diff --git a/.github/PULL_REQUEST_TEMPLATE.md b/.github/PULL_REQUEST_TEMPLATE.md new file mode 100644 index 0000000..ac3f36a --- /dev/null +++ b/.github/PULL_REQUEST_TEMPLATE.md @@ -0,0 +1,20 @@ + + +## What does this PR change? + + + +## Which guideline(s) are affected? + + + +## Checklist + +- [ ] H1 matches `title` in frontmatter +- [ ] All required body sections present and in order (`What & why`, `Scoring`, `Steps`, `References`, optionally `How Forter helps`) +- [ ] `forterApplies` and `How Forter helps` are in sync +- [ ] Every `#guideline-M-N` cross-reference points to an existing guideline +- [ ] References include at least one canonical source per claim diff --git a/.github/workflows/validate.yml b/.github/workflows/validate.yml new file mode 100644 index 0000000..3d09355 --- /dev/null +++ b/.github/workflows/validate.yml @@ -0,0 +1,24 @@ +name: validate + +on: + pull_request: + paths: + - "content/**" + - "audit/**" + - "scripts/**" + - "package.json" + - ".github/workflows/validate.yml" + push: + branches: [main] + +jobs: + validate: + runs-on: ubuntu-latest + steps: + - uses: actions/checkout@v4 + - uses: actions/setup-node@v4 + with: + node-version: "20" + cache: "npm" + - run: npm ci + - run: npm test diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..578e229 --- /dev/null +++ b/.gitignore @@ -0,0 +1,11 @@ +# OS / editor +.DS_Store +.vscode/ +.idea/ +*.swp + +# Just in case anyone runs tooling locally +node_modules/ + +# Audit reports — generated locally, never committed +report/ diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000..12d2718 --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,74 @@ +# Contributing + +Thanks for opening a PR. This repo is content-only - markdown files in `content/`, no build tooling. The rendered PDF and webinar live in a separate internal pipeline. + +## What to change + +- **Fix a fact, a link, or wording** in any existing guideline (`content/m*-*.md`). +- **Improve the steps or references** in a guideline. +- **Propose a new guideline** by opening an issue first - module numbering and scoping benefit from discussion before drafting. + +## Frontmatter schema + +Every guideline file starts with YAML frontmatter: + +```yaml +--- +id: m4-1-openapi-spec +module: actionable # discoverable | comprehensible | trustworthy | actionable | experiential +moduleNumber: 4 # 1-5, must match module +guidelineNumber: 1 # unique within module +title: Ship OpenAPI specification +complexity: 4 # 1 (trivial) to 5 (major engineering project); rendered as "Effort" in the guide body +impact: 5 # 1 (nice to have) to 5 (table stakes) +visualChange: low # optional: none | low | medium | high +forterApplies: partial # no | partial | yes | flagship +--- +``` + +Chapter and module-overview files have a lighter frontmatter: + +```yaml +--- +id: module-discoverable +title: Module 1 - Be Discoverable +kind: module-overview # front-matter | chapter | module-overview | appendix +moduleNumber: 1 # required for kind=module-overview +--- +``` + +## Required body sections + +Each guideline must have, in order: + +1. `# ` - H1 must match the frontmatter `title` exactly. +2. `## What & why` - what the guideline is and why it matters. +3. `## Scoring` - concrete, observable criteria for pass / partial / fail. This is what auditors (human or agent) will use. +4. `## Steps` - numbered, concrete actions to implement the guideline. +5. `## References` - links to specs, RFCs, blog posts, code examples. +6. `## How Forter helps` - **only** when `forterApplies` is `partial`, `yes`, or `flagship`. Skip this section when `forterApplies: no`. + +The internal validator (run in CI) enforces this structure and will fail the PR if a section is missing, mis-ordered, or if `forterApplies` doesn't match the presence of "How Forter helps". + +## Cross-references + +Link between guidelines with relative file paths, e.g. `[3.1](./m3-1-oauth-discovery.md)`. Audit rubrics link the same way (`[m3-2](./m3-2.md)`) and link back to content with `../content/...`. The validator (`npm test`) resolves every `(./mX-Y-*.md)` / `(../content|audit/...)` reference and fails the PR on a dangling link. + +## Checklist before opening a PR + +- [ ] H1 matches `title` in frontmatter. +- [ ] All required sections present and in order. +- [ ] `forterApplies` and `How Forter helps` are in sync (both present or both absent). +- [ ] Every `[x.y](./mX-Y-*.md)` cross-reference points to an existing file. +- [ ] Tags are lowercase, kebab-case. +- [ ] References include at least one canonical source per claim. + +## Tone + +- Concrete over abstract. "Add `Accept: application/ld+json`" beats "consider content negotiation". +- Cite RFCs and specs by number. Link to the canonical source, not a third-party tutorial. +- Don't sell Forter outside the "How Forter helps" section. The body of the guideline should be useful regardless of vendor choice. + +## License + +By submitting a PR you agree your contribution is licensed [CC BY 4.0](./LICENSE), the same as the rest of the repo. diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000..da6ab6c --- /dev/null +++ b/LICENSE @@ -0,0 +1,396 @@ +Attribution 4.0 International + +======================================================================= + +Creative Commons Corporation ("Creative Commons") is not a law firm and +does not provide legal services or legal advice. Distribution of +Creative Commons public licenses does not create a lawyer-client or +other relationship. Creative Commons makes its licenses and related +information available on an "as-is" basis. Creative Commons gives no +warranties regarding its licenses, any material licensed under their +terms and conditions, or any related information. Creative Commons +disclaims all liability for damages resulting from their use to the +fullest extent possible. + +Using Creative Commons Public Licenses + +Creative Commons public licenses provide a standard set of terms and +conditions that creators and other rights holders may use to share +original works of authorship and other material subject to copyright +and certain other rights specified in the public license below. The +following considerations are for informational purposes only, are not +exhaustive, and do not form part of our licenses. + + Considerations for licensors: Our public licenses are + intended for use by those authorized to give the public + permission to use material in ways otherwise restricted by + copyright and certain other rights. Our licenses are + irrevocable. Licensors should read and understand the terms + and conditions of the license they choose before applying it. + Licensors should also secure all rights necessary before + applying our licenses so that the public can reuse the + material as expected. Licensors should clearly mark any + material not subject to the license. This includes other CC- + licensed material, or material used under an exception or + limitation to copyright. More considerations for licensors: + wiki.creativecommons.org/Considerations_for_licensors + + Considerations for the public: By using one of our public + licenses, a licensor grants the public permission to use the + licensed material under specified terms and conditions. If + the licensor's permission is not necessary for any reason--for + example, because of any applicable exception or limitation to + copyright--then that use is not regulated by the license. Our + licenses grant only permissions under copyright and certain + other rights that a licensor has authority to grant. Use of + the licensed material may still be restricted for other + reasons, including because others have copyright or other + rights in the material. A licensor may make special requests, + such as asking that all changes be marked or described. + Although not required by our licenses, you are encouraged to + respect those requests where reasonable. More considerations + for the public: + wiki.creativecommons.org/Considerations_for_licensees + +======================================================================= + +Creative Commons Attribution 4.0 International Public License + +By exercising the Licensed Rights (defined below), You accept and agree +to be bound by the terms and conditions of this Creative Commons +Attribution 4.0 International Public License ("Public License"). To the +extent this Public License may be interpreted as a contract, You are +granted the Licensed Rights in consideration of Your acceptance of +these terms and conditions, and the Licensor grants You such rights in +consideration of benefits the Licensor receives from making the +Licensed Material available under these terms and conditions. + + +Section 1 -- Definitions. + + a. Adapted Material means material subject to Copyright and Similar + Rights that is derived from or based upon the Licensed Material + and in which the Licensed Material is translated, altered, + arranged, transformed, or otherwise modified in a manner requiring + permission under the Copyright and Similar Rights held by the + Licensor. For purposes of this Public License, where the Licensed + Material is a musical work, performance, or sound recording, + Adapted Material is always produced where the Licensed Material is + synched in timed relation with a moving image. + + b. Adapter's License means the license You apply to Your Copyright + and Similar Rights in Your contributions to Adapted Material in + accordance with the terms and conditions of this Public License. + + c. Copyright and Similar Rights means copyright and/or similar rights + closely related to copyright including, without limitation, + performance, broadcast, sound recording, and Sui Generis Database + Rights, without regard to how the rights are labeled or + categorized. For purposes of this Public License, the rights + specified in Section 2(b)(1)-(2) are not Copyright and Similar + Rights. + + d. Effective Technological Measures means those measures that, in the + absence of proper authority, may not be circumvented under laws + fulfilling obligations under Article 11 of the WIPO Copyright + Treaty adopted on December 20, 1996, and/or similar international + agreements. + + e. Exceptions and Limitations means fair use, fair dealing, and/or + any other exception or limitation to Copyright and Similar Rights + that applies to Your use of the Licensed Material. + + f. Licensed Material means the artistic or literary work, database, + or other material to which the Licensor applied this Public + License. + + g. Licensed Rights means the rights granted to You subject to the + terms and conditions of this Public License, which are limited to + all Copyright and Similar Rights that apply to Your use of the + Licensed Material and that the Licensor has authority to license. + + h. Licensor means the individual(s) or entity(ies) granting rights + under this Public License. + + i. Share means to provide material to the public by any means or + process that requires permission under the Licensed Rights, such + as reproduction, public display, public performance, distribution, + dissemination, communication, or importation, and to make material + available to the public including in ways that members of the + public may access the material from a place and at a time + individually chosen by them. + + j. Sui Generis Database Rights means rights other than copyright + resulting from Directive 96/9/EC of the European Parliament and of + the Council of 11 March 1996 on the legal protection of databases, + as amended and/or succeeded, as well as other essentially + equivalent rights anywhere in the world. + + k. You means the individual or entity exercising the Licensed Rights + under this Public License. Your has a corresponding meaning. + + +Section 2 -- Scope. + + a. License grant. + + 1. Subject to the terms and conditions of this Public License, + the Licensor hereby grants You a worldwide, royalty-free, + non-sublicensable, non-exclusive, irrevocable license to + exercise the Licensed Rights in the Licensed Material to: + + a. reproduce and Share the Licensed Material, in whole or + in part; and + + b. produce, reproduce, and Share Adapted Material. + + 2. Exceptions and Limitations. For the avoidance of doubt, where + Exceptions and Limitations apply to Your use, this Public + License does not apply, and You do not need to comply with + its terms and conditions. + + 3. Term. The term of this Public License is specified in Section + 6(a). + + 4. Media and formats; technical modifications allowed. The + Licensor authorizes You to exercise the Licensed Rights in + all media and formats whether now known or hereafter created, + and to make technical modifications necessary to do so. The + Licensor waives and/or agrees not to assert any right or + authority to forbid You from making technical modifications + necessary to exercise the Licensed Rights, including + technical modifications necessary to circumvent Effective + Technological Measures. For purposes of this Public License, + simply making modifications authorized by this Section 2(a) + (4) never produces Adapted Material. + + 5. Downstream recipients. + + a. Offer from the Licensor -- Licensed Material. Every + recipient of the Licensed Material automatically + receives an offer from the Licensor to exercise the + Licensed Rights under the terms and conditions of this + Public License. + + b. No downstream restrictions. You may not offer or impose + any additional or different terms or conditions on, or + apply any Effective Technological Measures to, the + Licensed Material if doing so restricts exercise of the + Licensed Rights by any recipient of the Licensed + Material. + + 6. No endorsement. Nothing in this Public License constitutes or + may be construed as permission to assert or imply that You + are, or that Your use of the Licensed Material is, connected + with, or sponsored, endorsed, or granted official status by, + the Licensor or others designated to receive attribution as + provided in Section 3(a)(1)(A)(i). + + b. Other rights. + + 1. Moral rights, such as the right of integrity, are not + licensed under this Public License, nor are publicity, + privacy, and/or other similar personality rights; however, to + the extent possible, the Licensor waives and/or agrees not to + assert any such rights held by the Licensor to the limited + extent necessary to allow You to exercise the Licensed + Rights, but not otherwise. + + 2. Patent and trademark rights are not licensed under this + Public License. + + 3. To the extent possible, the Licensor waives any right to + collect royalties from You for the exercise of the Licensed + Rights, whether directly or through a collecting society + under any voluntary or waivable statutory or compulsory + licensing scheme. In all other cases the Licensor expressly + reserves any right to collect such royalties. + + +Section 3 -- License Conditions. + +Your exercise of the Licensed Rights is expressly made subject to the +following conditions. + + a. Attribution. + + 1. If You Share the Licensed Material (including in modified + form), You must: + + a. retain the following if it is supplied by the Licensor + with the Licensed Material: + + i. identification of the creator(s) of the Licensed + Material and any others designated to receive + attribution, in any reasonable manner requested by + the Licensor (including by pseudonym if + designated); + + ii. a copyright notice; + + iii. a notice that refers to this Public License; + + iv. a notice that refers to the disclaimer of + warranties; + + v. a URI or hyperlink to the Licensed Material to the + extent reasonably practicable; + + b. indicate if You modified the Licensed Material and + retain an indication of any previous modifications; and + + c. indicate the Licensed Material is licensed under this + Public License, and include the text of, or the URI or + hyperlink to, this Public License. + + 2. You may satisfy the conditions in Section 3(a)(1) in any + reasonable manner based on the medium, means, and context in + which You Share the Licensed Material. For example, it may be + reasonable to satisfy the conditions by providing a URI or + hyperlink to a resource that includes the required + information. + + 3. If requested by the Licensor, You must remove any of the + information required by Section 3(a)(1)(A) to the extent + reasonably practicable. + + 4. If You Share Adapted Material You produce, the Adapter's + License You apply must not prevent recipients of the Adapted + Material from complying with this Public License. + + +Section 4 -- Sui Generis Database Rights. + +Where the Licensed Rights include Sui Generis Database Rights that +apply to Your use of the Licensed Material: + + a. for the avoidance of doubt, Section 2(a)(1) grants You the right + to extract, reuse, reproduce, and Share all or a substantial + portion of the contents of the database; + + b. if You include all or a substantial portion of the database + contents in a database in which You have Sui Generis Database + Rights, then the database in which You have Sui Generis Database + Rights (but not its individual contents) is Adapted Material; and + + c. You must comply with the conditions in Section 3(a) if You Share + all or a substantial portion of the contents of the database. + +For the avoidance of doubt, this Section 4 supplements and does not +replace Your obligations under this Public License where the Licensed +Rights include other Copyright and Similar Rights. + + +Section 5 -- Disclaimer of Warranties and Limitation of Liability. + + a. UNLESS OTHERWISE SEPARATELY UNDERTAKEN BY THE LICENSOR, TO THE + EXTENT POSSIBLE, THE LICENSOR OFFERS THE LICENSED MATERIAL AS-IS + AND AS-AVAILABLE, AND MAKES NO REPRESENTATIONS OR WARRANTIES OF + ANY KIND CONCERNING THE LICENSED MATERIAL, WHETHER EXPRESS, + IMPLIED, STATUTORY, OR OTHER. THIS INCLUDES, WITHOUT LIMITATION, + WARRANTIES OF TITLE, MERCHANTABILITY, FITNESS FOR A PARTICULAR + PURPOSE, NON-INFRINGEMENT, ABSENCE OF LATENT OR OTHER DEFECTS, + ACCURACY, OR THE PRESENCE OR ABSENCE OF ERRORS, WHETHER OR NOT + KNOWN OR DISCOVERABLE. WHERE DISCLAIMERS OF WARRANTIES ARE NOT + ALLOWED IN FULL OR IN PART, THIS DISCLAIMER MAY NOT APPLY TO YOU. + + b. TO THE EXTENT POSSIBLE, IN NO EVENT WILL THE LICENSOR BE LIABLE + TO YOU ON ANY LEGAL THEORY (INCLUDING, WITHOUT LIMITATION, + NEGLIGENCE) OR OTHERWISE FOR ANY DIRECT, SPECIAL, INDIRECT, + INCIDENTAL, CONSEQUENTIAL, PUNITIVE, EXEMPLARY, OR OTHER LOSSES, + COSTS, EXPENSES, OR DAMAGES ARISING OUT OF THIS PUBLIC LICENSE OR + USE OF THE LICENSED MATERIAL, EVEN IF THE LICENSOR HAS BEEN + ADVISED OF THE POSSIBILITY OF SUCH LOSSES, COSTS, EXPENSES, OR + DAMAGES. WHERE A LIMITATION OF LIABILITY IS NOT ALLOWED IN FULL OR + IN PART, THIS LIMITATION MAY NOT APPLY TO YOU. + + c. The disclaimer of warranties and limitation of liability provided + above shall be interpreted in a manner that, to the extent + possible, most closely approximates an absolute disclaimer and + waiver of all liability. + + +Section 6 -- Term and Termination. + + a. This Public License applies for the term of the Copyright and + Similar Rights licensed here. However, if You fail to comply with + this Public License, then Your rights under this Public License + terminate automatically. + + b. Where Your right to use the Licensed Material has terminated under + Section 6(a), it reinstates: + + 1. automatically as of the date the violation is cured, provided + it is cured within 30 days of Your discovery of the + violation; or + + 2. upon express reinstatement by the Licensor. + + For the avoidance of doubt, this Section 6(b) does not affect any + right the Licensor may have to seek remedies for Your violations + of this Public License. + + c. For the avoidance of doubt, the Licensor may also offer the + Licensed Material under separate terms or conditions or stop + distributing the Licensed Material at any time; however, doing so + will not terminate this Public License. + + d. Sections 1, 5, 6, 7, and 8 survive termination of this Public + License. + + +Section 7 -- Other Terms and Conditions. + + a. The Licensor shall not be bound by any additional or different + terms or conditions communicated by You unless expressly agreed. + + b. Any arrangements, understandings, or agreements regarding the + Licensed Material not stated herein are separate from and + independent of the terms and conditions of this Public License. + + +Section 8 -- Interpretation. + + a. For the avoidance of doubt, this Public License does not, and + shall not be interpreted to, reduce, limit, restrict, or impose + conditions on any use of the Licensed Material that could lawfully + be made without permission under this Public License. + + b. To the extent possible, if any provision of this Public License is + deemed unenforceable, it shall be automatically reformed to the + minimum extent necessary to make it enforceable. If the provision + cannot be reformed, it shall be severed from this Public License + without affecting the enforceability of the remaining terms and + conditions. + + c. No term or condition of this Public License will be waived and no + failure to comply consented to unless expressly agreed to by the + Licensor. + + d. Nothing in this Public License constitutes or may be interpreted + as a limitation upon, or waiver of, any privileges and immunities + that apply to the Licensor or You, including from the legal + processes of any jurisdiction or authority. + + +======================================================================= + +Creative Commons is not a party to its public +licenses. Notwithstanding, Creative Commons may elect to apply one of +its public licenses to material it publishes and in those instances +will be considered the “Licensor.” The text of the Creative Commons +public licenses is dedicated to the public domain under the CC0 Public +Domain Dedication. Except for the limited purpose of indicating that +material is shared under a Creative Commons public license or as +otherwise permitted by the Creative Commons policies published at +creativecommons.org/policies, Creative Commons does not authorize the +use of the trademark "Creative Commons" or any other trademark or logo +of Creative Commons without its prior written consent including, +without limitation, in connection with any unauthorized modifications +to any of its public licenses or any other arrangements, +understandings, or agreements concerning use of licensed material. For +the avoidance of doubt, this paragraph does not form part of the +public licenses. + +Creative Commons may be contacted at creativecommons.org. + diff --git a/README.md b/README.md index ce1c3bb..47de541 100644 --- a/README.md +++ b/README.md @@ -1,2 +1,137 @@ -# agentic-readiness-guide -Agentic Readiness Guide +# Forter Agentic Readiness Guide + +A practical, opinionated guide to making any website **agent-ready** - discoverable, comprehensible, trustworthy, actionable, and experiential - so that LLM-driven agents (and the humans behind them) can find, understand, trust, and act on your product. + +Five modules, 25 guidelines, each one a single markdown file in [`content/`](./content). Every guideline has a "What & why", a 1-5 complexity/impact score, concrete steps, and references. Where Forter ships infrastructure that satisfies a guideline, the file ends with a "How Forter helps" callout. + +## Download it offline + +https://github.com/user-attachments/assets/9817a638-6ffc-42b3-94b4-a0124f280cea + +**[Download the full guide (PDF)](https://l.forter.com/hubfs/Forter-agentic-readiness-guide.pdf)** + +## Read it online + +Browse [`content/`](./content) directly on GitHub. Files are organized by module: `m1-*` Discoverable, `m2-*` Comprehensible, `m3-*` Trustworthy, `m4-*` Actionable, `m5-*` Experiential. + +## Audit your own site + +This repo ships a Claude Code skill in [`SKILL.md`](./SKILL.md), backed by 25 machine-testable rubrics in [`audit/`](./audit), that turns the guide into an automated auditor. Point it at a site (and optionally its source) and it will: + +1. Run the probe in each `audit/m{M}-{N}.md` rubric against your site. +2. Score each guideline **Pass / Partial / Fail / N/A**, citing the literal probe response as evidence. +3. Rank Fails and Partials by **impact × (6 - complexity)** so the highest-leverage fixes float to the top. +4. Optionally apply fixes as commits when you give it your repo path. + +### Install the skill + +[Claude Code](https://docs.claude.com/claude-code) auto-discovers skills under `~/.claude/skills/`. Clone once and **symlink** the repo in - no copying, so `git pull` keeps the skill current and your skills folder stays clean: + +```bash +git clone https://github.com/forter/agentic-readiness-guide.git +mkdir -p ~/.claude/skills +ln -s "$(pwd)/agentic-readiness-guide" ~/.claude/skills/forter-agentic-readiness-audit +``` + +(The symlink's name matches the skill's `name`. The repo's root `SKILL.md` and `audit/` resolve straight through the symlink. To uninstall, `rm ~/.claude/skills/forter-agentic-readiness-audit` - that removes only the link, not your clone.) + +Then in any Claude Code session: + +```bash +claude "Audit https://your-site.example.com against the Agentic Readiness Guide" +``` + +Claude picks the skill up from its frontmatter `description` and runs it. Add `--add-dir /path/to/your/site` to include your source repo - fixes get applied as commits there. Type `/skills` inside Claude Code to confirm the skill is loaded. + +**Prereqs.** The probes shell out to `curl`, `jq`, and `python3` (for HTML parsing). All three are standard on macOS and most Linux distros; on a bare container, install via `apt-get install curl jq python3` or `brew install jq` (curl and python3 usually ship). + +### What you get + +A single `report/AUDIT.md` with, in order: + +- **One-line scoreboard** - `Score N/W (P%) · X Pass · Y Partial · Z Fail · K N/A`. +- **Headline** - two sentences a non-technical reader can grasp: where the site sits and the rough shape of the gap. +- **Action plan** - ordered fixes (smallest concrete change → guideline it unblocks → effort → point gain), with a projected post-fix score. +- **Cross-cutting blockers** - when one bug gates ≥ 3 guidelines, it gets a dedicated reproduce-and-fix subsection instead of being repeated per row. +- **Findings table** - 25 rows, one per guideline, with the literal probe response as inline evidence. + +Raw probe outputs land alongside in `report/*.out`; opt into `report/score.json` (machine-readable, schema below) and a PR-ready issue list by asking for them. A typical first audit on a real e-commerce site closes the discoverability tier in a day and surfaces 15-20 deeper items the team can sequence over a sprint. + +### `report/score.json` (machine-readable, opt-in) + +Ask for it explicitly ("also write `score.json`") and the skill emits a single JSON file next to `AUDIT.md` - the same scores as the report, structured for CI/CD gates, dashboards, or trend tracking. The findings table is for humans; `score.json` is for machines. + +```jsonc +{ + "host": "example.com", // bare host audited (no scheme) + "tested_at": "2026-06-02T14:30:00Z", // ISO-8601 UTC timestamp of the run + "scope": ["m1-1", "m1-2", "..."], // guideline ids actually scored this run (default: all 25) + "scoreboard": { + "pass": 4, "partial": 8, "fail": 11, "na": 2, // guideline counts by status + "blocked": 0, // guidelines gated by a cross-cutting blocker (↺) + "weighted": { "points": 31, "total": 142, "pct": 22 } // summed sub-check weights; pct = points/total + }, + "blockers": [ // cross-cutting issues gating ≥3 guidelines (may be empty) + { "id": "A", "title": "WAF challenges agent fetchers", "gates": ["m1-1", "m1-3", "m2-2"] } + ], + "guidelines": [ + { + "id": "m1-1", // matches content/m1-1-*.md and audit/m1-1.md + "title": "Discovery files", // the audit rubric's short title + "status": "partial", // "pass" | "partial" | "fail" | "na" | "blocked" + "points": 2, // sub-check weights earned + "weight_total": 10, // sum of the rubric's sub-check weights (matches audit frontmatter) + "complexity": 1, // 1-5, from the rubric/content frontmatter + "impact": 4, // 1-5, from the rubric/content frontmatter + "priority": 20, // impact × (6 - complexity); higher = fix first + "visual_change": "none", // "none" | "low" | "medium" | "high" + "sub_checks": [ // one entry per rubric row, in order + { "name": "sitemap.xml exists", "pass": true, "weight": 1, "evidence": "200 application/xml" }, + { "name": "Content-Signal present", "pass": false, "weight": 1, "evidence": "no Content-Signal: line" } + // weight: 0 rows are manual/bonus checks - surfaced but never drag the automated score + ], + "fix_summary": "Add Content-Signal directive; create llms.txt and index.md; emit Link: headers." + } + // ... one object per guideline in scope + ] +} +``` + +**Field notes.** `weighted.total` counts only guidelines that were scored (N/A and blocked guidelines are excluded), so `pct` reflects what was actually testable. `status` maps from `points / weight_total` per the rubric's thresholds (default **Pass ≥ 85%**, **Partial ≥ 30%**, **Fail < 30%** - some rubrics override this; see [`audit/README.md`](./audit/README.md)). A sub-check with `"weight": 0` is a `(manual)` or bonus row: it appears for visibility but can never lower the automated score. `priority` is the same `impact × (6 - complexity)` ranking the action plan uses. + +## Cite it + +Licensed CC BY 4.0 - use, adapt, and quote freely with attribution to Forter. + +## Contribute + +Spotted an outdated reference? Missing protocol? Better wording for a guideline? PRs welcome. + +1. Edit the relevant markdown file in `content/`. +2. Keep the frontmatter and H1 in sync (see [`CONTRIBUTING.md`](./CONTRIBUTING.md) for the schema). +3. Open a PR against `main`. + +The rendered PDF and webinar are produced from a separate build pipeline. CI (`npm test`) validates frontmatter, audit↔content alignment, and required section structure on every PR - so you'll find out before merge if something is off. + +## Structure + +``` +content/ Human-facing guide (markdown, the editorial source of truth) + 00-toc.md Table of contents + 01-introduction.md Lifecycle frame: Discover → Comprehend → Trust → Act → Experience + m{1-5}-0-module-*.md Five module overviews + m{1-5}-{1-N}-*.md Twenty-five guideline files + +audit/ Machine-testable rubrics - one per guideline, used by SKILL.md + m{1-5}-{1-N}.md Probe + weighted sub-checks + codebase hints + README.md Rubric file format & strict scoring rules + +SKILL.md Claude Code skill - orchestrates probes and writes report/AUDIT.md +CONTRIBUTING.md Frontmatter schema, body-section checklist, PR template +scripts/validate.mjs Self-contained frontmatter + alignment validator (npm test) +LICENSE CC BY 4.0 +``` + +## License + +Content is licensed [CC BY 4.0](./LICENSE). Built and maintained by [Forter](https://www.forter.com). diff --git a/SKILL.md b/SKILL.md new file mode 100644 index 0000000..d35ea63 --- /dev/null +++ b/SKILL.md @@ -0,0 +1,244 @@ +--- +name: forter-agentic-readiness-audit +description: Audit a website against the Forter Agentic Readiness Guide. Loads the 25 weighted rubrics in `audit/`, probes the target site (and optional source code), scores each guideline Pass/Partial/Fail/N/A with sub-check granularity, and produces a prioritized fix report. Use when a user asks "score my site against the agentic readiness guide", "audit https://… for agent readiness", or "what do I need to fix to be agent-ready". +--- + +# Forter Agentic Readiness Audit + +You score a website against the 25 guidelines in this repo. Each guideline has a machine-testable rubric in `audit/m{M}-{N}.md` (probe + weighted sub-checks + codebase hints). Your job: run the probes, score, prioritize, report. + +## Prerequisites + +The probes shell out to `curl`, `jq`, and `python3`. If any is missing, surface the error to the user with the install command for their platform (`brew install jq`, `apt-get install jq python3`, etc.) and stop - don't continue with degraded probes. + +## Inputs + +Ask the user for these if not provided: + +- **URL** - `https://example.com`. Required. +- **Local source path** (optional but strongly recommended) - enables framework detection and per-file fix hints. +- **Scope** (optional) - `all`, `top N`, or a comma-separated list of guideline IDs (`m1-1,m4-1,m4-4`). Default: all 25. +- **Output format(s)** - markdown report (always), plus opt-in `score.json` and PR-ready issue list. + +If only a URL is given, run with codebase hints set to generic-only. + +## How the skill is wired + +- **`content/`** is the human-facing guide. Don't read it during scoring - it's prose. Surface its URLs in references and fixes only. +- **`audit/m{M}-{N}.md`** is the source of truth for probes and scoring. Each file has frontmatter (`complexity`, `impact`, `weight_total`) and three sections: `## Probe`, `## Rubric`, `## Codebase hints`. +- **`audit/README.md`** documents the rubric format. Read it once if you're unfamiliar. + +## Process + +### 1. Set up + +Resolve and export shell variables once: + +```bash +URL='<user URL>' +HOST=$(printf '%s' "$URL" | sed -E 's|^https?://([^/]+).*|\1|') +export HOST ORIGIN="https://$HOST" +mkdir -p ./report && cd ./report +``` + +If a source path was given, detect the framework once and cache the result: + +```bash +REPO='<user path>' +# Detect: presence of files → framework label +# package.json + "next" → next.js (app router if app/ exists, else pages router) +# package.json + "express"|"fastify"|"@nestjs" → node-server +# Gemfile → rails +# requirements.txt|pyproject.toml + django|flask|fastapi → python-<framework> +# composer.json → php-<laravel|symfony|wordpress|custom> +# *.php in webroot, no composer.json → php-classic +# astro.config.* → astro · hugo.toml → hugo · config.yml + _posts → jekyll +echo "$FRAMEWORK" > ./report/framework +``` + +### 1.5 Pre-flight - can an agent even reach the site? + +Before scoring anything, run one cheap reachability probe. If the origin blocks or challenges agent fetchers, _every_ downstream guideline is moot - an agent bounces before it reads a byte. This mirrors how a real agent (ChatGPT-User, Claude-User, PerplexityBot) experiences the site. + +```bash +BASE=$(curl -fsS -A 'Mozilla/5.0' -o /dev/null -w '%{http_code}' "$ORIGIN/") +echo "baseline(browser) $BASE" +for UA in 'ChatGPT-User/1.0' 'Claude-User/1.0' 'PerplexityBot/1.0' 'OAI-SearchBot/1.0'; do + read code size < <(curl -fsS -A "$UA" -o /tmp/pf.html -w '%{http_code} %{size_download}' "$ORIGIN/" 2>/dev/null || echo "000 0") + grep -iqE 'just a moment|cf-browser-verification|captcha|enable javascript to continue' /tmp/pf.html && chal=" CHALLENGE" || chal="" + printf ' %-20s %s bytes=%s%s\n' "$UA" "$code" "$size" "$chal" +done +``` + +**If agent fetchers are blocked or challenged** (403/429/503, a Cloudflare/captcha interstitial, or a byte size wildly below baseline), treat it as **cross-cutting Blocker A** in the report. It gates m1-1, m1-3, m2-*, and every API/MCP/commerce guideline that needs the agent to fetch a real response. Score those as `↺` (blocked), and make "allowlist agent fetchers in your WAF / Cloudflare AI Crawl Control" action 1 in the plan. Don't let a high score on file-presence checks mask the fact that no agent can get through. Distinguish this from a site that *intentionally* blocks *training\* crawlers (GPTBot/CCBot) while staying open to fetchers - that's fine (see m1-1 sub-check 4). + +### 2. Run probes in parallel + +For each guideline in scope, copy the `## Probe` block from `audit/m{M}-{N}.md`, substitute `$ORIGIN`/`$HOST`, and execute. Probes are independent - run them in parallel using background shells or multiple Bash tool calls in a single message. + +Keep raw probe output in `./report/m{M}-{N}.out` so you can re-score without re-fetching. + +**Emit progress text.** The user cannot read tool output in real time the same way you can - emit a one-sentence text message before each probe batch so they see what's happening. Suggested cadence: one line on start ("Resolving target…"), one line per probe batch ("Probing m1-_ discovery files…", "Probing m4-_ actionable / commerce…"), one line on scoring ("Scoring 25 guidelines…"), one line on write ("Writing report/AUDIT.md…"). Don't narrate every curl - group by module or by batch. Keep each line under 80 chars. + +Probes that POST to live endpoints are safe - they're discovery calls with no side effects: m4-4 `initialize` + `tools/list`, m4-9 `OPTIONS` (and a `{}` POST that only reads back a structured validation error), m4-10 `/ask`, and m5-4's DCR probe (POSTs an intentionally-incomplete body so the server rejects it with a validation error rather than registering a real client). Don't run probes that explicitly require manual confirmation (m5-1, m5-4 sub-check 5); flag them as `(manual)` and surface in the report. + +### 3. Score each guideline + +For each guideline: + +1. Apply the `## Rubric` table from its `audit/` file. Walk row-by-row, derive each sub-check's pass condition from the probe output, sum weights. +2. Map `points / weight_total` to status: + - **Pass** ≥ 85% + - **Partial** 30-84% + - **Fail** < 30% + - **N/A** if the rubric's `N/A condition` matches the site (e.g., commerce protocols on a static blog). + + Each `audit/` file may override these thresholds - check the "Status:" line under the rubric table. + +3. Record the **exact probe** run and **exact response** observed - that's your evidence. + +**Evidence rules (non-negotiable)** + +These are the rules that separate a useful audit from a flattering one. Violating them produces over-claims that the user has to refute. + +- **Live-web sub-checks score from HTTP responses only.** If the rubric asks whether `/llms.txt` exists, the answer comes from `curl https://site/llms.txt`. It does not come from `find $REPO -name llms.txt`. A file existing in a repo does not mean a URL serves it; an endpoint file being present in `public/` does not mean nginx routes to it; a PHP doc-comment describing a webhook shape does not mean the live endpoint accepts that shape. +- **Repo inspection is for _fix hints only_.** When you've decided a sub-check Fails based on live evidence, you may consult the repo to suggest which file to edit. You may NOT consult the repo to upgrade a Fail to a Pass. +- **When a probe can't complete, the sub-check Fails.** This includes: auth-gated endpoints you don't have credentials for, manual checks (platform listings, Wikipedia presence) that require a human, JS-rendered content you'd need a headless browser to read, third-party services that timed out. Call out the specific reason. The user can re-run with credentials or confirm manually. +- **Ambiguous responses get the conservative reading.** A 400 with a domain-specific error message (e.g., `{"error":"Expected event.type = order.created"}`) is evidence the endpoint understands a specific protocol - that's a Pass on "endpoint exists and validates schema." It is NOT evidence of "the protocol is fully implemented" - that requires sending a valid payload and getting a domain-correct response. Score each sub-check at the resolution it asks for. +- **Cite the probe and the response verbatim, not your paraphrase.** "Got 200" is not evidence. "Got 200 with `content-type: application/json` and `.endpoints.checkout` field present" is. + +### 4. Prioritize fixes + +Rank Fail + Partial guidelines by: + +``` +priority = impact × (6 - complexity) +``` + +Both values come from the rubric frontmatter. Higher = fix first. Tie-break by `visualChange` (`none` and `low` before `medium`/`high`) - these ship faster. + +Show the cumulative impact too: "After top 5 fixes, projected score: X / Y (was A / Y)." + +### 4.5 Turn every gap into a concrete, environment-aware fix + +A score is half the value; the **fix** is the other half. Every Fail and Partial must ship with a fix the user can act on _in their stack_ - not generic advice. For each one: + +1. Start from the rubric's `## Codebase hints` and `## Auto-fix template`. +2. **Specialize to the detected framework** (from `./report/framework`). Surface only the matching hint row, with the real file path - `app/robots.ts` for Next.js app-router, `config/routes.rb` for Rails, webroot drop for PHP-classic, etc. +3. **If a repo path was given**, ground it in the actual tree: name the exact file to create/edit (`grep`/`ls` to confirm where headers/middleware/routes already live), and reference any half-built feature you found (e.g. "`agentic-oauth.php` exists in the repo but nothing routes `/.well-known/oauth-authorization-server` to it"). Repo inspection is for _fix hints only_ - it never upgrades a live Fail to a Pass. +4. Make the change the **smallest** one that moves the sub-check from Fail to Pass, and quote the literal snippet (robots line, JSON-LD block, header, well-known file) inlined from the rubric's auto-fix template with `$ORIGIN`/brand substituted. + +**Detect & propose only.** Put these fixes in the report (Action plan rows, Blocker `Fix` blocks, and the opt-in PR-issue list). **Do not write any files** - applying changes happens only in step 6, only when the user explicitly says "apply", and only against the repo path. + +### 5. Report + +Write a single markdown file at `./report/AUDIT.md`. The format is DRY - every fact appears in exactly one place. The user reads top-down and stops as deep as they need to go: header → headline → action plan → blockers → findings table. + +``` +# Agentic Readiness Audit - <host> + +**Score N / W (P%)** · X Pass · Y Partial · Z Fail · K N/A · <framework> · YYYY-MM-DD + +> **Headline.** <2-3 sentences, executive-summary style. State where the site sits overall ("foundational stage", "production-ready on discovery but gapped on actionability", etc.), the shape of the gap, and the rough cost/upside of closing it. Do NOT name specific endpoints, file paths, server bugs, or RFC numbers in the headline - those live in the Blockers and Findings sections. A non-technical reader (PM, founder, exec) should be able to grasp it in one read.> + +## Action plan - do in order + +| # | Action | Unblocks | Effort | +pts | +|---|--------|----------|--------|------| +| 1 | <smallest concrete change> | <guideline IDs it unblocks> | <h/d> | +N | +| ... | + +**Projected: <current> → <after-action-plan> (P%) after rows 1-N.** + +## Cross-cutting blockers + +For each blocker (typically 1-3 per audit) that gates ≥ 3 guidelines, write a dedicated subsection titled `### Blocker A - <one-line description>`. Inside: a `Reproduce` block with the literal probe + response, then a `Fix` block with the concrete change. Then in the Findings table use the `↺` status and reference "Blocker A" in the Unlocks column. Mention each specific bug exactly once, here - never repeat it across the guideline rows. + +## Findings - 25 guidelines + +Legend: ✅ Pass · ⚠️ Partial · ❌ Fail · ➖ N/A · ↺ blocked by a cross-cutting blocker. + +| ID | Guideline | Score | Live evidence (verbatim probe results) | Unlocks via | +|----|-----------|------:|-----------------------------------------|-------------| +| m1-1 | Discovery files | ⚠️ 2/6 | `sitemap.xml 404`; `robots.txt 200, has Content-Signal:`, no `Sitemap:` line; `llms.txt 404`; `index.md 404`; `Link: (none)` | action 3, 4 | +| m1-2 | Well-known agent files | ↺ 0/5 | All `/.well-known/*.json → 301 /well_known/ → 404` | Blocker A + action 2 | +| ... | + +## Quick wins outside the action plan + +3-5 bullets: each < 1 hour, gains a point, doesn't gate anything. + +## Methodology + +One paragraph: probes defined in audit/, evidence-only-from-HTTP, status thresholds. Link to `report/score.json` for the machine-readable breakdown. +``` + +**Rules for each section:** + +- **Headline** is the most-skipped-by-engineers, most-read-by-execs section. Lead with the _story_ (where does this site stand), not the _findings_ (what did probes return). Two sentences, max three. Don't name endpoints or files. +- **Action plan** is ordered by "what depends on what," not just by priority score. If action 2 needs action 1's fix, action 1 comes first even if it has lower priority. +- **Blockers** absorb the cross-cutting story (one nginx bug, one missing template, one missing auth setup). If you find yourself writing "same bug as m1-2" in another row, that's the cue to lift it into a Blocker. +- **Findings table** is one row per guideline. Evidence column carries the literal probe results inline - backtick-quote the response strings. No nested tables, no per-guideline detail sections. If a finding needs more than 200 characters to explain, it belongs in a Blocker, not the table. +- **Probe/Response/Verdict triplets** that you produced during scoring are kept in the raw `./report/*.out` files and `score.json` - they don't appear in the report markdown unless the reader explicitly asks for them. The Findings table's evidence column is the user-facing distillation. +- **Don't repeat the totals.** The score appears once at the top. Don't add a per-module breakdown - it's redundant with the Findings table. +- **Don't write "what changed since last run."** That's commit-message territory. + +**Score.json (opt-in)** - written to `report/score.json`. Schema documented in [`README.md`](./README.md#reportscorejson-machine-readable-opt-in); keep the two in sync. `weight_total` per guideline MUST match that guideline's `audit/` frontmatter. A `weight: 0` sub-check is a `(manual)`/bonus row that never lowers the automated score. + +```json +{ + "host": "example.com", + "tested_at": "YYYY-MM-DDTHH:MM:SSZ", + "scope": ["m1-1", "m1-2", "..."], + "scoreboard": { "pass": 4, "partial": 8, "fail": 11, "na": 2, "blocked": 0, "weighted": { "points": 31, "total": 142, "pct": 22 } }, + "blockers": [ + { "id": "A", "title": "WAF challenges agent fetchers", "gates": ["m1-1", "m1-3", "m2-2"] } + ], + "guidelines": [ + { + "id": "m1-1", "title": "Discovery files", + "status": "partial", "points": 2, "weight_total": 10, + "complexity": 1, "impact": 4, "priority": 20, "visual_change": "none", + "sub_checks": [ + { "name": "sitemap.xml exists", "pass": true, "weight": 1, "evidence": "200 application/xml" }, + { "name": "robots.txt exists", "pass": true, "weight": 1, "evidence": "200, Sitemap: line found" }, + { "name": "Content-Signal in robots", "pass": false, "weight": 1, "evidence": "no Content-Signal: line" }, + ... + ], + "fix_summary": "Add Content-Signal directive; create llms.txt and index.md; emit Link: headers." + } + ] +} +``` + +**PR-ready issue list (opt-in)**: + +One markdown block per Fail/Partial guideline, ready to paste as a GitHub issue body. Includes title (`agentic-readiness: <guideline title> (m{M}-{N})`), evidence, fix steps, file paths. + +### 6. Apply fixes (only when asked) + +If the user says "apply the top N" and gave you the repo path: + +- One commit per guideline. Message: `agentic-readiness: <guideline title> (m{M}-{N})`. +- Make the smallest change that pushes the guideline to Pass. Don't expand scope. +- After each commit, re-run only that guideline's probe to confirm. +- Stop at the user's quota. + +Never push, never open PRs without explicit consent. + +## Conventions + +- **Never fabricate scores.** If a probe doesn't complete (network error, JSON parse failure, JS-only render, auth required), the sub-check **Fails**. The Probe/Response/Verdict triplet captures the reason - `Verdict: FAIL - auth required, no token supplied`. The user can grant credentials and re-run. +- **Each guideline is independent.** Score from its own rubric, not your overall impression of the site. +- **Ignore `How Forter helps` sections** in `content/`. They're vendor commentary, not requirements. The audit must give the same score whether or not Forter is in use. +- **Don't recommend Forter** unless the user explicitly asks "should we use Forter for this?". +- **Keep responses verbatim.** Copy the response bytes (status line + relevant headers + first ~200 chars of body). Don't summarize "got an ACP-shaped error" - show `{"error":"Expected event.type = order.created"}`. +- **Repo grep is for fix hints, not evidence.** If a sub-check failed live but the repo shows the feature is half-built, note it under **Fix** (e.g., "agentic-oauth.php exists in repo but no nginx route") - don't promote the Fail to a Pass. +- **Keep the report tight.** The Probe/Response/Verdict triplet is the only ceremony. No essays. + +## When the user says "go" + +1. Ask for URL and (recommended) repo path if not given. +2. Ask which output formats - markdown only, or also `score.json` and/or PR issue list. +3. Run probes in parallel, score, prioritize, report. +4. If they then say "apply the top three", apply them as commits, then re-probe those three to confirm. diff --git a/audit/README.md b/audit/README.md new file mode 100644 index 0000000..ef64534 --- /dev/null +++ b/audit/README.md @@ -0,0 +1,111 @@ +# Audit rubrics + +Machine-testable companions to `content/`. One file per guideline (`m1-1.md` … `m5-4.md`). The skill in `SKILL.md` loads these to probe a target site and assign weighted scores. + +`content/` is the human-facing guide. `audit/` is the tester's rubric. Don't merge them - content evolves on its own cadence, rubrics evolve as probes improve. + +## Consistency contract + +Three layers must agree on module + guideline numbering: `content/m{M}-{N}-*.md` ↔ `audit/m{M}-{N}.md` ↔ the report's findings table (`m{M}-{N}` row id). The CI validator (`npm test`) enforces: + +- `audit/m{M}-{N}.md` exists for every content guideline (warning otherwise). +- `audit.id` matches `content.id`'s M-N prefix. +- `audit.complexity` and `audit.impact` equal the values in `content/`. +- `audit.visualChange` matches `content.visualChange` when both are set. +- `audit.weight_total` equals the sum of the rubric table's row weights (the final integer cell of each data row, across every table in the `## Rubric` section). Add a sub-check → bump `weight_total` to match, or CI fails. + +Titles intentionally diverge: `content/` uses imperative verbs ("Implement OAuth", "Verify bots cryptographically") because the guide reads top-to-bottom; `audit/` uses short noun-phrase tags ("OAuth discovery", "Web Bot Auth") that fit a scoreboard column. The report uses the audit title. + +## File format + +Every `audit/m{M}-{N}.md` must follow this structure: + +```markdown +--- +id: m{M}-{N} # MUST match the content guideline's M-N +title: <short title, ~3-5 words> # may differ from content's imperative title +complexity: <1-5, MUST match content frontmatter> +impact: <1-5, MUST match content frontmatter> +visualChange: <none|low|medium|high, MUST match content if both set> +weight_total: <sum of sub-check weights, usually 3-8> +--- + +# m{M}-{N} - <title> + +## Probe + +Exact shell commands that gather evidence. Use `$HOST` as the bare host (no scheme), `$ORIGIN` as `https://$HOST`. Probes should be: +- Idempotent (safe to re-run). +- Fast (<5s each; offload heavy work to optional sub-checks). +- Non-destructive (no POST to live endpoints unless explicitly a sandbox). +- Self-contained (no shared state between probes). + +​```bash +curl -fsSI $ORIGIN/sitemap.xml +curl -fsS $ORIGIN/robots.txt | grep -iE '^(sitemap|content-signal):' +# ... etc +​``` + +## Rubric + +A table of weighted sub-checks. Each row: a name, the pass condition (observable, derivable from probe output), and a point weight. + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | sitemap.xml | HTTP 200 and `content-type` includes `xml` | 1 | +| 2 | robots.txt | HTTP 200 and references at least one `Sitemap:` line | 1 | +| ... | + +**Status mapping** (default; override only with explicit reason): +- **Pass** - score ≥ 85% of `weight_total`. +- **Partial** - score ≥ 30% of `weight_total`. +- **Fail** - score < 30% of `weight_total`. +- **N/A** - the guideline genuinely does not apply (e.g., commerce protocols on a static marketing site). + +**Strict scoring rules** (binding on every rubric): +- Every sub-check's `Pass when` clause MUST be derivable from HTTP response bytes alone. "File present in repo at path X" is never a valid Pass condition; only "URL X returns Y" is. +- A sub-check that can't be probed (auth required, manual platform check, JS-only render) **Fails**. Don't introduce `Unknown` - the orchestrator surfaces such cases with the exact probe so a human can re-run. +- Ambiguous responses get the conservative reading. A 400 error message that names a protocol's event type is *only* evidence the endpoint validates schema, not evidence the protocol is fully implemented. +- Repo inspection is reserved for `## Codebase hints` - it MAY suggest which file to edit when a sub-check Fails. It MAY NOT promote a Fail to a Pass. + +## Codebase hints + +A short bulleted list mapping common frameworks → file paths to edit. The orchestrator detects framework from the repo (`package.json`, `Gemfile`, `composer.json`, `requirements.txt`, `next.config.js`, `astro.config.mjs`, etc.) and surfaces only the relevant row. + +- **Next.js (app router)**: `app/robots.ts`, `app/sitemap.ts` +- **Next.js (pages router)**: `pages/api/robots.ts`, `public/sitemap.xml` +- **Rails**: `config/routes.rb` + `app/views/robots.text.erb` +- **Django**: `django.contrib.sitemaps`, `urls.py` +- **Express / Node**: middleware route, set headers via `res.set()` +- **PHP / classic**: drop file at webroot (`public/`, `htdocs/`, `www/`) +- **Cloudflare Pages / Workers**: `_headers`, `public/` +- **Static (Hugo/Jekyll/Astro/11ty)**: `static/` or `public/` directory +- **Other / unknown**: drop file at webroot + +If a guideline is framework-agnostic (just a JSON file at a well-known path), say so and skip the per-framework table. + +## Auto-fix template (optional) + +A copy-pasteable starter for the most common stack. Keep minimal and correct over comprehensive. Include only when there's an obvious one-file change that covers ≥80% of sites. + +​```text +# /robots.txt +Sitemap: $ORIGIN/sitemap.xml + +User-agent: * +Content-Signal: search=yes, ai-input=yes, ai-train=no +​``` + +## References + +One line: link back to the source-of-truth guideline in `content/`. Specs/RFCs already live there - don't duplicate. + +`See: [content/m{M}-{N}-*.md](../content/m{M}-{N}-*.md)` + +## Notes for contributors + +- Sub-checks should be **independently testable** - a reader should be able to see each row's pass/fail from the probe output alone. +- Weights should be roughly proportional to user-visible impact within the guideline (don't weight a `Link:` header equal to publishing `llms.txt` itself). +- Composite guidelines (m1-1, m4-9, m2-2) should split into sub-checks per discrete artifact (file, protocol, endpoint). +- When a sub-check requires a paid/destructive action (e.g., actually completing an OAuth flow), mark it `weight: 0` and tag it `(manual)` - the orchestrator surfaces it as a checklist item rather than a probe failure. +- Keep each file under ~100 lines. If you're writing more, the rubric is too coarse - split into sub-checks. diff --git a/audit/m1-1.md b/audit/m1-1.md new file mode 100644 index 0000000..bc85243 --- /dev/null +++ b/audit/m1-1.md @@ -0,0 +1,171 @@ +--- +id: m1-1 +title: Discovery files +complexity: 1 +impact: 4 +visualChange: none +weight_total: 10 +--- + +# m1-1 - Discovery files + +## Probe + +```bash +# 1. sitemap +curl -fsSI $ORIGIN/sitemap.xml -o /dev/null -w '%{http_code} %{content_type}\n' + +# 2. robots.txt + content +curl -fsS $ORIGIN/robots.txt | tee /tmp/robots.txt +grep -iE '^sitemap:' /tmp/robots.txt +grep -iE '^content-signal:' /tmp/robots.txt + +# 2b. robots AI-policy quality: are agent *fetchers* (user-triggered) blocked, and +# is the Content-Signal grammar valid? Distinguish them from training crawlers. +python3 - <<'PY' +import re +try: + body=open("/tmp/robots.txt").read() +except FileNotFoundError: + print("robots_parse no-file"); raise SystemExit +# Group rules by user-agent +groups={}; cur=[] +for raw in body.splitlines(): + line=raw.split('#',1)[0].strip() + if not line: continue + k,_,v=line.partition(':'); k=k.strip().lower(); v=v.strip() + if k=='user-agent': + cur=groups.setdefault(v.lower(), []) + elif k=='disallow' and cur is not None: + cur.append(v) +def blocked(ua): + rules=groups.get(ua.lower()) + return rules is not None and any(r=='/' for r in rules) +# user-triggered agent fetchers - these MUST stay reachable for agent-readiness +FETCHERS=["ChatGPT-User","OAI-SearchBot","Claude-User","PerplexityBot","Perplexity-User"] +# training/bulk crawlers - blocking these is a legitimate, separate choice +TRAINERS=["GPTBot","CCBot","Google-Extended","ClaudeBot","anthropic-ai","Bytespider"] +star_blocked=blocked('*') +blocked_fetchers=[u for u in FETCHERS if blocked(u) or (star_blocked and u.lower() not in groups)] +print("agent_fetchers_blocked", blocked_fetchers) +print("training_crawlers_blocked", [u for u in TRAINERS if blocked(u)]) +# Content-Signal grammar: tokens must be <key>=<yes|no>, keys in {search,ai-input,ai-train} +sig=[l for l in body.splitlines() if l.lower().strip().startswith('content-signal:')] +valid=False +if sig: + toks=re.findall(r'([a-z-]+)\s*=\s*(yes|no)', sig[0].split(':',1)[1].lower()) + keys={k for k,_ in toks} + valid = bool(toks) and keys <= {"search","ai-input","ai-train"} +print("content_signal_present", bool(sig), "content_signal_valid", valid) +PY + +# 3. llms.txt (+ soft-404 guard: a 200 that is actually an HTML error page must fail) +curl -fsSI $ORIGIN/llms.txt -o /dev/null -w '%{http_code} %{content_type} %{size_download}\n' +curl -fsS $ORIGIN/llms.txt -o /tmp/llms.txt; wc -c < /tmp/llms.txt +head -c 200 /tmp/llms.txt | grep -iqE '<!doctype html|<html' && echo "llms.txt SOFT-404 (html body)" || echo "llms.txt non-html" + +# 4. /index.md +curl -fsSI $ORIGIN/index.md -o /dev/null -w '%{http_code} %{content_type}\n' + +# 5. Homepage Link: response headers +curl -fsSI $ORIGIN/ | grep -iE '^link:' + +# 6. sitemap freshness: at least one <lastmod> (the signal that tells an agent a page changed) +curl -fsS $ORIGIN/sitemap.xml -o /tmp/sitemap.xml 2>/dev/null +grep -iqE '<lastmod>[^<]+</lastmod>' /tmp/sitemap.xml && echo "sitemap_lastmod present" || echo "sitemap_lastmod absent" + +# 7. modular (per-area) llms.txt variants - an agent on a specific task pulls just its slice +for p in docs/llms.txt api/llms.txt developers/llms.txt; do + code=$(curl -fsS -o /tmp/mod.txt -w '%{http_code}' "$ORIGIN/$p" 2>/dev/null) + if [ "$code" = "200" ] && ! head -c 200 /tmp/mod.txt | grep -iqE '<!doctype html|<html'; then + echo "modular_llms $p 200"; fi +done +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | sitemap.xml exists | HTTP 200 and `content-type` includes `xml` | 1 | +| 2 | robots.txt exists | HTTP 200 and at least one `Sitemap:` line | 1 | +| 3 | Content-Signal present **and** well-formed | robots.txt has a `Content-Signal:` line whose tokens are all `<search\|ai-input\|ai-train>=<yes\|no>` (`content_signal_valid` true) | 1 | +| 4 | Agent fetchers not blocked | `agent_fetchers_blocked` is empty - no `Disallow: /` (directly or via `*`) hits `ChatGPT-User`, `Claude-User`, `PerplexityBot`, etc. Blocking *training* crawlers (GPTBot/CCBot) does **not** fail this. | 1 | +| 5 | llms.txt exists & non-stub | HTTP 200, `text/plain`/`text/markdown`, ≥ 200 bytes, **and not an HTML soft-404** | 1 | +| 6 | /index.md fallback | HTTP 200 and `content-type` includes `text/markdown` | 1 | +| 7 | Link headers on homepage | At least 1 `Link:` rel of `sitemap`, `describedby`, `service-desc`, or `api-catalog` | 1 | +| 8 | Link header targets resolve | Each `Link:` target URL returns HTTP < 400 (dead pointers don't count) | 1 | +| 9 | Sitemap carries `<lastmod>` | `/sitemap.xml` body contains ≥ 1 `<lastmod>…</lastmod>` element (`sitemap_lastmod present`) - the freshness signal that tells an agent a page changed | 1 | +| 10 | Modular llms.txt | At least one per-area variant (`/docs/llms.txt`, `/api/llms.txt`, `/developers/llms.txt`) returns HTTP 200 and is not an HTML soft-404 | 1 | + +Status: **Pass** ≥ 9/10 · **Partial** 3-8/10 · **Fail** 0-2/10. + +> Sub-check 4 encodes the most-missed nuance in agent-readiness: an agent acting *on behalf of a user* (e.g. `ChatGPT-User`, `Claude-User`) is not a training crawler. A site may legitimately block `GPTBot`/`CCBot` (training) while staying fully open to fetchers - that is a Pass. A blanket `User-agent: * / Disallow: /`, or an explicit fetcher block, fails it. + +## Codebase hints + +- **PHP (classic)**: drop static `sitemap.xml`, `robots.txt`, `llms.txt`, `index.md` in webroot; emit `Link:` via `header()` calls in a shared bootstrap (`inc/headers.php`). +- **Next.js (app router)**: `app/robots.ts`, `app/sitemap.ts`, `public/llms.txt`, `public/index.md`, `middleware.ts` for `Link:` headers. +- **Next.js (pages router)**: `pages/api/robots.ts`, `pages/sitemap.xml.ts`, `public/llms.txt`, `public/index.md`, custom server or middleware for `Link:`. +- **Rails**: `config/routes.rb` + `app/views/robots.text.erb`, `sitemap_generator` gem, `public/llms.txt`, `public/index.md`, `before_action` to set headers. +- **Django**: `django.contrib.sitemaps`, `urls.py` route for `robots.txt`, static `llms.txt`, middleware for `Link:`. +- **Express / Node**: routes for `/robots.txt`, `/sitemap.xml`, `/llms.txt`, `/index.md`; `res.setHeader('Link', …)` in shared middleware. +- **Cloudflare Pages / Workers**: `_headers` for `Link:`, `public/` for the four files. +- **Static (Hugo/Jekyll/Astro/11ty)**: `static/` or `public/` directory. + +## Auto-fix template + +```text +# /robots.txt +Sitemap: $ORIGIN/sitemap.xml + +# Default policy: declare AI preferences, allow everyone to crawl. +User-agent: * +Content-Signal: search=yes, ai-input=yes, ai-train=no +Disallow: + +# OPTIONAL - block *training* crawlers while staying open to agents. +# Do NOT add ChatGPT-User / Claude-User / PerplexityBot here: those are +# user-triggered fetchers, and blocking them fails agent-readiness (sub-check 4). +User-agent: GPTBot +Disallow: / + +User-agent: CCBot +Disallow: / + +User-agent: Google-Extended +Disallow: / +``` + +```text +# /llms.txt +# <Brand> + +> One sentence: what you do and who you serve. + +## Use cases +- ... +- ... + +## API & docs +- Docs: $ORIGIN/docs +- OpenAPI: $ORIGIN/openapi.json +``` + +```text +# /index.md +# <Brand> + +<Same value-prop paragraph as the HTML homepage, in plain markdown.> +``` + +```text +# Link: headers (set on every HTML response) +Link: </sitemap.xml>; rel="sitemap" +Link: </llms.txt>; rel="describedby" +Link: </openapi.json>; rel="service-desc" +Link: </.well-known/api-catalog>; rel="api-catalog" +``` + +## References + +See: [content/m1-1-discovery-files.md](../content/m1-1-discovery-files.md) diff --git a/audit/m1-2.md b/audit/m1-2.md new file mode 100644 index 0000000..5a668d3 --- /dev/null +++ b/audit/m1-2.md @@ -0,0 +1,124 @@ +--- +id: m1-2 +title: Well-known agent files +complexity: 1 +impact: 3 +visualChange: none +weight_total: 7 +--- + +# m1-2 - Well-known agent files + +## Probe + +```bash +# All well-known JSONs an agent looks for. Guard against HTML soft-404s: +# a 200 that returns an HTML error page must NOT count as a valid file. +for p in ai-plugin.json agent.json agent-card.json mcp.json mcp/server-card.json; do + printf '%s → ' "/.well-known/$p" + code=$(curl -fsS -o /tmp/wk.json -w '%{http_code}' "$ORIGIN/.well-known/$p" 2>/dev/null) + printf '%s ' "$code" + if jq -e 'type=="object" or type=="array"' /tmp/wk.json >/dev/null 2>&1; then + echo "valid JSON" + else + head -c 80 /tmp/wk.json | grep -iqE '<!doctype|<html' && echo "SOFT-404 (html)" || echo "not JSON" + fi +done + +# A2A agent card (newer SEP) - validate against the A2A shape, not just presence. +curl -fsS $ORIGIN/.well-known/agent-card.json -o /tmp/a2a.json -w 'a2a %{http_code}\n' +jq -e ' + (.name // .id) and (.url // .endpoint // .endpoints) and (.version // .protocolVersion) + and ((.skills // .capabilities // []) | length >= 0) +' /tmp/a2a.json >/dev/null 2>&1 && echo "a2a-card valid" || echo "a2a-card missing/invalid" + +# DNS-AID (draft-mozleywilliams-dnsop-dnsaid) - pure DNS-over-HTTPS, no resolver install. +# Org agent index lives at _index._agents.<domain> as an SVCB record (protocol carried in the +# `alpn` SvcParam, not in _mcp/_a2a labels). A _agents-challenge.<domain> TXT proves domain +# control. Records SHOULD be DNSSEC-signed - the DoH `AD` flag reports authenticated data. +BARE=$(echo "$HOST" | sed -E 's/^www\.//') +dohq() { curl -fsS -H 'accept: application/dns-json' \ + "https://cloudflare-dns.com/dns-query?name=$1&type=$2" 2>/dev/null; } +dohq "_index._agents.$BARE" SVCB \ + | jq -r '"dnsaid_index answers=" + ((.Answer // []) | length | tostring) + " AD=" + ((.AD // false)|tostring)' \ + 2>/dev/null || echo "dnsaid_index answers=0 AD=false" +dohq "_agents-challenge.$BARE" TXT \ + | jq -r '"dnsaid_challenge answers=" + ((.Answer // []) | length | tostring)' \ + 2>/dev/null || echo "dnsaid_challenge answers=0" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | `/.well-known/ai-plugin.json` | HTTP 200, valid JSON (not HTML soft-404) with `name_for_model` + `api.url` | 1 | +| 2 | `/.well-known/agent.json` OR `agent-card.json` | HTTP 200, valid JSON describing agent identity | 1 | +| 3 | `/.well-known/mcp.json` OR `mcp/server-card.json` | HTTP 200, valid JSON pointing at an MCP server endpoint (`url`/`serverUrl`/`endpoint`, incl. inside `mcpServers[]`) and/or declaring a `transport` (e.g. `streamable-http`) | 1 | +| 4 | Each file declared has `name` + `description` ≥ 40 chars | Names+descriptions present (not placeholders) | 1 | +| 5 | Each file references a reachable endpoint | URLs inside resolve to HTTP < 500 | 1 | +| 6 | A2A agent-card conforms to schema | `agent-card.json` has identity (`name`/`id`) **+** endpoint (`url`/`endpoint`) **+** `version` (the `a2a-card valid` line) | 1 | +| 7 | DNS-AID record published | `_index._agents.<domain>` returns ≥ 1 SVCB answer (the org agent index), or a `_agents-challenge.<domain>` TXT is present. DNSSEC-authenticated (`AD=true`) is a bonus, never required | 1 | + +Status: **Pass** ≥ 6/7 · **Partial** 3-5/7 · **Fail** 0-2/7. + +> Sub-checks 6-7 are additive: a site passes comfortably on the classic well-known files alone (5/7 → Partial→Pass boundary), while a forward-looking setup also publishes a schema-valid A2A card and DNS-AID records so agents can discover it without first fetching HTML. DNS-AID (`draft-mozleywilliams-dnsop-dnsaid`) is an early IETF draft - treat its presence as a forward-looking signal, never penalise its absence. + +## Codebase hints + +Framework-agnostic - these are static JSON files at fixed paths. + +- **Any framework**: serve from webroot under `.well-known/`. On nginx ensure `location ~ /\. { allow all; }` doesn't block dotfile directories. +- **PHP**: `.well-known/*.json` as plain files in webroot. +- **Next.js**: `public/.well-known/*.json` (App Router serves them automatically). +- **Cloudflare Pages**: `public/.well-known/*.json` works as-is. +- **GitHub Pages / Jamstack**: include `.well-known/` in build output, set `include: [.well-known]` in `_config.yml` for Jekyll. +- **DNS-AID** (sub-check 7): add records at your DNS provider - no app change. Publish an `SVCB` record at `_index._agents.<domain>` pointing at your agent index (per-agent `SVCB` records at `agent-name.<domain>` carry the protocol in the `alpn` SvcParam, e.g. `alpn="mcp"`/`"a2a"`). Sign the zone with DNSSEC so consumers can trust the records (Cloudflare, Route 53, NS1 all support SVCB + DNSSEC). +- **A2A card** (sub-check 6): the `url`/`endpoint` should point at your A2A server; if you don't run one, you can still publish identity + `skills: []` so agents resolve who you are. + +## Auto-fix template + +```json +// /.well-known/ai-plugin.json +{ + "schema_version": "v1", + "name_for_human": "<Brand>", + "name_for_model": "<brand_snake>", + "description_for_human": "<one-sentence what we do>", + "description_for_model": "Use this tool to <core capability>. Auth: OAuth.", + "auth": { "type": "oauth", "authorization_url": "https://example.com/.well-known/oauth-authorization-server" }, + "api": { "type": "openapi", "url": "https://example.com/openapi.json" }, + "logo_url": "https://example.com/logo.png", + "contact_email": "support@example.com", + "legal_info_url": "https://example.com/legal" +} +``` + +```json +// /.well-known/mcp.json (or a 307 from /.well-known/mcp). Discovery-file field names +// track the MCP discovery SEPs (SEP-1649 / SEP-1960), still stabilizing; the firm part +// is transport: "streamable-http". Same shape as the content guideline's example. +{ + "mcpServers": [ + { + "name": "<brand>", + "description": "<what tools this server exposes>", + "version": "1.0.0", + "url": "https://mcp.example.com", + "transport": "streamable-http", + "authorization": { "type": "oauth2", "metadata": "https://example.com/.well-known/oauth-authorization-server" } + } + ] +} +``` + +```text +# DNS-AID - DNS records (draft-mozleywilliams-dnsop-dnsaid; set at your DNS provider). +# Org index → SVCB pointing at an agent-index host; per-agent SVCB carries the protocol in alpn. +_index._agents.example.com. 3600 IN SVCB 1 agent-index.example.com. ( alpn="a2a,mcp" ) +mcp._agents.example.com. 3600 IN SVCB 1 mcp.example.com. ( alpn="mcp" port=443 ) +# Sign the zone with DNSSEC so consumers can authenticate these records (RFC 9364). +``` + +## References + +See: [content/m1-2-well-known-agent-files.md](../content/m1-2-well-known-agent-files.md) diff --git a/audit/m1-3.md b/audit/m1-3.md new file mode 100644 index 0000000..7c90d9e --- /dev/null +++ b/audit/m1-3.md @@ -0,0 +1,91 @@ +--- +id: m1-3 +title: Render content without JavaScript +complexity: 3 +impact: 4 +visualChange: low +weight_total: 9 +--- + +# m1-3 - Render content without JavaScript + +## Probe + +```bash +# Fetch raw HTML (no JS execution) of homepage and a representative deep page +curl -fsSL -A 'Mozilla/5.0 (compatible; AgentAudit/1.0)' $ORIGIN/ -o /tmp/home.html +wc -c /tmp/home.html + +# Strip script/style/comments, count visible text + content-efficiency ratio +python3 -c ' +import re,sys +raw=open("/tmp/home.html").read() +h=re.sub(r"(?is)<(script|style|noscript)\b.*?</\1>","",raw) +h=re.sub(r"(?is)<!--.*?-->","",h) +txt=re.sub(r"(?is)<[^>]+>"," ",h) +txt=re.sub(r"\s+"," ",txt).strip() +nchars=len(txt); nbytes=len(raw) +print("text_chars",nchars) +print("html_bytes",nbytes) +print("content_efficiency", round(nchars/nbytes,3) if nbytes else 0) # visible-text / total-bytes +print("est_tokens", nchars//4) # rough token cost to read the page +print("h1_count", len(re.findall(r"(?is)<h1\b",raw))) +print("anchor_count", len(re.findall(r"(?is)<a\s+[^>]*href=",raw))) +# empty-shell tells: a near-empty root mount that JS later fills +print("shell_markers", bool(re.search(r"(?is)id=[\"\x27](root|app|__next|__nuxt)[\"\x27]\s*>\s*</",raw)) or "enable javascript" in raw.lower()) +# alt-text coverage: multimodal agents read alt as the primary image signal +imgs=re.findall(r"(?is)<img\b[^>]*>",raw) +with_alt=sum(1 for t in imgs if re.search(r"(?i)\balt\s*=",t)) +print("img_total",len(imgs),"img_with_alt",with_alt,"alt_ratio", round(with_alt/len(imgs),2) if imgs else 1.0) +# document <head> completeness: AI systems lean on these to resolve/disambiguate pages +print("has_canonical", bool(re.search(r"(?is)<link[^>]+rel=[\"\x27]?canonical",raw))) +print("has_lang", bool(re.search(r"(?is)<html[^>]+\blang=",raw))) +print("has_og", bool(re.search(r"(?is)<meta[^>]+property=[\"\x27]?og:(title|description|image)",raw))) +' + +# Bot reachability: do real agent fetchers get the same page, or a 403/challenge/cloaked stub? +BASE=$(curl -fsS -A 'Mozilla/5.0' -o /dev/null -w '%{http_code}' $ORIGIN/) +echo "baseline(browser-UA) $BASE" +for UA in "ChatGPT-User/1.0" "Claude-User/1.0" "PerplexityBot/1.0" "Googlebot/2.1 (Google-Extended)" "ClaudeBot/1.0"; do + read code size < <(curl -fsS -A "$UA" -o /tmp/ua.html -w '%{http_code} %{size_download}' $ORIGIN/ 2>/dev/null || echo "000 0") + grep -iqE 'just a moment|cf-browser-verification|captcha|enable javascript to continue' /tmp/ua.html && chal=" CHALLENGE" || chal="" + printf ' %-38s %s bytes=%s%s\n' "$UA" "$code" "$size" "$chal" +done + +# Probe one product/content page too - find a link from sitemap +DEEP=$(curl -fsS $ORIGIN/sitemap.xml | grep -oE '<loc>[^<]+</loc>' | head -2 | tail -1 | sed -E 's|<[^>]+>||g') +[ -n "$DEEP" ] && curl -fsSL "$DEEP" -o /tmp/deep.html && wc -c /tmp/deep.html +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | Homepage SSR body ≥ 2 KB visible text | Stripped text ≥ 2000 chars | 1 | +| 2 | Exactly one `<h1>` on homepage | Exactly 1 `<h1>` match | 1 | +| 3 | ≥ 10 real `<a href>` links in raw HTML | Anchor count ≥ 10 | 1 | +| 4 | Deep page SSR body ≥ 1 KB visible text | Stripped text on a sample product/article page ≥ 1000 chars | 1 | +| 5 | No "JS required" or empty `<body>` shell | `shell_markers` false - no `enable javascript` and no empty `#root`/`#app`/`#__next` mount | 1 | +| 6 | Agent fetchers reach the page un-challenged | Every agent UA returns 2xx/3xx (not 403/429/503), no Cloudflare/captcha challenge page, and byte size within 50% of baseline (no cloaking/stub) | 1 | +| 7 | Content efficiency reasonable | `content_efficiency` ≥ 0.05 (visible text is at least 5% of bytes - flags JS-bloated shells that cost agents tokens for little signal) | 1 | +| 8 | Alt text on ≥ 80% of images | `alt_ratio` ≥ 0.8 (or no `<img>` on the page) - multimodal agents read `alt` as the primary image signal | 1 | +| 9 | `<head>` resolves & disambiguates | `has_canonical` **and** `has_lang` **and** `has_og` all true (self-referential `rel=canonical`, `<html lang>`, ≥ 1 `og:title`/`description`/`image`) | 1 | + +Status: **Pass** ≥ 8/9 · **Partial** 3-7/9 · **Fail** 0-2/9. + +> Sub-check 6 is the one a static-HTML audit usually misses: a site can be perfectly server-rendered yet sit behind bot-management that returns `403`/`Just a moment…` to `ChatGPT-User` or `PerplexityBot`. To an agent that is indistinguishable from "no content." A large byte-size delta between an agent UA and a browser UA is a cloaking signal - surface it verbatim. + +## Codebase hints + +- **Next.js**: prefer `app/` Server Components or `getServerSideProps`/`getStaticProps` in `pages/`. Avoid `dynamic(() => import(...), { ssr: false })` for above-the-fold copy. +- **React SPA (Vite/CRA)**: migrate to Next.js or add prerender (`react-snap`, `prerender.io`, `vite-plugin-prerender`). +- **Vue SPA**: switch to Nuxt SSR/SSG or use `vite-plugin-vue-ssr-prerender`. +- **PHP (classic / WordPress / Laravel)**: already SSR by default - make sure no critical content is injected by JS after page load. +- **Rails / Django / Phoenix**: SSR by default; audit any `data-react-component` or Stimulus shells that replace content. +- **Static (Hugo/Jekyll/Astro/11ty)**: SSR by default - check that `client:only` islands (Astro) aren't hiding key copy. +- **Bot challenges (sub-check 6)**: if agent UAs get `403`/`Just a moment…`, allowlist them in your WAF. **Cloudflare**: AI Crawl Control → set verified/agent bots to *Allow* (don't leave them under Bot Fight Mode). **Akamai/Fastly**: add the fetcher UAs/ASNs to an allow rule. Prefer Web Bot Auth ([m3-2](./m3-2.md)) over UA allowlists where supported. +- **Content efficiency (sub-check 7)**: trim render-blocking inline JSON state, lazy-load below-the-fold, or serve a markdown twin ([m2-3](./m2-3.md)) so agents pay fewer tokens to read you. + +## References + +See: [content/m1-3-readable-without-js.md](../content/m1-3-readable-without-js.md) diff --git a/audit/m1-4.md b/audit/m1-4.md new file mode 100644 index 0000000..0925fa2 --- /dev/null +++ b/audit/m1-4.md @@ -0,0 +1,84 @@ +--- +id: m1-4 +title: Topical authority & coding rules +complexity: 3 +impact: 5 +visualChange: high +weight_total: 6 +--- + +# m1-4 - Topical authority & coding rules + +## Probe + +```bash +# 1. AGENTS.md / .cursorrules (origin first; many sites only ship these in the public repo). +# Fetch the body so we can judge substance, not just presence. +AGENTS_OK=0; AGENTS_SUBSTANTIVE=0 +for p in AGENTS.md .cursorrules .claude/CLAUDE.md docs/AGENTS.md; do + code=$(curl -fsS -o /tmp/agents.txt -w '%{http_code}' "$ORIGIN/$p" 2>/dev/null) + printf '%s %s\n' "$p" "$code" + if [ "$code" = "200" ]; then AGENTS_OK=1 + bytes=$(wc -c < /tmp/agents.txt) + if [ "$bytes" -ge 600 ] && grep -iqE 'build|test|run|install|convention|lint|structure' /tmp/agents.txt; then AGENTS_SUBSTANTIVE=1; fi + fi +done +echo "agents_md_ok $AGENTS_OK agents_md_substantive $AGENTS_SUBSTANTIVE" + +# 2. Comparison / "best for" pages - fetch the body and count words (content asks for 1500+) +BEST_WORDS=0 +for p in compare alternatives vs best; do + code=$(curl -fsSL -o /tmp/best.html -w '%{http_code}' "$ORIGIN/$p" 2>/dev/null) + if [ "$code" = "200" ]; then + w=$(python3 -c 'import re;raw=open("/tmp/best.html").read();h=re.sub(r"(?is)<(script|style)\b.*?</\1>"," ",raw);t=re.sub(r"(?is)<[^>]+>"," ",h);print(len(re.findall(r"\S+",t)))') + printf '/%s 200 words=%s\n' "$p" "$w"; [ "$w" -gt "$BEST_WORDS" ] && BEST_WORDS=$w + else printf '/%s %s\n' "$p" "$code"; fi +done +echo "best_page_words $BEST_WORDS" + +# 3. Wikipedia / Wikidata presence - best-effort via public APIs +BRAND=$(printf '%s' "$HOST" | sed -E 's/^www\.//; s/\..*$//') +curl -fsS "https://en.wikipedia.org/api/rest_v1/page/summary/$BRAND" -o /tmp/wp.json -w 'wikipedia %{http_code}\n' +curl -fsS "https://www.wikidata.org/w/api.php?action=wbsearchentities&search=$BRAND&language=en&format=json" -o /tmp/wd.json +jq -r '.search | length' /tmp/wd.json 2>/dev/null + +# 4. GitHub org/repo link from homepage +curl -fsS $ORIGIN/ | grep -oE 'https://github\.com/[A-Za-z0-9._-]+' | sort -u | head -3 +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | `AGENTS.md` or `.cursorrules` discoverable | HTTP 200 on origin (`agents_md_ok`) OR linked from a public GitHub repo named in homepage | 1 | +| 2 | `AGENTS.md` is substantive (not a stub) | `agents_md_substantive` - the served file is ≥ 600 bytes and names build/test/run/conventions/structure guidance (only scorable when served at origin; a repo-only file Fails this row) | 1 | +| 3 | "Best X for Y" / compare page ≥ 1500 words | `best_page_words` ≥ 1500 on a `/compare`, `/alternatives`, `/vs`, or `/best` page - the depth answer engines cite | 1 | +| 4 | Wikipedia article exists | Wikipedia REST summary returns 200 with `extract` field | 1 | +| 5 | Wikidata entity with P856 = origin | wbsearchentities returns ≥ 1 entity matching brand, with official-website claim | 1 | +| 6 | Public GitHub org linked from homepage | At least one `github.com/<org>` URL in homepage HTML, repo exists | 1 | + +Status: **Pass** ≥ 5/6 · **Partial** 2-4/6 · **Fail** 0-1/6 · **N/A** for purely internal tools. + +## Codebase hints + +- **All frameworks**: comparison/alternatives pages live in the marketing site routing (e.g., `pages/compare/[competitor].tsx`, `app/compare/[slug]/page.tsx`, Rails `config/routes.rb`, Hugo `content/compare/*.md`). +- **AGENTS.md**: drop at repo root next to `README.md`. Same file works for Cursor, Claude Code, Cody, Continue. Pair with a shorter `.cursorrules` for IDE-level rules (preferred imports, banned patterns, version constraints). +- **JSON-LD on comparison pages**: add `Product` + `ComparisonPage` schema (or `FAQPage` for "X vs Y" Q&A). +- **Wikipedia/Wikidata**: not a codebase task - earn third-party press coverage, then draft article with cited references. Update Wikidata P856 (official website) and P31 (instance of) for the brand entity. + +## Auto-fix template + +```markdown +# /AGENTS.md +This repo powers <Brand> at $ORIGIN. AI coding agents working here should: + +- Treat `src/` as the production codebase. Tests live in `tests/`. +- Run `npm test` (or `pytest`, `bundle exec rspec`) before committing. +- Match the prevailing style in nearby files - ESLint/Prettier configs are authoritative. +- Public docs: $ORIGIN/docs. API spec: $ORIGIN/openapi.json. MCP server: $ORIGIN/mcp. +- Don't run migrations, deploys, or destructive scripts without explicit user confirmation. +``` + +## References + +See: [content/m1-4-topical-authority.md](../content/m1-4-topical-authority.md) diff --git a/audit/m2-1.md b/audit/m2-1.md new file mode 100644 index 0000000..cd096b0 --- /dev/null +++ b/audit/m2-1.md @@ -0,0 +1,164 @@ +--- +id: m2-1 +title: JSON-LD structure +complexity: 2 +impact: 4 +visualChange: none +weight_total: 9 +--- + +# m2-1 - JSON-LD structure + +## Probe + +```bash +curl -fsSL $ORIGIN/ -o /tmp/home.html + +# Extract every <script type="application/ld+json"> block + required-prop + head metadata +python3 - <<'PY' +import re,json +h=open("/tmp/home.html").read() +blocks=re.findall(r'(?is)<script[^>]+type=["\']application/ld\+json["\'][^>]*>(.*?)</script>', h) +print("blocks_found", len(blocks)) +types=[]; sameAs=[]; has_org=False; has_contact=False; has_addr=False +# Required props per @type - having the type without these is a hollow schema. +REQ={"Product":["name","offers"], "SoftwareApplication":["name","applicationCategory"], + "Article":["headline"], "FAQPage":["mainEntity"], "Service":["name"], + "Offer":["price"], "BreadcrumbList":["itemListElement"]} +type_ok={} # type -> bool(all required props present on at least one node of that type) +for b in blocks: + try: d=json.loads(b.strip()) + except Exception: continue + nodes=d.get("@graph") if isinstance(d,dict) and "@graph" in d else (d if isinstance(d,list) else [d]) + for n in nodes: + if not isinstance(n,dict): continue + t=n.get("@type") + tl=(t if isinstance(t,list) else [t]) if t else [] + for one in tl: + types.append(one) + if one in REQ: + type_ok[one]= type_ok.get(one,False) or all(n.get(p) for p in REQ[one]) + if isinstance(n.get("sameAs"),list): sameAs += n["sameAs"] + elif isinstance(n.get("sameAs"),str): sameAs.append(n["sameAs"]) + if (t if isinstance(t,str) else (t[0] if isinstance(t,list) and t else "")) in ("Organization","Corporation","LocalBusiness"): + has_org=True + if n.get("contactPoint"): has_contact=True + if n.get("address"): has_addr=True +rich=[k for k in type_ok if k!="Organization"] +print("types", sorted(set(types))) +print("sameAs", sameAs[:10]) +print("organization", has_org, "contactPoint", has_contact, "address", has_addr) +print("rich_types_complete", {k:type_ok[k] for k in rich}) # which non-Org types have all required props + +# Head metadata completeness (canonical/lang/OG/twitter) +def meta(name): + m=re.search(r'(?is)<meta[^>]+(?:property|name)=["\']%s["\'][^>]*content=["\']([^"\']+)' % re.escape(name), h) + return bool(m) +lang=bool(re.search(r'(?is)<html[^>]+lang=', h)) +canon=bool(re.search(r'(?is)<link[^>]+rel=["\']canonical["\']', h)) +present={"canonical":canon,"html-lang":lang,"og:title":meta("og:title"), + "og:description":meta("og:description"),"og:image":meta("og:image"), + "og:type":meta("og:type"),"twitter:card":meta("twitter:card")} +print("metadata", present, "metadata_complete", sum(present.values())>=6) + +# Speakable: a `speakable` property (on WebPage/Article) or a SpeakableSpecification node. +# Its cssSelectors are only useful if they actually resolve to elements on the page. +sel=[] +for b in blocks: + try: d=json.loads(b.strip()) + except Exception: continue + nodes=d.get("@graph") if isinstance(d,dict) and "@graph" in d else (d if isinstance(d,list) else [d]) + for n in nodes: + if not isinstance(n,dict): continue + spec=n.get("speakable") if isinstance(n.get("speakable"),dict) else (n if n.get("@type")=="SpeakableSpecification" else None) + if isinstance(spec,dict) and isinstance(spec.get("cssSelector"),list): + sel+=[s for s in spec["cssSelector"] if isinstance(s,str)] +def resolves(s): + s=s.strip().split()[-1] # last simple selector in a descendant chain + if s.startswith('.'): return bool(re.search(r'(?is)class=["\'][^"\']*\b%s\b'%re.escape(s[1:]), h)) + if s.startswith('#'): return bool(re.search(r'(?is)id=["\']%s["\']'%re.escape(s[1:]), h)) + return bool(re.search(r'(?is)<%s\b'%re.escape(s), h)) +ok_sel=[s for s in sel if resolves(s)] +print("speakable_selectors", sel, "speakable_resolves", len(ok_sel)) +PY + +# Resolve the first sameAs URL - a dead authority link is no entity link at all +FIRST_SAMEAS=$(python3 -c "import re,json;h=open('/tmp/home.html').read(); +b=re.findall(r'\"sameAs\"\s*:\s*\[?\s*\"(https?://[^\"]+)',h);print(b[0] if b else '')") +[ -n "$FIRST_SAMEAS" ] && curl -fsSI "$FIRST_SAMEAS" -o /dev/null -w "sameAs[0] $FIRST_SAMEAS → %{http_code}\n" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | At least 1 JSON-LD block parses cleanly | `blocks_found ≥ 1` and `json.loads` succeeds | 1 | +| 2 | Organization (or Corporation / LocalBusiness) declared | `has_org` true | 1 | +| 3 | Organization has `contactPoint` | `has_contact` true | 1 | +| 4 | Organization has `address` (PostalAddress) | `has_addr` true | 1 | +| 5 | `sameAs` to a known entity **that resolves** | ≥ 1 link matches `wikipedia.org`/`wikidata.org`/`github.com`/`linkedin.com`/`x.com` **and** `sameAs[0]` returns HTTP < 400 | 1 | +| 6 | Page-appropriate `@type` beyond Organization | At least one of: `Product`, `SoftwareApplication`, `Article`, `FAQPage`, `BreadcrumbList`, `WebSite`, `Service` | 1 | +| 7 | Rich types carry their required props | Every non-Organization type in `rich_types_complete` is `true` (e.g. `Product` has `name`+`offers`, `FAQPage` has `mainEntity`) - no hollow `@type` declarations | 1 | +| 8 | Head metadata complete | `metadata_complete` true - ≥ 6 of `canonical`, `html lang`, `og:title/description/image/type`, `twitter:card` present | 1 | +| 9 | Speakable markup that resolves | `speakable`/`SpeakableSpecification` declares ≥ 1 `cssSelector` **and** `speakable_resolves ≥ 1` (at least one selector matches an element on the page - not a dangling selector) | 1 | + +Status: **Pass** ≥ 8/9 · **Partial** 3-7/9 · **Fail** 0-2/9. + +> Sub-check 7 replaces the old "validate at schema.org" manual row with something probe-derivable: declaring `@type: Product` with no `name`/`offers` is what trips agents and rich-result parsers, and we can see it in the bytes. Sub-check 8 folds in the OG/canonical/lang metadata that agents read for entity resolution. Sub-check 9 rewards `speakable` markup only when its selectors actually point at on-page elements - a `SpeakableSpecification` whose `cssSelector` matches nothing is dead weight voice agents can't use. + +## Codebase hints + +- **Next.js**: `<Script type="application/ld+json" strategy="afterInteractive">` in `app/layout.tsx`, or use [`schema-dts`](https://github.com/google/schema-dts) for typed objects. +- **PHP (WordPress)**: Yoast SEO or Rank Math emits Organization automatically - extend with `wpseo_schema_*` filters for Product/FAQ. +- **PHP (custom)**: inline `<script type="application/ld+json">` in `inc/header.php`; per-page Product schema in product templates. +- **Rails**: `json-ld` gem, render in `application.html.erb`. +- **Django**: `django-meta` or hand-rolled template tag. +- **Hugo/Jekyll/Astro**: include in `<head>` partial; per-page front matter feeds the JSON. +- **All**: per-page schemas (Product, FAQPage, Article) live in the page template, not the global layout. + +## Auto-fix template + +```html +<script type="application/ld+json"> +{ + "@context": "https://schema.org", + "@graph": [ + { + "@type": "Organization", + "@id": "$ORIGIN/#org", + "name": "<Brand>", + "url": "$ORIGIN", + "logo": "$ORIGIN/logo.png", + "description": "<one-sentence value prop>", + "sameAs": [ + "https://en.wikipedia.org/wiki/<Brand>", + "https://www.wikidata.org/wiki/Q<NNN>", + "https://github.com/<org>", + "https://x.com/<handle>", + "https://www.linkedin.com/company/<slug>" + ], + "contactPoint": [{ + "@type": "ContactPoint", + "contactType": "customer support", + "email": "support@example.com", + "telephone": "+1-555-0100", + "availableLanguage": ["en"] + }], + "address": { + "@type": "PostalAddress", + "streetAddress": "100 Main St", + "addressLocality": "City", + "addressRegion": "ST", + "postalCode": "12345", + "addressCountry": "US" + } + }, + { "@type": "WebSite", "@id": "$ORIGIN/#site", "url": "$ORIGIN", "name": "<Brand>", "publisher": { "@id": "$ORIGIN/#org" } } + ] +} +</script> +``` + +## References + +See: [content/m2-1-json-ld.md](../content/m2-1-json-ld.md) diff --git a/audit/m2-2.md b/audit/m2-2.md new file mode 100644 index 0000000..61640ff --- /dev/null +++ b/audit/m2-2.md @@ -0,0 +1,102 @@ +--- +id: m2-2 +title: Useful llms.txt content +complexity: 2 +impact: 5 +visualChange: none +weight_total: 9 +--- + +# m2-2 - Useful llms.txt content + +## Probe + +```bash +curl -fsS $ORIGIN/llms.txt -o /tmp/llms.txt +curl -fsS $ORIGIN/llms-full.txt -o /tmp/llms-full.txt +curl -fsS $ORIGIN/docs/llms.txt -o /tmp/llms-docs.txt +curl -fsS $ORIGIN/api/llms.txt -o /tmp/llms-api.txt + +# Stats + rough token cost (≈ chars/4) +for f in /tmp/llms.txt /tmp/llms-full.txt /tmp/llms-docs.txt /tmp/llms-api.txt; do + [ -s "$f" ] || continue + c=$(wc -c < "$f") + printf '%s chars=%d est_tokens=%d headings=%d links=%d code_fences=%d\n' "$f" \ + "$c" "$((c/4))" \ + "$(grep -cE '^#{1,3} ' "$f")" \ + "$(grep -cE '\]\(https?://' "$f")" \ + "$(grep -c '^```' "$f")" +done + +# Agent-instruction block? +grep -iE '^(##|###) *(for agents|agent instructions|when to use|when to recommend)' /tmp/llms.txt + +# Sample-resolve up to 3 absolute links in the index - dead links erode trust +grep -oE 'https?://[^)" ]+' /tmp/llms.txt | sort -u | head -3 | while read u; do + curl -fsSI "$u" -o /dev/null -w "link $u → %{http_code}\n" 2>/dev/null || echo "link $u → DEAD" +done +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | `/llms.txt` exists, non-stub | ≥ 500 bytes, ≥ 3 markdown headings | 1 | +| 2 | Index links to docs / API / pricing | ≥ 3 absolute markdown links | 1 | +| 3 | `llms-full.txt` exists | ≥ 5 KB and < 200,000 bytes (the content guideline's one-shot-ingestion ceiling) | 1 | +| 4 | Modular llms.txt per area | At least 1 of `/docs/llms.txt`, `/api/llms.txt`, `/developers/llms.txt` returns 200 | 1 | +| 5 | Agent-instruction block | `llms.txt` has a `## For agents` / `## When to use` / `## When to recommend us` section | 1 | +| 6 | API + auth coverage in llms-full | `llms-full.txt` (or `llms.txt`) mentions both `endpoint`/`api` and `oauth`/`api key`/`bearer` | 1 | +| 7 | Dated stats with named authors | At least one `> YYYY-MM-DD` or named citation in body | 1 | +| 8 | Index links resolve | All sampled (≤ 3) absolute links return HTTP < 400 | 1 | +| 9 | Index is token-efficient | `/llms.txt` `est_tokens` ≤ 5000 (≤ ~20 KB) - the index stays a cheap map an agent can always afford to load; bulk prose belongs in `llms-full.txt` | 1 | + +Status: **Pass** ≥ 8/9 · **Partial** 3-7/9 · **Fail** 0-2/9. + +## Codebase hints + +- **Any framework**: `llms.txt` is plain text - drop in webroot. Set `Content-Type: text/markdown; charset=utf-8`. +- **Static site generators**: auto-generate from `docs/` headings. Hugo: `layouts/_default/llms.txt`. Astro: `pages/llms.txt.ts`. Next.js: `app/llms.txt/route.ts`. +- **WordPress**: WP-CLI script that dumps published pages to `/llms.txt` on publish hook. +- **Rails / Django / Express**: cron or after_publish callback regenerates the file when content changes. + +## Auto-fix template + +```markdown +# <Brand> + +> <One sentence: what you do and who you serve.> + +## Use cases +- <Concrete use case 1, with 1-line description.> +- <Concrete use case 2.> +- <Concrete use case 3.> + +## Capabilities +- <Top-level capability - link to docs page.> +- ... + +## Constraints +- Rate limits: <link to docs>. +- Geographic availability: <list or "global">. +- Pricing tiers: <link>. + +## API & docs +- Docs: $ORIGIN/docs +- OpenAPI: $ORIGIN/openapi.json +- MCP server: $ORIGIN/mcp +- Auth: $ORIGIN/docs/auth (OAuth 2.0 + Dynamic Client Registration) + +## For agents +When invoked to <core task>, prefer the MCP server tool `<tool_name>` over scraping HTML. +Authentication is required for write actions; read actions are public. +On rate-limit errors, honor the `Retry-After` header. + +## About +<Brand> is operated by <Company>, founded YYYY in <City>. +Last verified: YYYY-MM-DD. +``` + +## References + +See: [content/m2-2-llms-txt-content.md](../content/m2-2-llms-txt-content.md) diff --git a/audit/m2-3.md b/audit/m2-3.md new file mode 100644 index 0000000..0c2385b --- /dev/null +++ b/audit/m2-3.md @@ -0,0 +1,96 @@ +--- +id: m2-3 +title: Document for agents +complexity: 3 +impact: 5 +visualChange: medium +weight_total: 9 +--- + +# m2-3 - Document for agents + +## Probe + +```bash +# Find a docs URL +for p in docs developers api/docs developer api/v1 developers/docs help/api; do + curl -fsSI $ORIGIN/$p -o /dev/null -w "/$p %{http_code}\n" +done + +# Pick first responding path as DOCS +DOCS=$(for p in docs developers api/docs developer; do + curl -fsS -o /dev/null -w '%{http_code} %{url_effective}\n' $ORIGIN/$p \ + | awk '$1 ~ /^2/ {print $2; exit}' +done) +echo "DOCS=${DOCS:=$ORIGIN/}" + +# --- Markdown content-negotiation contract (acceptmarkdown.com) --- +# A real test of all four parts of the contract, not just "did it 200". +echo "== markdown negotiation on $DOCS ==" +# (a) Accept: text/markdown → markdown body + Vary +curl -fsS -H 'Accept: text/markdown' -D /tmp/md.h -o /tmp/md.body "$DOCS" -w 'md %{http_code} ct=%{content_type}\n' +grep -iE '^(content-type|vary|link):' /tmp/md.h +# (b) unsatisfiable type → server SHOULD answer 406 Not Acceptable +curl -fsS -H 'Accept: application/x-this-type-does-not-exist' -o /dev/null \ + -w '406-test status=%{http_code}\n' "$DOCS" 2>/dev/null || echo "406-test status=406" +# (c) q=0 on markdown MUST fall back to HTML (honors quality values) +curl -fsS -H 'Accept: text/markdown;q=0, text/html' -o /dev/null \ + -w 'qval ct=%{content_type}\n' "$DOCS" +# (d) sibling discovery: Link: rel="alternate" type="text/markdown", or a .md twin +grep -iqE '^link:.*type="?text/markdown' /tmp/md.h && echo "link-alternate=yes" || echo "link-alternate=no" +curl -fsSI "${DOCS%/}.md" -o /dev/null -w 'twin-md status=%{http_code} ct=%{content_type}\n' 2>/dev/null + +# Code sample languages - fetch a docs page and count fences +curl -fsSL "$DOCS" -o /tmp/docs.html +python3 -c ' +import re +h=open("/tmp/docs.html").read() +langs=set(re.findall(r"language-([a-z]+)", h)) +print("languages_seen", sorted(langs)) +print("quickstart_keyword", "quickstart" in h.lower() or "getting started" in h.lower()) +print("auth_keyword", "oauth" in h.lower() or "api key" in h.lower()) +# Citability cues: a named-author byline and dated content survive the LLM citability filter +print("author_byline", bool(re.search(r"(?is)(\"author\"\s*:|rel=[\"\x27]author|\bby\s+[A-Z][a-z]+\s+[A-Z][a-z]+|<address\b)", h))) +print("dated_content", bool(re.search(r"(?is)(datePublished|dateModified|<time[^>]+datetime=|\b20[12][0-9]-[01][0-9]-[0-3][0-9]\b)", h))) +' + +# Glossary page - lets an agent resolve domain jargon without leaving your origin +GLOSSARY=0 +for p in glossary docs/glossary docs/reference/glossary help/glossary; do + code=$(curl -fsSI -o /dev/null -w '%{http_code}' "$ORIGIN/$p" 2>/dev/null) + [ "$code" = "200" ] && GLOSSARY=1 +done +echo "glossary_present $GLOSSARY" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | `/docs` (or equivalent) reachable | First 2xx among standard doc paths | 1 | +| 2 | Quickstart present | "quickstart" or "getting started" heading reachable in first crawl | 1 | +| 3 | Auth walkthrough present | "oauth" / "api key" / "authentication" section reachable | 1 | +| 4 | ≥ 4 language code samples | `languages_seen` ≥ 4 distinct (e.g. `language-curl`, `language-js`/`javascript`, `language-python`, `language-go`) - the content guideline's curl/JS/Python/Go floor | 1 | +| 5 | Per-endpoint reference page | At least one `/docs/api/.+` or `/api/reference/.+` route returns 200 with code samples | 1 | +| 6 | Markdown served **with `Vary: Accept`** | `Accept: text/markdown` → HTTP 200, `content-type` includes `text/markdown`, **and** response carries `Vary: Accept` (so caches don't serve HTML to agents) | 1 | +| 7 | Negotiation is strict (media type + q-values) | Served media type is canonical `text/markdown` (**not** `text/x-markdown`/`application/markdown`) **and** at least one of: `406-test` returns `406`, **or** `qval` falls back to `text/html` (i.e. `q=0` is honored) | 1 | +| 8 | Markdown siblings are discoverable | `link-alternate=yes` (`Link: rel="alternate" type="text/markdown"`) **or** the `.md` twin returns `text/markdown` - an agent can find the markdown without guessing | 1 | +| 9 | Citability signals | ≥ 2 of: `author_byline` true, `dated_content` true, `glossary_present` - named authors, dated specific numbers, and a jargon glossary are what survive the LLM citability filter | 1 | + +Status: **Pass** ≥ 7/9 · **Partial** 3-6/9 · **Fail** 0-2/9 · **N/A** for marketing-only sites with no API. + +> Sub-checks 6-8 implement the full [acceptmarkdown.com](https://acceptmarkdown.com) contract. The common half-implementation - serving markdown but omitting `Vary: Accept` - fails sub-check 6 because a CDN will then cache the HTML and hand it to the next agent. Serving `text/x-markdown`, or ignoring `q=0`, fails sub-check 7. Shipping a `.md` sibling with no `Link: rel="alternate"` and no content-negotiation on the canonical URL fails sub-check 8 (the agent has no way to discover it). + +## Codebase hints + +- **Docusaurus / Nextra / Starlight / Mintlify**: built-in markdown delivery; configure middleware to honor `Accept: text/markdown` (Mintlify does this OOTB). Verify `Vary: Accept` is emitted. +- **ReadMe.io / Stoplight / Redocly**: enable "agent-friendly" or "developer experience" markdown feed. +- **Next.js custom docs**: middleware reads `Accept`, returns `.mdx` source with `Content-Type: text/markdown`; **always set `Vary: Accept`**, return `406` for unsatisfiable types, and parse `q=0`. +- **Cloudflare**: enable *Markdown for Agents* (AI Crawl Control) - it converts HTML→markdown at the edge on `Accept: text/markdown`, sets `Vary: Accept`, and adds an `x-markdown-tokens` header; zero app change. +- **Nginx/Apache/Caddy**: `map`/rewrite on the `Accept` header to a pre-rendered `.md`; add `Vary: Accept`; return `406` when no representation matches. +- **Static markdown**: keep `.md` source alongside HTML output; advertise it with `Link: </page.md>; rel="alternate"; type="text/markdown"`. +- **WordPress + docs plugin**: a markdown-export plugin that honors the `Accept` header (not just `/page.md` URLs - those alone fail sub-check 6). + +## References + +See: [content/m2-3-document-for-agents.md](../content/m2-3-document-for-agents.md) diff --git a/audit/m2-4.md b/audit/m2-4.md new file mode 100644 index 0000000..1393913 --- /dev/null +++ b/audit/m2-4.md @@ -0,0 +1,62 @@ +--- +id: m2-4 +title: Competitive positioning +complexity: 2 +impact: 4 +visualChange: high +weight_total: 6 +--- + +# m2-4 - Competitive positioning + +## Probe + +```bash +# Comparison / alternatives pages +for p in compare alternatives "vs" pricing pricing.md; do + curl -fsSI "$ORIGIN/$p" -o /dev/null -w "/$p %{http_code}\n" +done + +# Markdown content negotiation on /pricing - same contract as m2-3 (markdown + Vary + strictness) +curl -fsS -H 'Accept: text/markdown' -D /tmp/pm.h -o /dev/null $ORIGIN/pricing -w 'pricing-md %{http_code} ct=%{content_type}\n' +grep -iE '^(content-type|vary):' /tmp/pm.h +curl -fsSI "$ORIGIN/pricing.md" -o /dev/null -w 'pricing.md twin %{http_code} %{content_type}\n' 2>/dev/null + +# Differentiator keywords on homepage +curl -fsSL $ORIGIN/ -o /tmp/home.html +python3 -c ' +import re +h=open("/tmp/home.html").read().lower() +hits=[w for w in ["only","first","alternative to","unlike","compared to","built for","versus"] if w in h] +print("positioning_keywords", hits) +' + +# JSON-LD Offer / AggregateRating on pricing +curl -fsSL $ORIGIN/pricing -o /tmp/pricing.html 2>/dev/null +grep -E 'application/ld\+json' /tmp/pricing.html | head -1 +grep -oE '"@type": *"(Offer|PriceSpecification|AggregateRating|Product)"' /tmp/pricing.html | sort -u +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | At least one `/compare/<competitor>` or `/alternatives` page | HTTP 200 with body ≥ 1 KB | 1 | +| 2 | `/pricing` reachable | HTTP 200 | 1 | +| 3 | Pricing has structured JSON-LD | `Offer` or `PriceSpecification` schema on pricing page | 1 | +| 4 | Pricing served as markdown | `Accept: text/markdown` on `/pricing` returns `content-type: text/markdown` **or** the `/pricing.md` twin returns `text/markdown` | 1 | +| 5 | Markdown delivery sets `Vary: Accept` | When sub-check 4 passes via content negotiation, the response also carries `Vary: Accept` (caches won't serve HTML pricing to agents). If only a `.md` twin exists, this passes when that twin is reachable. | 1 | +| 6 | Homepage uses positioning keywords | At least 1 of `only`, `first`, `alternative to`, `unlike`, `built for` | 1 | + +Status: **Pass** ≥ 5/6 · **Partial** 2-4/6 · **Fail** 0-1/6 · **N/A** for sites with no competitors or pricing. + +## Codebase hints + +- **Comparison pages**: generate one per competitor from a YAML/JSON data file. Next.js: `app/compare/[competitor]/page.tsx`. Hugo: `content/compare/*.md`. Rails: `config/routes.rb resources :compare`. +- **Schema.org/Table for feature comparisons**: wrap the comparison `<table>` with `itemscope itemtype="https://schema.org/Table"` or emit `FAQPage` JSON-LD if rendered as Q&A. +- **Pricing JSON-LD**: emit `Product` with `offers: [Offer]` block per tier. +- **pricing.md**: content-neg via middleware (Express, Next.js) or sibling file (`public/pricing.md` mirroring `public/pricing.html`). + +## References + +See: [content/m2-4-competitive-positioning.md](../content/m2-4-competitive-positioning.md) diff --git a/audit/m3-1.md b/audit/m3-1.md new file mode 100644 index 0000000..3aeabc0 --- /dev/null +++ b/audit/m3-1.md @@ -0,0 +1,108 @@ +--- +id: m3-1 +title: OAuth discovery +complexity: 5 +impact: 4 +visualChange: medium +weight_total: 8 +--- + +# m3-1 - OAuth discovery + +## Probe + +```bash +# Try discovery on both apex and api subdomain +for H in $HOST api.$(echo $HOST | sed -E 's/^www\.//') $(echo $HOST | sed -E 's/^www\.//'); do + for p in oauth-authorization-server openid-configuration oauth-protected-resource; do + URL="https://$H/.well-known/$p" + printf '%s ' "$URL" + curl -fsS -o /tmp/$p.json -w '%{http_code}\n' "$URL" 2>/dev/null || true + done +done + +# AS metadata: RFC 8414 doc, falling back to the OIDC discovery doc (same shape). +AS=/tmp/oauth-authorization-server.json +[ -s "$AS" ] || AS=/tmp/openid-configuration.json +echo "AS_SOURCE=$AS" +[ -s "$AS" ] && jq -r '{ + issuer, authorization_endpoint, token_endpoint, jwks_uri, + grant_types_supported, scopes_supported, registration_endpoint, + code_challenge_methods_supported +}' "$AS" + +# jwks_uri must actually resolve and contain keys (a dangling jwks_uri breaks token validation) +JWKS=$(jq -r '.jwks_uri // empty' "$AS" 2>/dev/null) +if [ -n "$JWKS" ]; then + curl -fsS "$JWKS" -o /tmp/as-jwks.json -w "jwks_uri $JWKS → %{http_code}\n" + jq -e '.keys | length > 0' /tmp/as-jwks.json >/dev/null 2>&1 && echo "jwks keys present" || echo "jwks empty/invalid" +fi + +# Parse PRM +[ -s /tmp/oauth-protected-resource.json ] && jq -r '{ + resource, authorization_servers, scopes_supported, bearer_methods_supported +}' /tmp/oauth-protected-resource.json + +# WWW-Authenticate hint on unauthenticated API calls +for p in api api/v1 v1 mcp; do + curl -fsS -o /dev/null -D /tmp/h.txt $ORIGIN/$p + grep -iE '^www-authenticate' /tmp/h.txt +done +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | AS metadata discoverable | HTTP 200, valid JSON at `/.well-known/oauth-authorization-server` **or** `/.well-known/openid-configuration` (`AS_SOURCE` is non-empty) | 1 | +| 2 | AS metadata has `issuer` + `authorization_endpoint` + `token_endpoint` + `jwks_uri` | All four fields non-empty | 1 | +| 3 | PKCE supported | `code_challenge_methods_supported` includes `S256` | 1 | +| 4 | Dynamic Client Registration (RFC 7591) | `registration_endpoint` present | 1 | +| 5 | `/.well-known/oauth-protected-resource` (RFC 9728) | HTTP 200, JSON with `resource` + `authorization_servers` | 1 | +| 6 | Scopes use resource-prefixed naming | At least one scope of pattern `<resource>:read` / `<resource>:write` (e.g., `orders:read`) | 1 | +| 7 | `WWW-Authenticate: Bearer resource_metadata=…` on 401 | Header present on at least one unauthenticated API probe | 1 | +| 8 | `jwks_uri` resolves with keys | The advertised `jwks_uri` returns HTTP 200 and a non-empty `keys[]` array (agents can actually validate issued tokens) | 1 | + +Status: **Pass** ≥ 7/8 · **Partial** 3-6/8 · **Fail** 0-2/8 · **N/A** for sites with no API surface. + +## Codebase hints + +- **Auth0 / Okta / Clerk / WorkOS / Stytch**: enable OIDC discovery in tenant settings; metadata is auto-published. Map PRM via "resource server" config. +- **Keycloak**: discovery is built-in at `/realms/{realm}/.well-known/openid-configuration`. Front with a static `/.well-known/oauth-authorization-server` that redirects. +- **Custom (Node)**: `node-oidc-provider`, `oauth2orize`. +- **Custom (Ruby)**: `doorkeeper-openid_connect`. +- **Custom (Python)**: `authlib`, `django-oauth-toolkit`. +- **Custom (PHP)**: `league/oauth2-server` + `web-token/jwt-framework` for JWKS. +- **Cloudflare**: front-end via Cloudflare Access for the discovery doc; back-end IdP issues tokens. + +## Auto-fix template + +```json +// /.well-known/oauth-authorization-server (RFC 8414) +{ + "issuer": "$ORIGIN", + "authorization_endpoint": "$ORIGIN/oauth/authorize", + "token_endpoint": "$ORIGIN/oauth/token", + "jwks_uri": "$ORIGIN/.well-known/jwks.json", + "registration_endpoint": "$ORIGIN/oauth/register", + "grant_types_supported": ["authorization_code", "refresh_token"], + "response_types_supported": ["code"], + "code_challenge_methods_supported": ["S256"], + "scopes_supported": ["openid", "profile", "orders:read", "orders:write"], + "token_endpoint_auth_methods_supported": ["client_secret_basic", "private_key_jwt", "none"] +} +``` + +```json +// /.well-known/oauth-protected-resource (RFC 9728) +{ + "resource": "$ORIGIN", + "authorization_servers": ["$ORIGIN"], + "scopes_supported": ["orders:read", "orders:write"], + "bearer_methods_supported": ["header"] +} +``` + +## References + +See: [content/m3-1-oauth-discovery.md](../content/m3-1-oauth-discovery.md) diff --git a/audit/m3-2.md b/audit/m3-2.md new file mode 100644 index 0000000..40cb758 --- /dev/null +++ b/audit/m3-2.md @@ -0,0 +1,125 @@ +--- +id: m3-2 +title: Web Bot Auth +complexity: 4 +impact: 3 +visualChange: none +weight_total: 5 +--- + +# m3-2 - Web Bot Auth + +## Probe + +```bash +# RFC 9421 / Web Bot Auth signature directory +curl -fsS $ORIGIN/.well-known/http-message-signatures-directory -o /tmp/dir.json -w 'sig-dir %{http_code} %{content_type}\n' + +# Validate the JWKS structure +jq -e ' + .keys + | length > 0 + and all(.[]; .kty == "OKP" and .crv == "Ed25519" and .kid and .nbf and .exp) +' /tmp/dir.json && echo "valid Ed25519 JWKS" || echo "invalid or missing" + +# Inspect key rotation window + whether at least one signing key is currently valid (kid not expired) +jq -r '.keys[] | "\(.kid) nbf=\(.nbf) exp=\(.exp)"' /tmp/dir.json 2>/dev/null +NOW=$(date +%s) +jq -e --argjson now "$NOW" '.keys | map(select((.nbf//0) <= $now and $now <= (.exp//9999999999))) | length > 0' \ + /tmp/dir.json >/dev/null 2>&1 && echo "signing key currently valid" || echo "no currently-valid signing key" + +# Signature-Input header on a sample request? (sites that *consume* signatures may surface info via headers/docs only) +curl -fsSI $ORIGIN/ | grep -iE '^(accept-signature|signature-input):' + +# Generic JWKS endpoint (the keys an agent uses to validate tokens / signatures) - +# usually at /.well-known/jwks.json or wired via the OAuth AS metadata's jwks_uri. +curl -fsS $ORIGIN/.well-known/jwks.json -o /tmp/jwks.json -w 'jwks %{http_code} %{content_type}\n' +jq -e '.keys | length > 0' /tmp/jwks.json >/dev/null 2>&1 && echo "jwks valid" || echo "jwks empty/invalid" +jq -r '.keys[] | {kid, kty, crv, alg, use}' /tmp/jwks.json 2>/dev/null + +# DNS-AID (draft-mozleywilliams-dnsop-dnsaid) - emerging, weight-0 bonus. Org index is an +# SVCB record at _index._agents.<domain> (protocol carried in the `alpn` SvcParam, not the label). +# Pure DNS-over-HTTPS; the DoH `AD` flag reports DNSSEC-authenticated data. +BARE=$(printf '%s' "$ORIGIN" | sed -E 's#^https?://##; s#[:/].*##; s/^www\.//') +curl -fsS -H 'accept: application/dns-json' "https://cloudflare-dns.com/dns-query?name=_index._agents.$BARE&type=SVCB" 2>/dev/null \ + | jq -r '"dns-aid index answers=" + ((.Answer // []) | length | tostring) + " AD=" + ((.AD // false)|tostring)' \ + 2>/dev/null || echo "dns-aid index answers=0 AD=false" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | Signature directory exists | HTTP 200 at `/.well-known/http-message-signatures-directory` | 1 | +| 2 | Valid Ed25519 JWKS structure (signing) | `keys[]` non-empty, each has `kty=OKP`, `crv=Ed25519`, `kid` | 1 | +| 3 | A signing key is currently valid (`kid` not expired) | Keys carry `nbf`+`exp` **and** at least one key satisfies `nbf ≤ now ≤ exp` (`signing key currently valid`) | 1 | +| 4 | Documented in a web-fetchable artifact | `GET /llms.txt` OR `GET /docs` (or /docs/security) returns a body containing case-insensitive `web bot auth`, `http message signatures`, or `rfc 9421` | 1 | +| 5 | Generic JWKS endpoint at `/.well-known/jwks.json` valid | HTTP 200 with `application/json`, parseable `keys[]` array (≥ 1 key) - the keys an agent uses to validate issued tokens/signatures | 1 | +| 6 | (Bonus) ≥ 2 signing keys for rotation overlap | `keys.length >= 2` - a published rotation window, never required for Pass | 0 | +| 7 | (Bonus) JWKS carries an encryption key | At least one key has `use=enc` OR `alg` in `ECDH-ES*`/`RSA-OAEP*`/`A256GCM`-family - supports an end-to-end-encryption layer some agentic-commerce stacks add | 0 | +| 8 | (Emerging, bonus) DNS-AID index published | SVCB record at `_index._agents.<domain>` resolves (`dns-aid index answers ≥ 1`); DNSSEC-authenticated (`AD=true`) is a further bonus | 0 | + +Status: **Pass** ≥ 4/5 · **Partial** 2-3/5 · **Fail** 0-1/5 · **N/A** for sites with no bot-distinguishing needs *and* no agentic-commerce surface. (Pass threshold is 80%, an explicit override of the 85% default, so a site can miss one scored row and still Pass.) + +Sub-checks 1-5 are the scored core: 1-4 are the Web Bot Auth signing layer (who is calling), and 5 is the JWKS an agent validates against. Sub-checks 6-8 are weight-0 bonuses - multiple keys (rotation overlap), an encryption key, and DNS-AID - surfaced as "forward-looking" signals but **never** counted toward `weight_total` or the Pass threshold, and never penalised when absent. Rotation is good practice but does not gate Pass. + +## Codebase hints + +- **Cloudflare**: enable Web Bot Auth in dashboard (AI Crawl Control → Verified Bots). Cloudflare publishes the signing directory for you. +- **Fastly**: VCL snippet + custom edge dictionary; publish directory as a static asset. +- **Custom (Node)**: [`http-message-signatures`](https://www.npmjs.com/package/http-message-signatures), generate Ed25519 keypair, serve directory from `/.well-known/`. For JWKS publishing reuse the same OIDC library you use for [m3-1](./m3-1.md) (`node-oidc-provider`, `jose`). +- **Custom (PHP)**: [`http-signatures-php`](https://github.com/99designs/http-signatures-php), publish JWKS via `web-token/jwt-framework`. +- **Custom (Python)**: [`http-message-signatures`](https://pypi.org/project/http-message-signatures/) + `python-jose`. +- **Custom (Go)**: [`go-httpsig`](https://pkg.go.dev/github.com/yaronf/httpsign) + `go-jose`. +- **Key management**: store private keys in HSM / KMS, rotate every 90 days, keep `exp` 90-180 days out for signing keys; overlap windows 7-14 days so signers in flight don't fail mid-roll. +- **Generic JWKS** (sub-check 5): publish the keys an agent validates against at `/.well-known/jwks.json`, or point your OAuth AS metadata's `jwks_uri` at it (reuse the OIDC library from [m3-1](./m3-1.md)). +- **DNS-AID** (emerging, bonus): publish a ServiceMode SVCB record for your org index at `_index._agents.<domain>` per [RFC 9460](https://www.rfc-editor.org/rfc/rfc9460) and the [DNS-AID draft](https://datatracker.ietf.org/doc/draft-mozleywilliams-dnsop-dnsaid/) (per-agent records carry the protocol in `alpn`, not the label), then DNSSEC-sign the zone. Most managed DNS providers (Route 53, Cloudflare, NS1) sign on toggle; if you sign your own zone, stage key rollovers carefully - a broken DS/RRSIG chain takes the whole zone dark, not just the discovery records. + +## Auto-fix template + +```json +// /.well-known/http-message-signatures-directory +{ + "keys": [ + { + "kty": "OKP", + "crv": "Ed25519", + "kid": "agent-key-2026-01", + "x": "<base64url Ed25519 public key>", + "nbf": 1735689600, + "exp": 1751328000 + }, + { + "kty": "OKP", + "crv": "Ed25519", + "kid": "agent-key-2026-04", + "x": "<base64url Ed25519 public key>", + "nbf": 1751328000, + "exp": 1766966400 + } + ] +} +``` + +```json +// /.well-known/jwks.json - signing + encryption keys side-by-side +{ + "keys": [ + { "kty": "OKP", "crv": "Ed25519", "kid": "sig-2026-01", "use": "sig", "alg": "EdDSA", "x": "..." }, + { "kty": "OKP", "crv": "X25519", "kid": "enc-2026-01", "use": "enc", "alg": "ECDH-ES", "x": "..." } + ] +} +``` + +```dns +; DNS-AID (emerging, bonus) - ServiceMode SVCB records under _agents (RFC 9460), then DNSSEC-sign the zone. +; Org index points at an agent-index host; per-agent records carry the protocol in `alpn`. +_index._agents.example.com. 3600 IN SVCB 1 agent-index.example.com. ( alpn="a2a,mcp" ) +mcp._agents.example.com. 3600 IN SVCB 1 mcp.example.com. ( alpn="mcp" port=443 ) +; sign (BIND): dnssec-signzone -S -K keys -o example.com db.example.com +; then upload the resulting DS record to the parent zone via your registrar +``` + +## References + +See: [content/m3-2-web-bot-auth.md](../content/m3-2-web-bot-auth.md) diff --git a/audit/m3-3.md b/audit/m3-3.md new file mode 100644 index 0000000..5e3175d --- /dev/null +++ b/audit/m3-3.md @@ -0,0 +1,127 @@ +--- +id: m3-3 +title: Self-serve credentials +complexity: 3 +impact: 5 +visualChange: medium +weight_total: 9 +--- + +# m3-3 - Self-serve credentials + +## Probe + +```bash +# Signup page reachable +for p in signup sign-up register get-started developers/signup; do + curl -fsSI $ORIGIN/$p -o /dev/null -w "/$p %{http_code}\n" +done + +# Dynamic Client Registration (RFC 7591) endpoint advertised? +jq -r '.registration_endpoint' /tmp/oauth-authorization-server.json 2>/dev/null + +# Docs mention sandbox / test / playground +curl -fsSL $ORIGIN/docs 2>/dev/null | grep -ciE 'sandbox|test (mode|environment)|playground|demo (key|account)' + +# auth.md walkthrough +curl -fsSI $ORIGIN/auth.md -o /dev/null -w 'auth.md %{http_code} %{content_type}\n' +curl -fsS $ORIGIN/auth.md -o /tmp/auth.md +grep -ciE '^#{1,3} *(discover|register|claim|use|errors|revocation)' /tmp/auth.md + +# Self-identifying H1: the file must OPEN with a top-level `# ` heading that +# names it as the auth doc. isitagentready.com's authMd check fails a file whose +# H1 doesn't contain "auth.md" (e.g. "# Forter Agent Authentication" → fail). +FIRST_H=$(grep -m1 -E '^#{1,6} ' /tmp/auth.md) +echo "first_heading: $FIRST_H" +printf '%s' "$FIRST_H" | grep -qE '^# ' && echo "h1_is_atx=yes" || echo "h1_is_atx=no" +printf '%s' "$FIRST_H" | grep -qiE '^# .*auth' && echo "h1_names_auth=yes" || echo "h1_names_auth=no" + +# Structured agent_auth block - machine-readable hook telling an agent how to get credentials. +grep -iE 'agent_auth' /tmp/auth.md +jq -e '.agent_auth // .["agent-auth"]' /tmp/oauth-protected-resource.json >/dev/null 2>&1 && echo "agent_auth in PRM" + +# Public-key model (content step 3): instead of (or alongside) DCR, the agent generates its own +# keypair, registers a JWKS / key directory, and signs every request (RFC 9421). Credit a reachable +# key directory as a no-shared-secret path. +curl -fsS $ORIGIN/.well-known/jwks.json -o /tmp/m33-jwks.json -w 'jwks %{http_code}\n' 2>/dev/null +jq -e '.keys | length > 0' /tmp/m33-jwks.json >/dev/null 2>&1 && echo "pubkey jwks present" || echo "pubkey jwks absent" +curl -fsSI $ORIGIN/.well-known/http-message-signatures-directory -o /dev/null -w 'sig-dir %{http_code}\n' 2>/dev/null +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | Public signup page | At least one of `/signup`, `/register`, `/get-started` returns 200 | 1 | +| 2 | Self-serve key issuance described in a fetched page | `GET /docs` or `/auth.md` returns body containing both `create.{0,15}(api key|access token)` (regex) AND does NOT require `contact sales` on the same page | 1 | +| 3 | Sandbox / test environment | Docs mention sandbox / test mode / playground | 1 | +| 4 | Dynamic Client Registration | `registration_endpoint` present in AS metadata | 1 | +| 5 | `/auth.md` prose walkthrough | HTTP 200, `text/markdown`, ≥ 1 KB | 1 | +| 6 | `/auth.md` has all six sections | Headings for Discover, Register, Claim, Use, Errors, Revocation | 1 | +| 7 | Machine-readable `agent_auth` hook | `/auth.md` contains an `agent_auth` block (fenced JSON/keyed section) **or** the protected-resource metadata exposes an `agent_auth` key - an agent can parse *how to self-register* without reading prose | 1 | +| 8 | `/auth.md` opens with a self-identifying H1 | The file's **first** markdown heading is a top-level `# ` ATX H1 (`h1_is_atx=yes`) whose text contains `auth`, case-insensitive (`h1_names_auth=yes`) - e.g. `# Auth.md - Acme Agent Authentication`. (isitagentready.com's `authMd` check is stricter: it wants the literal substring `auth.md` in the H1, so naming the file in the heading satisfies both.) | 1 | +| 9 | Public-key / keypair auth path | `pubkey jwks present` (`/.well-known/jwks.json` has keys) **or** the signature directory returns 200 - the agent can authenticate by signing requests with its own keypair (no shared secret), the content guideline's "better still" option | 1 | + +Status: **Pass** ≥ 8/9 · **Partial** 3-7/9 · **Fail** 0-2/9 · **N/A** for sites with no API. + +## Codebase hints + +- **Auth provider (Auth0/Clerk/WorkOS/Stytch)**: enable DCR in dashboard; expose API-keys settings page in your app. For `/auth.md`, follow the open [WorkOS auth.md spec](https://github.com/workos/auth.md). +- **Custom**: build `POST /oauth/register` per RFC 7591, returning `client_id` + (optional) `client_secret`. +- **Sandbox**: separate database/environment with seed data; expose `https://sandbox.<domain>` or honor `X-Sandbox: true` header. +- **PHP**: `league/oauth2-server` + custom registration controller. +- **Node**: `node-oidc-provider` supports DCR via `features.registration.enabled = true`. +- **Python**: `authlib` `register_client_id_issuer`. +- **Rails**: `doorkeeper` + `doorkeeper-dynamic-client-registration` gem. + +## Auto-fix template + +```markdown +# Auth.md - <Brand> Agent Authentication + +## Discover +Discovery metadata: $ORIGIN/.well-known/oauth-authorization-server +Protected-resource metadata: $ORIGIN/.well-known/oauth-protected-resource + +## Pick a method +- **API key** - fastest, for server-to-server. Generate at $ORIGIN/dashboard/api-keys. +- **OAuth 2.0 + PKCE** - for agents acting on behalf of a user. Authorize at $ORIGIN/oauth/authorize. +- **Dynamic Client Registration (RFC 7591)** - for agent frameworks that register on the fly. POST to $ORIGIN/oauth/register. + +## Register +POST $ORIGIN/oauth/register +Content-Type: application/json +{ "client_name": "MyAgent", "redirect_uris": ["..."] } +→ 201 { "client_id": "...", "client_secret": "..." } + +## Claim +Exchange auth code for tokens: POST $ORIGIN/oauth/token (grant_type=authorization_code, code_verifier=<PKCE>). + +## Use the credential +Authorization: Bearer <access_token> +Scopes: orders:read, orders:write, ... + +## Errors +401 → re-authenticate. 403 → wrong scope. 429 → honor Retry-After. + +## Revocation +POST $ORIGIN/oauth/revoke with the token. Returns 200 on success. + +## agent_auth +```json +{ + "agent_auth": { + "preferred": "dynamic_client_registration", + "registration_endpoint": "$ORIGIN/oauth/register", + "token_endpoint": "$ORIGIN/oauth/token", + "scopes_supported": ["orders:read", "orders:write"], + "self_serve": true, + "sandbox": "https://sandbox.example.com" + } +} +``` +``` + +## References + +See: [content/m3-3-self-serve-credentials.md](../content/m3-3-self-serve-credentials.md) diff --git a/audit/m4-1.md b/audit/m4-1.md new file mode 100644 index 0000000..8becf3c --- /dev/null +++ b/audit/m4-1.md @@ -0,0 +1,150 @@ +--- +id: m4-1 +title: OpenAPI specification +complexity: 4 +impact: 5 +visualChange: none +weight_total: 9 +--- + +# m4-1 - OpenAPI specification + +## Probe + +```bash +# Common spec paths +for p in openapi.json openapi.yaml openapi/v1.json api/openapi.json api/openapi.yaml spec/openapi.json swagger.json; do + curl -fsSI $ORIGIN/$p -o /dev/null -w "/$p %{http_code} %{content_type}\n" +done + +# RFC 9727 API catalog +curl -fsS $ORIGIN/.well-known/api-catalog -o /tmp/catalog.json -w 'catalog %{http_code} %{content_type}\n' +jq -e '.linkset[0].anchor' /tmp/catalog.json 2>/dev/null + +# Pull whichever JSON spec responded +for p in openapi.json api/openapi.json openapi/v1.json spec/openapi.json swagger.json; do + curl -fsS -o /tmp/spec -w '%{http_code}\n' $ORIGIN/$p 2>/dev/null | grep -q '^200' && break +done + +# Spectral-free lint: depth checks an agent (or a function-calling layer) actually needs. +python3 - <<'PY' +import json +try: + s=json.load(open("/tmp/spec")) +except Exception: + print("spec_parse FAILED (not JSON - if YAML, lint checks 5/7/8 are limited)"); raise SystemExit +paths=s.get("paths",{}) or {} +ops=[] +for p,item in paths.items(): + if not isinstance(item,dict): continue + for m,op in item.items(): + if m.lower() in ("get","post","put","patch","delete") and isinstance(op,dict): + ops.append((p,m,op)) +n=len(ops) +oids=[op.get("operationId") for _,_,op in ops] +oids_present=all(oids) and len(set(oids))==len(oids) # present AND unique +described=sum(1 for _,_,op in ops if op.get("description") or op.get("summary")) +with_resp=sum(1 for _,_,op in ops if op.get("responses")) +# function-calling compat: params + requestBody fields carry type+description +def param_ok(op): + for prm in op.get("parameters",[]) or []: + if isinstance(prm,dict) and prm.get("description") and (prm.get("schema",{}) or {}).get("type"): + return True + return "parameters" not in op # no params is fine +fc_ok=sum(1 for _,_,op in ops if param_ok(op)) +comps=(s.get("components",{}) or {}).get("schemas",{}) or {} +err_schema=any(k for k in comps if "error" in k.lower() or "problem" in k.lower()) +servers=[sv.get("url","") for sv in s.get("servers",[]) or []] +abs_servers=bool(servers) and all(u.startswith("http") for u in servers) +print("openapi", s.get("openapi")) +print("operations", n) +print("operationId_present_unique", oids_present) +print("all_described", n>0 and described==n) +print("all_have_responses", n>0 and with_resp==n) +print("function_calling_ready", n>0 and fc_ok==n) +print("error_schema", err_schema) +print("servers_absolute", abs_servers, servers[:3]) +PY + +# Agent-friendly view negotiation (content step 6): Accept: text/markdown → markdown + Vary: Accept, +# or a ?mode=agent fallback for clients that can't set the header. +curl -fsS -H 'Accept: text/markdown' -D /tmp/m41md.h -o /dev/null "$ORIGIN/" -w 'md %{http_code} ct=%{content_type}\n' 2>/dev/null +grep -iqE '^content-type:.*text/markdown' /tmp/m41md.h && echo "md-served yes" || echo "md-served no" +grep -iqE '^vary:.*accept' /tmp/m41md.h && echo "vary-accept yes" || echo "vary-accept no" +MA_CT=$(curl -fsS -o /dev/null -w '%{content_type}' "$ORIGIN/?mode=agent" 2>/dev/null); echo "mode-agent ct=$MA_CT" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | OpenAPI doc reachable | Any common path returns 200 with JSON/YAML | 1 | +| 2 | OpenAPI 3.x | `openapi` field starts with `3.` | 1 | +| 3 | `/.well-known/api-catalog` (RFC 9727) | HTTP 200, `linkset[0].anchor` present | 1 | +| 4 | securitySchemes declared | Spec has `components.securitySchemes` (OAuth/bearer/API key) | 1 | +| 5 | Shared Error/Problem schema | `error_schema` true - `components.schemas` has an `Error`/`Problem` shape | 1 | +| 6 | Pagination documented | Operations have `page` / `cursor` / `limit` parameters | 1 | +| 7 | Every operation is agent-legible | `operationId_present_unique` **and** `all_described` **and** `all_have_responses` (no spectral needed - derived from the parsed spec) | 1 | +| 8 | Function-calling ready | `function_calling_ready` true (params/body carry `type` + `description`, so the spec drops cleanly into a tool-calling layer) **and** `servers_absolute` true | 1 | +| 9 | Agent-friendly view negotiation | `Accept: text/markdown` returns `content-type: text/markdown` **with** `Vary: Accept` (`md-served yes` **and** `vary-accept yes`), **or** `?mode=agent` returns a non-`text/html` agent view - content step 6 | 1 | + +Status: **Pass** ≥ 8/9 · **Partial** 3-7/9 · **Fail** 0-2/9 · **N/A** for sites with no API. + +> Sub-checks 7-8 replace the old "lint with Spectral / manual" row with conditions computed directly from the fetched JSON spec - no extra tooling. (A YAML-only spec limits 5/7/8; note that in the evidence and treat the unparseable checks as Fail per the strict-scoring rule.) + +## Codebase hints + +- **Node (Express/Fastify/Nest)**: `swagger-jsdoc` (annotation-driven) or `@nestjs/swagger`. Serve via `/openapi.json` route. +- **Python (FastAPI)**: free - `/openapi.json` is built-in. Add `tags`, `summary`, `responses` per endpoint. +- **Python (Flask)**: `flask-smorest` / `apispec`. +- **Ruby (Rails)**: `rswag` or `apipie-rails`. +- **PHP (Laravel)**: `darkaonline/l5-swagger` or `openapi-php/openapi-generator`. PHP custom: `zircote/swagger-php` annotations. +- **Go**: `swaggo/swag` annotations. +- **Java/Kotlin (Spring)**: `springdoc-openapi`. +- **Lint with**: `npx @stoplight/spectral-cli lint openapi.json` (Spectral) before publishing. + +## Auto-fix template + +```json +// /.well-known/api-catalog (RFC 9727) +{ + "linkset": [ + { + "anchor": "$ORIGIN", + "service-desc": [ + { "href": "$ORIGIN/openapi.json", "type": "application/openapi+json" }, + { "href": "$ORIGIN/openapi.yaml", "type": "application/openapi+yaml" } + ], + "service-doc": [{ "href": "$ORIGIN/docs/api" }], + "status": [{ "href": "$ORIGIN/status" }] + } + ] +} +``` + +```yaml +# Skeleton openapi.yaml - replace with annotations from your framework +openapi: 3.1.0 +info: { title: <Brand> API, version: "1.0.0" } +servers: [{ url: $ORIGIN/v1 }] +components: + securitySchemes: + oauth2: + type: oauth2 + flows: { authorizationCode: { authorizationUrl: $ORIGIN/oauth/authorize, tokenUrl: $ORIGIN/oauth/token, scopes: { "orders:read": "", "orders:write": "" } } } + schemas: + Error: + type: object + required: [type, message, request_id] + properties: + type: { type: string, example: "rate_limited" } + message: { type: string } + request_id: { type: string } + retry_hint: { type: integer, description: "seconds" } +security: [{ oauth2: [orders:read] }] +paths: {} +``` + +## References + +See: [content/m4-1-openapi-spec.md](../content/m4-1-openapi-spec.md) diff --git a/audit/m4-10.md b/audit/m4-10.md new file mode 100644 index 0000000..be0216c --- /dev/null +++ b/audit/m4-10.md @@ -0,0 +1,54 @@ +--- +id: m4-10 +title: NLWeb (/ask + Schema Feeds) +complexity: 3 +impact: 3 +visualChange: none +weight_total: 5 +--- + +# m4-10 - NLWeb endpoint + +## Probe + +```bash +# Schemamap directive in robots.txt +grep -iE '^schemamap:' /tmp/robots.txt + +# Schema map manifest +curl -fsSI $ORIGIN/.well-known/schema-map.xml -o /dev/null -w 'schema-map %{http_code}\n' + +# /ask endpoint (NLWeb) +curl -fsS -X POST $ORIGIN/ask \ + -H 'Content-Type: application/json' \ + --data '{"query":"what do you sell?","stream":false}' \ + -o /tmp/ask.json -w '/ask POST %{http_code} %{content_type}\n' +jq -r '. | {has_results: (.results // [] | length > 0), meta: ._meta}' /tmp/ask.json 2>/dev/null + +# SSE streaming probe +curl -fsSI -H 'Accept: text/event-stream' -X POST $ORIGIN/ask \ + -o /dev/null -w '/ask SSE %{http_code} %{content_type}\n' +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | `Schemamap:` directive in `robots.txt` | Line present pointing to schema-map | 1 | +| 2 | Schema map manifest | `/.well-known/schema-map.xml` (or `.json`) returns 200 | 1 | +| 3 | `/ask` endpoint exists | POST returns 2xx with JSON body | 1 | +| 4 | Response has `_meta` (`response_type`, `version`) | Both keys present | 1 | +| 5 | SSE streaming supported on `/ask` | `Accept: text/event-stream` returns `text/event-stream` | 1 | + +Status: **Pass** ≥ 4/5 · **Partial** 2-3/5 · **Fail** 0-1/5 · **N/A** for sites with no Q&A surface. + +## Codebase hints + +- **Reference impl**: [`microsoft/NLWeb`](https://github.com/microsoft/NLWeb) - Python + JSON-LD ingest, /ask served via FastAPI. +- **Schema feeds**: emit your existing Schema.org JSON-LD as a JSONL/RSS feed (`feeds/products.jsonl`, `feeds/articles.rss`); list them in `schema-map.xml`. +- **/ask answer engine**: wire to your existing search/RAG (Algolia, Pinecone, Weaviate, local pgvector). Stream tokens as SSE `event: result`. +- **MCP parity**: expose the same `ask` capability as an MCP tool (see [m4-4](./m4-4.md)) - many agents prefer MCP over NLWeb. + +## References + +See: [content/m4-10-nlweb.md](../content/m4-10-nlweb.md) diff --git a/audit/m4-2.md b/audit/m4-2.md new file mode 100644 index 0000000..ec307e2 --- /dev/null +++ b/audit/m4-2.md @@ -0,0 +1,88 @@ +--- +id: m4-2 +title: Rate limits & errors +complexity: 2 +impact: 5 +visualChange: none +weight_total: 7 +--- + +# m4-2 - Rate limits & errors + +## Probe + +```bash +# Pick a known API endpoint or fall back to /api +EP=$ORIGIN/api/v1/ping +curl -fsS -D /tmp/h.txt -o /tmp/b.json $EP 2>/dev/null +grep -iE '^(x-)?ratelimit-(limit|remaining|reset)|^retry-after' /tmp/h.txt + +# Versioning / deprecation signalling (RFC 8594 Sunset, Deprecation header) +grep -iE '^(sunset|deprecation):' /tmp/h.txt + +# Trigger an error +curl -fsS -D /tmp/eh.txt -o /tmp/eb.json $ORIGIN/api/v1/this-does-not-exist-$RANDOM +cat /tmp/eb.json | jq -r '{type, message, request_id, retry_hint}' 2>/dev/null + +# Status endpoint +curl -fsS -H 'Accept: application/json' $ORIGIN/status -o /tmp/st.json -w 'status %{http_code} %{content_type}\n' +jq -r '{status, last_updated, incidents: (.incidents // [] | length)}' /tmp/st.json 2>/dev/null +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | RateLimit headers on success | At least one of `RateLimit-Limit`/`X-RateLimit-Limit`/`RateLimit-Remaining` on a 2xx response | 1 | +| 2 | Reset header in epoch seconds or seconds-to-reset | `RateLimit-Reset` or `X-RateLimit-Reset` present and numeric | 1 | +| 3 | `Retry-After` on 429 | When forced to a 429 (or documented) `Retry-After` is set | 1 | +| 4 | Structured JSON errors | Error body parses as JSON with `type` and `message` (and ideally `request_id`) | 1 | +| 5 | Errors include `retry_hint` for transient classes | Rate-limit / server errors include `retry_hint` field | 1 | +| 6 | `/status` JSON endpoint | Returns JSON with `status` + `last_updated` (+ `incidents` array) | 1 | +| 7 | Versioning / deprecation signalled | Versioned path (e.g. `/v1/`) **or** `Sunset`/`Deprecation` header (RFC 8594) present - an agent can tell when an endpoint is going away | 1 | + +Status: **Pass** ≥ 6/7 · **Partial** 3-5/7 · **Fail** 0-2/7 · **N/A** for sites with no API. + +## Codebase hints + +- **Express / Fastify**: `express-rate-limit`, `@fastify/rate-limit`; both emit `RateLimit-*` headers per RFC 9598. +- **Nest**: `@nestjs/throttler`. +- **FastAPI**: `slowapi`. +- **Django**: `django-ratelimit` + custom middleware to emit headers. +- **Flask**: `Flask-Limiter`. +- **Rails**: `rack-attack`. +- **PHP (Laravel)**: built-in `throttle` middleware emits headers. +- **PHP (custom)**: middleware sets `RateLimit-Limit`/`Remaining`/`Reset`; central error handler returns `application/problem+json` body. +- **Cloudflare**: configure Rate Limiting rules; map response action to a JSON response template with structured fields. +- **Status page**: StatusPage.io / Better Stack / Instatus expose `/api/v2/summary.json`; proxy or alias to `/status`. + +## Auto-fix template + +```json +// Error response body (RFC 7807-ish + agent-friendly fields) +{ + "type": "rate_limited", + "message": "You have exceeded 60 requests/minute on orders:write.", + "request_id": "req_01HZ7Q2W…", + "retry_hint": 12 +} +``` + +```text +// Success response headers +HTTP/2 200 +ratelimit-limit: 60 +ratelimit-remaining: 47 +ratelimit-reset: 34 +``` + +```text +// Deprecation signalling on a sunsetting endpoint (RFC 8594) +Deprecation: true +Sunset: Sat, 31 Oct 2026 23:59:59 GMT +Link: <https://example.com/docs/migrate-v2>; rel="deprecation" +``` + +## References + +See: [content/m4-2-rate-limits-and-errors.md](../content/m4-2-rate-limits-and-errors.md) diff --git a/audit/m4-3.md b/audit/m4-3.md new file mode 100644 index 0000000..145e1f1 --- /dev/null +++ b/audit/m4-3.md @@ -0,0 +1,67 @@ +--- +id: m4-3 +title: Streaming +complexity: 5 +impact: 4 +visualChange: none +weight_total: 7 +--- + +# m4-3 - Streaming long-running operations + +## Probe + +```bash +# Inspect openapi for x-streaming / text/event-stream / application/x-ndjson +jq -r ' + ([.. | objects | select(.["content"]? != null) | .content | keys[]] | unique) + as $ct | + { streaming_content_types: ($ct | map(select(test("event-stream|x-ndjson|jsonl"))))} +' /tmp/spec 2>/dev/null + +# Webhooks documented +jq -r 'has("webhooks") or any(.paths[]?; .post?.callbacks)' /tmp/spec 2>/dev/null + +# Async job pattern: any operation returning 202, plus a poll/status path +jq -r '"has_202=" + ([.paths[]?[]? | objects | select(.responses?["202"]?)] | length > 0 | tostring)' /tmp/spec 2>/dev/null +jq -r '"poll_paths=" + ([.paths | keys[]? | select(test("(job|task|operation)s?/|/status|/poll"))] | tostring)' /tmp/spec 2>/dev/null + +# SSE handshake on a known endpoint (if discoverable) +for p in events stream notifications; do + curl -fsSI -H 'Accept: text/event-stream' $ORIGIN/api/v1/$p -o /dev/null -w "/$p %{http_code} %{content_type}\n" +done + +# Cancellation contract (content step 4): can an agent abort a long-running stream/job? +CANCEL=no +grep -iqE 'event: *cancel|cancelled|/cancel|abort the (stream|job|run|operation)' /tmp/spec 2>/dev/null && CANCEL=yes +jq -e '[.paths | to_entries[] | select(.key|test("(job|task|operation|run)s?/";"i")) | .value | has("delete")] | any' /tmp/spec >/dev/null 2>&1 && CANCEL=yes +echo "cancel_documented $CANCEL" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | At least one streaming operation declared | OpenAPI declares `text/event-stream` or `application/x-ndjson` content type | 1 | +| 2 | SSE endpoint responds with `text/event-stream` | Probe returns 200 with correct content-type | 1 | +| 3 | Webhooks documented in a fetched artifact | (Fetched) OpenAPI declares a `webhooks:` block OR any path has `callbacks:`, OR `GET /docs/webhooks` returns 200 with ≥ 1 KB body | 1 | +| 4 | Webhook signatures (HMAC) documented in a fetched artifact | OpenAPI spec OR `/docs` body (whichever URL responded earlier) contains case-insensitive match of `X-Signature`, `Webhook-Signature`, `HMAC`, or `RFC 9421` | 1 | +| 5 | Idempotency-Key header documented in a fetched artifact | OpenAPI spec OR docs body contains literal `Idempotency-Key` or `X-Idempotency-Key` | 1 | +| 6 | Async-job pattern for long ops | Spec has an operation returning `202` **and** a job/task/status poll path (e.g. `/jobs/{id}` or `…/status`) - agents can kick off and poll instead of holding a connection | 1 | +| 7 | Cancellation contract documented | `cancel_documented yes` - the spec/docs define how to abort a long-running stream or job (`event: cancelled`, a `/cancel` path, or `DELETE /jobs/{id}`), so an agent can stop work it no longer needs | 1 | + +Status: **Pass** ≥ 6/7 · **Partial** 2-5/7 · **Fail** 0-1/7 · **N/A** for sites with no long-running ops. + +## Codebase hints + +- **Node**: write SSE with `res.write('event: progress\ndata: {...}\n\n')` or use [`sse-express`](https://github.com/dpskvn/express-sse); webhooks via [`svix`](https://www.svix.com). +- **Python**: FastAPI `StreamingResponse(generator, media_type='text/event-stream')`. +- **Ruby (Rails)**: `ActionController::Live`. +- **PHP**: long-poll via flush + `Content-Type: text/event-stream`; for production use, a dedicated SSE server (e.g., Mercure) is more reliable than PHP-FPM. +- **Go**: built-in via `http.Flusher`. +- **Webhook signatures**: HMAC-SHA256 over body, header `Webhook-Signature: t=<ts>,v1=<hex>` (Svix/Stripe convention). +- **Idempotency**: store `Idempotency-Key` → response in Redis with 24h TTL. + +## References + +See: [content/m4-3-streaming.md](../content/m4-3-streaming.md) diff --git a/audit/m4-4.md b/audit/m4-4.md new file mode 100644 index 0000000..52aea99 --- /dev/null +++ b/audit/m4-4.md @@ -0,0 +1,105 @@ +--- +id: m4-4 +title: MCP server +complexity: 4 +impact: 5 +visualChange: none +weight_total: 9 +--- + +# m4-4 - MCP server + +## Probe + +```bash +# Discovery +curl -fsS $ORIGIN/.well-known/mcp.json -o /tmp/mcp1.json -w 'mcp.json %{http_code}\n' +curl -fsS $ORIGIN/.well-known/mcp/server-card.json -o /tmp/mcp2.json -w 'server-card %{http_code}\n' + +# Common transport paths +for p in mcp mcp/sse mcp/stream api/mcp; do + curl -fsSI $ORIGIN/$p -o /dev/null -w "/$p %{http_code} %{content_type}\n" +done + +# Try MCP initialize on /mcp (Streamable HTTP). Capture the session id header if any. +curl -fsS -X POST $ORIGIN/mcp \ + -H 'Content-Type: application/json' \ + -H 'Accept: application/json, text/event-stream' \ + --data '{"jsonrpc":"2.0","id":1,"method":"initialize","params":{"protocolVersion":"2025-06-18","capabilities":{},"clientInfo":{"name":"audit","version":"1.0"}}}' \ + -D /tmp/init.h -o /tmp/init.json -w 'initialize %{http_code} %{content_type}\n' +SID=$(grep -i '^mcp-session-id:' /tmp/init.h | sed -E 's/^[^:]+:[[:space:]]*//; s/\r//') +jq -r '.result | {protocolVersion, serverInfo, capabilities}' /tmp/init.json 2>/dev/null + +# tools/list - pass the session id if the server issued one. SSE replies come back as data: lines. +curl -fsS -X POST $ORIGIN/mcp \ + -H 'Content-Type: application/json' -H 'Accept: application/json, text/event-stream' \ + ${SID:+-H "Mcp-Session-Id: $SID"} \ + --data '{"jsonrpc":"2.0","id":2,"method":"tools/list","params":{}}' \ + -o /tmp/tools.raw -w 'tools/list %{http_code}\n' +# normalise SSE → JSON, then assess tool quality +sed -n 's/^data: //p' /tmp/tools.raw > /tmp/tools.json 2>/dev/null; [ -s /tmp/tools.json ] || cp /tmp/tools.raw /tmp/tools.json +python3 - <<'PY' +import json +try: d=json.load(open("/tmp/tools.json")) +except Exception: print("tools_parse FAILED"); raise SystemExit +tools=(d.get("result",{}) or {}).get("tools",[]) or [] +print("tool_count", len(tools)) +def good(t): + s=t.get("inputSchema",{}) or {} + return bool(t.get("name")) and len((t.get("description") or ""))>=20 \ + and s.get("type")=="object" and isinstance(s.get("properties"),dict) +print("tools_well_described", sum(good(t) for t in tools), "of", len(tools)) +print("has_annotations", any((t.get("annotations") or {}) for t in tools)) +PY +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | MCP discovery file | `mcp.json` OR `mcp/server-card.json` returns 200 with valid JSON | 1 | +| 2 | Server card has `name` + `serverUrl` + `transport` | All three present | 1 | +| 3 | MCP endpoint reachable | A common transport path responds 200 (or 405 on GET, accepting POST) | 1 | +| 4 | `initialize` handshake succeeds | POST returns JSON-RPC `result` with `serverInfo` | 1 | +| 5 | Tools exposed (list_tools works) | `tools/list` returns ≥ 1 tool with `name` + `inputSchema` (`tool_count ≥ 1`) | 1 | +| 6 | OAuth required on tools | `authorization` field on server card OR 401 with `WWW-Authenticate` on protected tool | 1 | +| 7 | Tool annotations (readOnlyHint / destructiveHint) | `has_annotations` true - at least one tool declares `annotations` | 1 | +| 8 | Modern transport (Streamable HTTP) | `initialize` succeeded over `POST /mcp` (Streamable HTTP) **and** negotiated `protocolVersion` ≥ `2025-03-26`; a server-card `transport` of legacy `sse`-only with no Streamable HTTP fails this | 1 | +| 9 | Tools are well-described | Every listed tool has a `description` ≥ 20 chars **and** an `inputSchema` of `type: object` with `properties` (`tools_well_described` == `tool_count`) - agents can pick and call tools without guessing | 1 | + +Status: **Pass** ≥ 8/9 · **Partial** 3-7/9 · **Fail** 0-2/9 · **N/A** for sites with no API. + +> Sub-checks 8-9 follow Anthropic's MCP best-practice tightening - mirroring the shift the Ora ranker made in its v1.1 scoring: a server that merely *exists* but speaks only deprecated HTTP+SSE, or exposes terse unschematized tools, is not actually usable by current agent clients. Score tool quality, not tool presence. + +## Codebase hints + +- **TypeScript / Node**: [`@modelcontextprotocol/sdk`](https://www.npmjs.com/package/@modelcontextprotocol/sdk) `Server` + `StreamableHTTPServerTransport`. +- **Python**: `mcp` SDK (`pip install mcp`). +- **Generate tools from OpenAPI**: [`mcpify`](https://github.com/mcpify/mcpify) or hand-map operations to MCP tools. +- **Cloudflare Workers**: [`workers-mcp`](https://github.com/cloudflare/workers-mcp) generates a server from a Worker class. +- **OAuth wiring**: front MCP with the same `/.well-known/oauth-protected-resource` from [m3-1](./m3-1.md); per-tool scopes. +- **Hosting**: typically a separate route on the same origin (`/mcp`), not a separate subdomain - keeps OAuth simple. + +## Auto-fix template + +```json +// /.well-known/mcp/server-card.json +{ + "name": "<Brand> MCP", + "description": "Tools to <core capability>.", + "version": "1.0.0", + "serverUrl": "$ORIGIN/mcp", + "transport": "streamable-http", + "authorization": { + "type": "oauth2", + "metadata": "$ORIGIN/.well-known/oauth-protected-resource" + }, + "tools": [ + { "name": "<tool_name>", "description": "<what it does>", "annotations": { "readOnlyHint": true } } + ] +} +``` + +## References + +See: [content/m4-4-mcp-server.md](../content/m4-4-mcp-server.md) diff --git a/audit/m4-5.md b/audit/m4-5.md new file mode 100644 index 0000000..ee4789a --- /dev/null +++ b/audit/m4-5.md @@ -0,0 +1,85 @@ +--- +id: m4-5 +title: WebMCP +complexity: 2 +impact: 3 +visualChange: none +weight_total: 4 +--- + +# m4-5 - WebMCP + +## Probe + +```bash +# Look for navigator.modelContext usage in homepage JS +curl -fsSL $ORIGIN/ -o /tmp/home.html + +# Inline script content (spec uses document.modelContext; polyfill/earlier Chrome uses navigator.modelContext) +grep -oE '(document|navigator)\.modelContext[^"]{0,200}' /tmp/home.html | head -5 + +# Extract first few same-origin <script src> and grep them +python3 - <<'PY' +import re,sys,urllib.request,os +h=open("/tmp/home.html").read() +srcs=re.findall(r'(?is)<script[^>]+src=["\']([^"\']+)["\']', h) +origin=os.environ["ORIGIN"] +for s in srcs[:10]: + if s.startswith("http") and not s.startswith(origin): continue + url = s if s.startswith("http") else origin.rstrip("/") + "/" + s.lstrip("/") + try: + b=urllib.request.urlopen(url, timeout=5).read().decode(errors="ignore") + if "modelContext" in b or "provideContext" in b or "registerTool" in b: + print("WebMCP found in", url) + except Exception: pass +PY +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | Any `modelContext` reference | `document.modelContext` or `navigator.modelContext` found in homepage HTML or first 10 same-origin scripts | 1 | +| 2 | Tool registration call | `registerTool(` (current spec) or `provideContext(` (polyfill/legacy bulk form) present | 1 | +| 3 | Tools have `inputSchema` | Source contains `inputSchema` next to tool name | 1 | +| 4 | Tools clean up on navigation | `AbortSignal` / `signal:` passed to registration | 1 | + +Status: **Pass** ≥ 3/4 · **Partial** 1-2/4 · **Fail** 0/4 · **N/A** for sites with no client-side tool actions to expose. + +## Codebase hints + +- **Any client-side**: register tools on page load behind a feature detect. The current W3C spec uses `document.modelContext.registerTool(tool, { signal })`; earlier Chrome builds / the MCP-B polyfill expose `navigator.modelContext` (with an older `provideContext({ tools })` bulk form). Detect both: `const mc = document.modelContext || navigator.modelContext;`. Today only Chrome supports it. +- **React**: hook (`useWebMCPTool({ name, inputSchema, execute })`) that registers on mount and aborts on unmount. +- **Vue/Svelte**: same pattern via `onMounted` / `$effect` with an `AbortController`. +- **PHP-rendered pages**: include the registration `<script>` in the page template; reference page-specific tools by URL param. + +## Auto-fix template + +```html +<script> +// Spec exposes document.modelContext; polyfill/earlier Chrome use navigator.modelContext. +const mc = (typeof document !== 'undefined' && document.modelContext) || navigator.modelContext; +if (mc) { + const ac = new AbortController(); + mc.registerTool({ + name: 'add_to_cart', + description: 'Add an item to the visitor\'s shopping cart.', + inputSchema: { + type: 'object', + properties: { sku: { type: 'string' }, quantity: { type: 'integer', minimum: 1 } }, + required: ['sku'] + }, + annotations: { readOnlyHint: false, destructiveHint: false }, + async execute({ sku, quantity }) { + const r = await fetch('/api/cart', { method: 'POST', body: JSON.stringify({ sku, quantity }) }); + return { content: [{ type: 'text', text: r.ok ? 'Added.' : 'Failed: ' + r.status }] }; + } + }, { signal: ac.signal }); // AbortSignal unregisters the tool on navigation + addEventListener('pagehide', () => ac.abort()); +} +</script> +``` + +## References + +See: [content/m4-5-webmcp.md](../content/m4-5-webmcp.md) diff --git a/audit/m4-6.md b/audit/m4-6.md new file mode 100644 index 0000000..b183a23 --- /dev/null +++ b/audit/m4-6.md @@ -0,0 +1,114 @@ +--- +id: m4-6 +title: Agent registries +complexity: 1 +impact: 3 +visualChange: low +weight_total: 6 +--- + +# m4-6 - Agent registries + +## Probe + +```bash +BRAND=$(printf '%s' "$HOST" | sed -E 's/^www\.//; s/\..*$//') + +# skills.sh - community + official +curl -fsS "https://skills.sh/api/skills?q=$BRAND" -o /tmp/sk.json -w 'skills.sh %{http_code}\n' +jq -r '.skills // [] | map(select(.official == true)) | length' /tmp/sk.json 2>/dev/null + +# mcp.run / smithery.ai / mcphub.io (registries shift - probe with site:domain search) +for r in mcp.run mcphub.io smithery.ai; do + curl -fsSI "https://$r/search?q=$BRAND" -o /dev/null -w "$r %{http_code}\n" +done + +# Self-published Agent Skills index (well-known, v0.2.0) - the site's own skill registry +curl -fsS $ORIGIN/.well-known/agent-skills/index.json -o /tmp/aski.json -w 'agent-skills %{http_code}\n' +python3 - <<'PY' +import json,os +try: d=json.load(open("/tmp/aski.json")) +except Exception: print("agent_skills_index invalid/absent"); raise SystemExit +sk=d if isinstance(d,list) else (d.get("skills") or []) +ok=all(isinstance(s,dict) and s.get("name") and (s.get("path") or s.get("url") or s.get("skill")) for s in sk) and len(sk)>=1 +print("agent_skills_count", len(sk), "schema_ok", ok) +# emit first skill ref so the orchestrator can resolve it +if sk: + ref=sk[0].get("path") or sk[0].get("url") or sk[0].get("skill") + print("first_skill_ref", ref) +PY + +# Own repo signals +curl -fsSI $ORIGIN/SKILL.md -o /dev/null -w 'SKILL.md %{http_code}\n' +curl -fsSI $ORIGIN/smithery.yaml -o /dev/null -w 'smithery.yaml %{http_code}\n' +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | Listed on `skills.sh` (official) | API returns ≥ 1 result where `official == true` | 1 | +| 2 | Listed on `mcp.run` or `mcphub.io` | Either registry has a result page for the brand | 1 | +| 3 | Listed on `smithery.ai` | Search page returns 200 with the brand name | 1 | +| 4 | `SKILL.md` web-discoverable | `GET $ORIGIN/SKILL.md` returns 200, OR a `github.com/<org>/<repo>` URL appears in homepage HTML AND `GET https://raw.githubusercontent.com/<org>/<repo>/HEAD/SKILL.md` returns 200 | 1 | +| 5 | `smithery.yaml` web-discoverable | Same pattern: 200 at `$ORIGIN/smithery.yaml`, OR fetched from the homepage-linked GitHub repo's HEAD, AND body parses as YAML with `runtime:` and `start:` keys | 1 | +| 6 | Agent Skills index published | `/.well-known/agent-skills/index.json` returns 200 with `schema_ok` true (≥ 1 skill, each with a `name` and a resolvable `path`/`url`) | 1 | + +Status: **Pass** ≥ 5/6 · **Partial** 2-4/6 · **Fail** 0-1/6. + +## Codebase hints + +- **`SKILL.md`**: at repo root with YAML frontmatter (`name`, `description`). Register with `npx skills add`. +- **`smithery.yaml`**: at repo root. Specifies `runtime` (`node`/`python`/`docker`), `start` command, `env` variables. +- **mcp.run / mcphub.io**: web-form submission per registry; no codebase change. +- **Agent Skills index**: publish `/.well-known/agent-skills/index.json` (v0.2.0) listing each skill with `name`, `description`, and a `path` to its `SKILL.md`. Static JSON - drop in `.well-known/`. +- **Cross-link**: list registry pages in `/llms.txt` so agents can discover them by reading you. + +## Auto-fix template + +```json +// /.well-known/agent-skills/index.json (Agent Skills v0.2.0) +{ + "version": "0.2.0", + "skills": [ + { "name": "<brand>-mcp", "description": "<one-sentence what it does>", "path": "/SKILL.md" } + ] +} +``` + +```yaml +# /smithery.yaml (repo root) +runtime: node +language: typescript +build: + command: npm run build +start: + command: node dist/server.js +env: + - OAUTH_CLIENT_ID + - OAUTH_CLIENT_SECRET +``` + +```markdown +# /SKILL.md (repo root) +--- +name: <brand>-mcp +description: <One-sentence what the MCP server does.> +--- + +# <Brand> MCP + +Tools to <core capability>. Auth via OAuth at $ORIGIN/oauth/authorize. + +## Tools +- `<tool_name>` - <description>. + +## Setup +1. `npm install -g @<brand>/mcp` (or `npx <brand>-mcp`). +2. `<brand> login` to OAuth. +3. Add to Claude Desktop / Cursor config as `mcp.<brand>`. +``` + +## References + +See: [content/m4-6-agent-registries.md](../content/m4-6-agent-registries.md) diff --git a/audit/m4-7.md b/audit/m4-7.md new file mode 100644 index 0000000..fc89f1a --- /dev/null +++ b/audit/m4-7.md @@ -0,0 +1,71 @@ +--- +id: m4-7 +title: SDKs & CLI +complexity: 3 +impact: 4 +visualChange: none +weight_total: 5 +--- + +# m4-7 - SDKs & CLI + +## Probe + +```bash +BRAND=$(printf '%s' "$HOST" | sed -E 's/^www\.//; s/\..*$//') + +# npm - try bare, scoped, and -sdk/-js variants; record the first that resolves + whether it ships a bin (CLI) +NPM_HIT=""; NPM_BIN=no +for n in "$BRAND" "@$BRAND/$BRAND" "@$BRAND/sdk" "@$BRAND/node" "$BRAND-sdk" "$BRAND-js" "$BRAND-node"; do + enc=$(printf '%s' "$n" | sed 's|/|%2F|') + if curl -fsS "https://registry.npmjs.org/$enc" -o /tmp/npm.json 2>/dev/null && jq -e '.["dist-tags"].latest' /tmp/npm.json >/dev/null 2>&1; then + NPM_HIT="$n"; jq -e '.versions[.["dist-tags"].latest].bin' /tmp/npm.json >/dev/null 2>&1 && NPM_BIN=yes + break + fi +done +echo "npm=$NPM_HIT latest=$(jq -r '.["dist-tags"].latest // empty' /tmp/npm.json 2>/dev/null) bin=$NPM_BIN" + +# PyPI - bare, -sdk, -python, -client +PYPI_HIT="" +for n in "$BRAND" "$BRAND-sdk" "$BRAND-python" "$BRAND-client" "${BRAND}sdk"; do + if curl -fsS "https://pypi.org/pypi/$n/json" -o /tmp/pypi.json 2>/dev/null && jq -e '.info.version' /tmp/pypi.json >/dev/null 2>&1; then + PYPI_HIT="$n"; break + fi +done +echo "pypi=$PYPI_HIT version=$(jq -r '.info.version // empty' /tmp/pypi.json 2>/dev/null)" + +# Other languages: Go, RubyGems, Packagist +curl -fsSI "https://pkg.go.dev/github.com/$BRAND/$BRAND-go" -o /dev/null -w 'go %{http_code}\n' +curl -fsS "https://rubygems.org/api/v1/gems/$BRAND.json" -o /dev/null -w 'rubygems %{http_code}\n' 2>/dev/null +curl -fsS "https://repo.packagist.org/p2/$BRAND/$BRAND.json" -o /dev/null -w 'packagist %{http_code}\n' 2>/dev/null + +# Homebrew tap +curl -fsSI "https://github.com/$BRAND/homebrew-$BRAND" -o /dev/null -w 'brew tap %{http_code}\n' + +# CLI hint in docs +curl -fsSL $ORIGIN/docs 2>/dev/null | grep -ciE "\b$BRAND (login|init|deploy|run|exec)\b" | head -1 +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | npm package | `NPM_HIT` non-empty - bare, scoped (`@brand/sdk`), or `<brand>-js` variant returns a latest version | 1 | +| 2 | PyPI package | `PYPI_HIT` non-empty - bare or `<brand>-sdk`/`-python` variant returns a version | 1 | +| 3 | At least one other language (Go / Ruby / PHP / Java) SDK | Reachable on a language-native registry (pkg.go.dev / RubyGems / Packagist) | 1 | +| 4 | CLI installable (Homebrew / npm bin / scoop) | Homebrew tap exists OR `NPM_BIN=yes` (npm package declares `bin`) | 1 | +| 5 | Docs show idiomatic SDK usage | `GET /docs` (or whichever doc URL responded for m2-3) body contains at least one regex match of `client\.[a-z_]+\.(create|list|retrieve|update|delete)\(` OR `[A-Z][a-z]+::client` (Ruby-style) - proves the page actually demos SDK syntax, not just lists endpoints | 1 | + +Status: **Pass** ≥ 4/5 · **Partial** 2-3/5 · **Fail** 0-1/5 · **N/A** for sites with no API. + +## Codebase hints + +- **Generate from OpenAPI**: `openapi-generator-cli`, `Speakeasy`, `Stainless`, `Fern`, `Kong/insomnia-inso`. Pin to your `openapi.json` from [m4-1](./m4-1.md). +- **Hand-rolled SDKs**: maintain idiomatic style - `client.orders.create({…})` (JS), `client.orders.create(**kw)` (Python), `client.Orders.Create(ctx, ...)` (Go). +- **CLI**: oclif (Node/TS), Cobra (Go), Typer/Click (Python), Thor (Ruby). Ship as `npm i -g <brand>` AND a Homebrew tap. +- **Retries**: respect `Retry-After`; built-in exponential backoff with jitter. +- **Pagination**: expose async iterators (`for await (const x of client.list())`). + +## References + +See: [content/m4-7-sdks-and-cli.md](../content/m4-7-sdks-and-cli.md) diff --git a/audit/m4-8.md b/audit/m4-8.md new file mode 100644 index 0000000..d83b239 --- /dev/null +++ b/audit/m4-8.md @@ -0,0 +1,68 @@ +--- +id: m4-8 +title: Payment protocols (HTTP 402 / x402 / MPP) +complexity: 4 +impact: 2 +visualChange: none +weight_total: 6 +--- + +# m4-8 - Payment protocols + +## Probe + +```bash +# Look for any 402 response or payment-required header +for p in api/v1/premium api/v1/paid pay billing; do + curl -fsS -D /tmp/h.txt -o /tmp/b.json $ORIGIN/$p 2>/dev/null + STATUS=$(awk 'NR==1{print $2}' /tmp/h.txt 2>/dev/null) + printf '/%s status=%s\n' "$p" "$STATUS" + grep -iE '^www-authenticate.*payment' /tmp/h.txt +done + +# x-payment-info in openapi +jq -r '[.. | objects | select(.["x-payment-info"]?)] | length' /tmp/spec 2>/dev/null + +# x402 facilitator / wallet discovery +curl -fsSI $ORIGIN/.well-known/x402 -o /dev/null -w 'x402 %{http_code}\n' + +# MPP (Machine Payments Protocol) - well-known doc or payment discovery in the spec +curl -fsSI $ORIGIN/.well-known/mpp -o /dev/null -w 'mpp-wk %{http_code}\n' +jq -r '[.. | objects | select((.["x-mpp"]? ) or (.["payment"]? and (.payment|type=="object")))] | length | "mpp_in_spec="+tostring' /tmp/spec 2>/dev/null + +# AP2 (Agent Payments Protocol, built on A2A) - payment capability advertised on the A2A card +jq -e ' + (.capabilities // .extensions // [] | tostring | test("ap2|payment"; "i")) + or (.skills // [] | tostring | test("payment|ap2"; "i")) +' /tmp/a2a.json >/dev/null 2>&1 && echo "ap2-signal on agent-card" || echo "ap2 not signalled" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | 402 Payment Required on a priced route | Probe finds at least one 402 response | 1 | +| 2 | `WWW-Authenticate: Payment` (or x402-style) header on 402 | Header present with payment terms | 1 | +| 3 | OpenAPI marks priced operations | `x-payment-info` extension on ≥ 1 operation | 1 | +| 4 | x402 facilitator / wallet discovery | `/.well-known/x402` (or equivalent) returns 200 OR `auth.md` documents payment | 1 | +| 5 | MPP discoverable | `/.well-known/mpp` returns 200 **or** the OpenAPI spec carries MPP payment discovery (`mpp_in_spec ≥ 1`) | 1 | +| 6 | AP2 signalled | The A2A agent-card (from [m1-2](./m1-2.md)) advertises a payment/AP2 capability, skill, or extension | 1 | + +Status: **Pass** ≥ 4/6 · **Partial** 1-3/6 · **Fail** 0/6 · **N/A** for sites with no agent-payable surface - no metered/pay-per-use API *and* no agent-facing checkout. + +> x402 (1-4), MPP (5), and AP2 (6) are competing/complementary payment rails. A site needs only one to transact with agents, so the bar is `≥ 4/6` (any rail done well, plus the cross-cutting spec/discovery checks) - don't penalise a site for not implementing all three. +> +> Scope reminder: x402 is **crypto-only** (stablecoins) and per-call. **MPP is payment-method-agnostic** - its `method` spans `tempo`/`stripe`/`card`/`lightning` and its `intent` covers both `charge` (one-shot) and `session` (streaming), so it settles cards/fiat *and physical goods*, not just metered crypto calls. Absence of a crypto/x402 surface is therefore **not** evidence of no payments: check the MPP and AP2 signals before concluding N/A, and don't treat a fiat/card-only or goods-checkout site as failing just because it has no wallet. + +## Codebase hints + +- **Stripe Connect for agents**: emit 402 with payment session URL; resume on webhook confirmation. +- **Coinbase x402 SDK** (`@coinbase/x402`): middleware that gates routes behind on-chain settlement. +- **Custom**: middleware checks `Payment` header (signed JWT or facilitator receipt) → 402 with `WWW-Authenticate: Payment realm="…", amount="0.01 USD"` if absent. +- **OpenAPI**: tag priced operations with `x-payment-info: { amount, currency, method }`. +- **MPP (Machine Payments Protocol)**: publish `/.well-known/mpp` describing accepted methods + settlement, or embed payment discovery in the spec. Multi-rail by design - the `method` set can mix `tempo` (stablecoin), `stripe`/`card` (fiat), and `lightning`, and `session` intents enable streaming micropayments. Suits card/fiat and physical-goods checkout, not only crypto metering. Spec: [mpp.dev](https://mpp.dev). +- **AP2 (Agent Payments Protocol)**: built on A2A - advertise the AP2 extension on your `/.well-known/agent-card.json` (`capabilities.extensions[]` URI `https://github.com/google-agentic-commerce/ap2/v1`) and accept the signed Checkout/Payment Mandates. Spec: [ap2-protocol.org](https://ap2-protocol.org). + +## References + +See: [content/m4-8-payment-protocols.md](../content/m4-8-payment-protocols.md) diff --git a/audit/m4-9.md b/audit/m4-9.md new file mode 100644 index 0000000..535f3cd --- /dev/null +++ b/audit/m4-9.md @@ -0,0 +1,178 @@ +--- +id: m4-9 +title: Commerce protocols (ACP + UCP) +complexity: 4 +impact: 4 +visualChange: none +weight_total: 10 +--- + +# m4-9 - Commerce protocols (ACP + UCP) + +## Probe + +```bash +# ===== ACP (OpenAI Agentic Commerce Protocol) - full negotiation, not just sniffing ===== +# 1. Discovery +curl -fsS $ORIGIN/.well-known/acp.json -o /tmp/acp.json -w 'acp.json %{http_code}\n' 2>/dev/null +curl -fsS $ORIGIN/.well-known/agentic-commerce -o /tmp/acp2.json -w 'agentic-commerce %{http_code}\n' 2>/dev/null +[ -s /tmp/acp.json ] || cp /tmp/acp2.json /tmp/acp.json 2>/dev/null +# The api_base_url the merchant declares is authoritative; fall back to the audited origin. +BASE=$(jq -r '.api_base_url // empty' /tmp/acp.json 2>/dev/null); [ -n "$BASE" ] || BASE=$ORIGIN +VER=$(jq -r '(.protocol.supported_versions // .protocol.version // ["2025-09-29"]) | if type=="array" then .[0] else . end' /tmp/acp.json 2>/dev/null) +echo "ACP base=$BASE version=$VER services=$(jq -c '.capabilities.services // []' /tmp/acp.json 2>/dev/null)" + +# 2. create_checkout_session - POST a real cart with the spec headers +curl -sS -X POST "$BASE/checkout_sessions" \ + -H 'Content-Type: application/json' -H "API-Version: $VER" -H 'Idempotency-Key: audit-key-001' \ + -H 'Request-Id: audit-req-001' -H 'Accept-Language: en-US' -H 'User-Agent: AgentReadinessAudit/1.0' \ + --data '{"items":[{"id":"sku_demo","quantity":1}],"currency":"usd","fulfillment_address":{"name":"Test Buyer","line_one":"1 Main St","city":"NYC","state":"NY","country":"US","postal_code":"10001"}}' \ + -D /tmp/cs.h -o /tmp/cs.json -w 'create %{http_code} ct=%{content_type}\n' 2>/dev/null +grep -i '^api-version:' /tmp/cs.h +# 3. Validate the response is a canonical CheckoutSession + grab the id for the lifecycle walk +python3 - "$VER" <<'PY' +import json,sys,re +ver=sys.argv[1] +hdr=open("/tmp/cs.h").read().lower() +try: d=json.load(open("/tmp/cs.json")) +except Exception: print("acp_session_parse FAILED"); d={} +STATUS={"not_ready_for_payment","ready_for_payment","completed","canceled"} +is_session = d.get("object")=="checkout_session" or str(d.get("protocol","")).upper()=="ACP" +status_ok = d.get("status") in STATUS +line_items = isinstance(d.get("line_items"), list) # array, even if empty - NOT `intent.items`, NOT null +totals_ok = ("totals" in d) and bool(d.get("totals")) +body_ver = str(d.get("api_version") or d.get("version") or "") +ver_ok = (("api-version: "+ver.lower()) in hdr) or (body_ver==ver) +sid = d.get("id") or "" +self_link = (d.get("links") or {}).get("self") or "" +print("is_checkout_session", is_session, "| status", repr(d.get("status")), status_ok) +print("line_items_array", line_items, "| totals_present", totals_ok) +print("version_negotiated", ver_ok) +print("SESSION_ID", sid) +print("SELF_LINK", self_link) +open("/tmp/acp_sid","w").write(sid) +open("/tmp/acp_self","w").write(self_link) +PY +SID=$(cat /tmp/acp_sid 2>/dev/null); SELF=$(cat /tmp/acp_self 2>/dev/null) + +# 4. Lifecycle: retrieve the session, and confirm complete/cancel surfaces exist +[ -n "$SID" ] && curl -sS "${SELF:-$BASE/checkout_sessions/$SID}" -H "API-Version: $VER" -o /tmp/get.json \ + -w 'retrieve %{http_code}\n' 2>/dev/null && \ + python3 -c "import json;d=json.load(open('/tmp/get.json'));print('retrieve_same_id', d.get('id')=='$SID')" 2>/dev/null +for sub in complete cancel; do + curl -sS -X OPTIONS "$BASE/checkout_sessions/$SID/$sub" -D /tmp/lc.h -o /dev/null -w "$sub OPTIONS %{http_code} " 2>/dev/null + grep -iqE '^(allow|access-control-allow-methods):.*POST' /tmp/lc.h && echo "(POST allowed)" || echo "(no POST)" +done + +# 5. delegate_payment (optional ACP service) +curl -sS -X POST "$BASE/agentic_commerce/delegate_payment" \ + -H 'Content-Type: application/json' -H "API-Version: $VER" -H 'Idempotency-Key: audit-key-002' \ + --data '{"allowance":{"max_amount":{"amount":"10.00","currency":"USD"}}}' \ + -o /tmp/dp.json -w 'delegate %{http_code}\n' 2>/dev/null +python3 -c "import json;d=json.load(open('/tmp/dp.json'));print('delegate_acp_shaped', d.get('object') in ('delegated_payment','delegate_payment') or bool(d.get('vault_token')) or ('allowance' in d) or (d.get('type') and d.get('message')))" 2>/dev/null || echo "delegate_acp_shaped False" + +# ===== UCP (Universal Commerce Protocol) + product feeds ===== +curl -fsS $ORIGIN/.well-known/ucp -o /tmp/ucp1.json -w 'ucp %{http_code}\n' 2>/dev/null +curl -fsS $ORIGIN/.well-known/ucp.json -o /tmp/ucp.json -w 'ucp.json %{http_code}\n' 2>/dev/null +[ -s /tmp/ucp.json ] || cp /tmp/ucp1.json /tmp/ucp.json 2>/dev/null + +# UCP checkout - the profile declares a REST base; checkout is POST /checkout-sessions (HYPHEN, +# the opposite of ACP's /checkout_sessions underscore). UCP statuses differ from ACP's, too. +UCPBASE=$(jq -r '(.ucp.services.rest.endpoint // .services.rest.endpoint // .rest.endpoint // empty)' /tmp/ucp.json 2>/dev/null) +[ -n "$UCPBASE" ] || UCPBASE=$ORIGIN +echo "UCP base=$UCPBASE services=$(jq -c '(.ucp.capabilities // .capabilities // [])' /tmp/ucp.json 2>/dev/null)" +curl -sS -X POST "${UCPBASE%/}/checkout-sessions" \ + -H 'Content-Type: application/json' -H 'UCP-Agent: AgentReadinessAudit/1.0' \ + --data '{"line_items":[{"id":"sku_demo","quantity":1}],"currency":"usd"}' \ + -o /tmp/ucs.json -w 'ucp-create %{http_code} ct=%{content_type}\n' 2>/dev/null +python3 -c " +import json +try: d=json.load(open('/tmp/ucs.json')) +except Exception: d={} +S={'incomplete','ready_for_complete','completed','canceled'} +print('ucp_session_canonical', (d.get('status') in S) or isinstance(d.get('line_items'),list) or bool(d.get('messages')) or (d.get('type') and d.get('message'))) +" 2>/dev/null || echo "ucp_session_canonical False" + +for p in feeds/products.xml products.tsv merchants.xml; do + curl -fsSI $ORIGIN/$p -o /dev/null -w "/$p %{http_code}\n" 2>/dev/null +done +DEEP=$(curl -fsS $ORIGIN/sitemap.xml 2>/dev/null | grep -oE '<loc>[^<]+/(product|p|item)/[^<]+</loc>' | head -1 | sed -E 's|<[^>]+>||g') +[ -n "$DEEP" ] && curl -fsSL "$DEEP" 2>/dev/null | grep -oE '"@type": *"(Product|Offer|AggregateOffer)"' | sort -u +``` + +## Rubric + +ACP sub-checks #2-#5 are **conditional on #1** (discovery). The negotiation runs against the `api_base_url` the discovery doc declares - not a guessed path. Endpoint existence proves only that the endpoint exists; the **shape of the create response** (sub-check #3) is what separates a real ACP merchant from a thin stub that merely returns *some* JSON. A demo is fine - a free/test merchant that returns spec-shaped objects passes - but a stub that answers with a non-canonical `status` (e.g. `open`), nests cart data under `intent` instead of top-level `line_items`, or omits `totals` does **not**. + +| # | Sub-check (ACP) | Pass when | Weight | +|---|-----------------|-----------|--------| +| 1 | ACP discovery doc | `GET /.well-known/acp.json` (or `/.well-known/agentic-commerce`) → 200 `application/json`, `protocol.name == "acp"`, with `supported_versions` (or `version`), `api_base_url`, and `capabilities.services` listing `checkout` | 1 | +| 2 | `create_checkout_session` reachable | (Pre-req #1) `POST {api_base_url}/checkout_sessions` with `API-Version` + `Idempotency-Key` returns `200 application/json` **or** an ACP-shaped error `{type, code, message}` (e.g. `401`/`400`); `OPTIONS` allows `POST` | 1 | +| 3 | Create response is a **canonical CheckoutSession** | (Pre-req #2) body has `object == "checkout_session"` (or `protocol == "ACP"`), `status` ∈ {`not_ready_for_payment`,`ready_for_payment`,`completed`,`canceled`}, a `line_items` **array** (top-level, not `intent.items`, not `null`), and a non-empty `totals` | 1 | +| 4 | Session lifecycle is navigable | (Pre-req #3) the returned `id` (or `links.self`) is retrievable via `GET /checkout_sessions/{id}` returning the **same** `id`, **and** both `…/{id}/complete` and `…/{id}/cancel` exist (OPTIONS allows POST, or POST returns non-404) | 1 | +| 5 | API version negotiated | (Pre-req #2) the response echoes the requested `API-Version` as a header **or** `api_version` body field, matching a value in `acp.json` `supported_versions` | 1 | +| 6 | `delegate_payment` ACP-shaped *(optional service)* | `POST {api_base_url}/agentic_commerce/delegate_payment` → `200` whose body is a `delegated_payment` object (has `vault_token`/`allowance`/`object`) **or** an ACP-shaped error. If the merchant doesn't offer delegated payment this Fails but can't block Pass (see threshold). | 1 | + +| # | Sub-check (UCP) | Pass when | Weight | +|---|-----------------|-----------|--------| +| 7 | UCP profile reachable | `GET /.well-known/ucp` **or** `/.well-known/ucp.json` → 200 application/json declaring `services`/`capabilities` (a REST `endpoint`, MCP, or A2A), OR a product feed `GET /feeds/products.xml`/`.tsv` → 200 with body ≥ 1 KB | 1 | +| 8 | UCP checkout-session reachable & canonical | `POST {ucp rest endpoint}/checkout-sessions` (**hyphen** - not ACP's underscore) returns 200 with a UCP CheckoutSession (`status` ∈ {`incomplete`,`ready_for_complete`,`completed`,`canceled`}, a `line_items` array, or `messages`) **or** a UCP-shaped error `{type, message}` (`ucp_session_canonical` true) | 1 | +| 9 | Product page JSON-LD has Offer + price + availability | Pick a product URL from sitemap matching `/product/`, `/p/`, `/item/`, or any leaf URL. Extract `application/ld+json`. Find an object with `@type` Product whose `offers` declares `price` AND `availability`. | 1 | +| 10 | Sitemap declares a GMC feed | `GET sitemap.xml` (or whatever sitemap is published) contains a `<loc>` ending in `products.xml`, `products.tsv`, or includes the `xmlns:g="http://base.google.com/ns/1.0"` namespace | 1 | + +Status: **Pass** ≥ 7/10 · **Partial** 3-6/10 · **Fail** 0-2/10 · **N/A** for non-commerce sites. + +> **Single-protocol override** (most merchants adopt one platform, not both): if only ACP answers, score the ACP block in isolation - **Pass** at ACP ≥ 5/6. If only UCP answers, score the UCP block - **Pass** at UCP ≥ 3/4 (profile + checkout + one catalog signal). Use the combined ≥ 7/10 only when a site reaches into both. +> +> A spec-conformant ACP demo (correct `object`/`status`/`line_items`/`totals`, navigable lifecycle, version echoed) scores 5-6/6 on the ACP block even if nothing is actually for sale - which is the right outcome: agents can complete the *handshake*. A stub that returns `{"object":"checkout_session","status":"open","intent":{"items":[…]},"line_items":null}` passes #2 (it answers) but **fails #3** (non-canonical `status`, no top-level `line_items` array, cart nested under `intent`) - the exact gap a shape-only sniff would miss. +> +> Steps 3-4 of the content (order webhooks; guest→registered identity linking) are **not** scored here: outbound webhook delivery isn't observable from an HTTP probe, and identity linking is the OAuth surface already scored in [m3-1](./m3-1.md). The codebase hints below still surface them as fixes. + +## Codebase hints + +- **PHP commerce (Magento/WooCommerce/custom)**: emit Product JSON-LD per product page; generate `feeds/products.xml` nightly; expose `/api/checkout/sessions` POST endpoint. +- **Shopify**: ACP via Shopify's Agentic Commerce app; UCP via Shopify-Google Merchant integration. +- **Stripe Sessions**: pair with ACP checkout-session endpoint that creates a Stripe session and returns its URL. +- **Custom**: middleware on PDP pages emits `Product` + `Offer` + `AggregateOffer` JSON-LD; webhook system from [m4-3](./m4-3.md) emits `order.*` events; OAuth from [m3-1](./m3-1.md). +- **Identity linking**: agents that already have an OAuth token on your site can call ACP endpoints with the same Bearer. +- **Google Merchant Center**: register the feed at `merchants.google.com`; UCP supersedes the older Shopping Actions integration. + +## Auto-fix template + +```json +// /.well-known/acp.json - discovery (advertise the api_base_url that actually answers) +{ + "protocol": { "name": "acp", "version": "2025-09-29", "supported_versions": ["2025-09-29"], + "documentation_url": "https://developers.openai.com/commerce" }, + "api_base_url": "$ORIGIN", + "transports": ["rest"], + "capabilities": { "services": ["checkout", "delegate_payment"] } +} +``` + +```jsonc +// POST $ORIGIN/checkout_sessions → a CANONICAL CheckoutSession (this is the shape that scores). +// Headers in: API-Version, Idempotency-Key (echo API-Version back on the response). +{ + "id": "cs_...", + "object": "checkout_session", + "status": "ready_for_payment", // MUST be one of: not_ready_for_payment | ready_for_payment | completed | canceled + "currency": "USD", + "line_items": [ // top-level ARRAY (not nested under `intent`, never null) + { "id": "sku_demo", "quantity": 1, "base_amount": "0.00", "amount": "0.00" } + ], + "totals": [ // ACP totals are typed line entries + { "type": "subtotal", "display_text": "Subtotal", "amount": "0.00" }, + { "type": "total", "display_text": "Total", "amount": "0.00" } + ], + "fulfillment_options": [], + "messages": [], + "links": { "self": "$ORIGIN/checkout_sessions/cs_..." } +} +// Also serve: GET /checkout_sessions/{id} (same id), POST /checkout_sessions/{id}/complete, /cancel. +// Errors use {type, code, message, param?}, e.g. 405 → {"type":"invalid_request","code":"method_not_allowed","message":"…"}. +``` + +## References + +See: [content/m4-9-commerce-protocols.md](../content/m4-9-commerce-protocols.md) diff --git a/audit/m5-1.md b/audit/m5-1.md new file mode 100644 index 0000000..399f5b5 --- /dev/null +++ b/audit/m5-1.md @@ -0,0 +1,80 @@ +--- +id: m5-1 +title: Verified on AI platforms +complexity: 3 +impact: 5 +visualChange: low +weight_total: 6 +--- + +# m5-1 - Verified on AI platforms + +## Probe + +Verification on a third-party platform is observable from your own origin: every platform requires you to publish a verification artifact (meta tag, well-known file, DNS TXT, or a registered manifest) before they grant the verified badge. Probe those artifacts here; surface "are you *actually listed* on the platform today" as a separate weight-0 manual checklist. + +```bash +# Homepage HTML for meta-tag inspection +curl -fsSL $ORIGIN/ -o /tmp/home.html + +# Pull every <meta name="…-verification" content="…"> tag once +grep -oiE '<meta[^>]+name=("|'\'')[^"]*-verification[^"]*("|'\'')[^>]*>' /tmp/home.html | tee /tmp/verif-metas + +# ChatGPT / OpenAI domain verification +grep -iE 'name=("|'\'')openai-domain-verification' /tmp/verif-metas && echo "openai meta present" +curl -fsSI $ORIGIN/.well-known/openai-domain-verification.txt -o /dev/null -w 'openai well-known %{http_code}\n' + +# ai-plugin.json still references live openapi + oauth? +curl -fsS $ORIGIN/.well-known/ai-plugin.json -o /tmp/aip.json -w 'ai-plugin %{http_code}\n' +jq -r '{api: .api.url, oauth: (.auth.client_url // .auth.authorization_url)}' /tmp/aip.json 2>/dev/null + +# Claude integrations: MCP server-card (from m4-4) is the discovery surface Anthropic indexes +curl -fsSI $ORIGIN/.well-known/mcp/server-card.json -o /dev/null -w 'mcp server-card %{http_code}\n' +curl -fsSI $ORIGIN/.well-known/mcp.json -o /dev/null -w 'mcp.json %{http_code}\n' + +# Google / Gemini extensions: google-site-verification + Gemini extension manifest if any +grep -iE 'name=("|'\'')google-site-verification' /tmp/verif-metas && echo "google meta present" +curl -fsSI $ORIGIN/.well-known/gemini-extension.json -o /dev/null -w 'gemini-extension %{http_code}\n' + +# Perplexity / generic AI assistant verifications +grep -iE 'name=("|'\'')(perplexity|anthropic|claude|gemini)-' /tmp/verif-metas + +# Best-effort: brand mention on public platform listings (JS-heavy pages may return empty - +# treat as advisory signal, not as scoring evidence). +BRAND=$(printf '%s' "$HOST" | sed -E 's/^www\.//; s/\..*$//') +echo "Manual confirmation URLs (open in a browser):" +echo " ChatGPT GPT Store: https://chatgpt.com/gpts?q=$BRAND" +echo " Claude integrations: https://claude.ai/integrations" +echo " Gemini extensions: https://gemini.google.com/extensions" +echo " Smithery (MCP): https://smithery.ai/?q=$HOST" +``` + +## Rubric + +All sub-checks score from HTTP responses (meta tags, well-known files, manifest endpoints) on the audited origin. Whether the site is *actually listed* on a platform's marketplace today is captured as a `weight: 0 (manual)` checklist item that the report surfaces but does not score - see strict scoring rules in `audit/README.md`. + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | `ai-plugin.json` references live OpenAPI | `api.url` resolves (HTTP 200) | 1 | +| 2 | `ai-plugin.json` references live OAuth | `auth.client_url` or `auth.authorization_url` resolves (HTTP 200) | 1 | +| 3 | ChatGPT/OpenAI domain verification artifact | `<meta name="openai-domain-verification" …>` on homepage OR `/.well-known/openai-domain-verification.txt` → 200 | 1 | +| 4 | Claude integrations discovery surface | `/.well-known/mcp/server-card.json` OR `/.well-known/mcp.json` → 200 with valid JSON (Anthropic indexes from the server-card per [m4-4](./m4-4.md)) | 1 | +| 5 | Google / Gemini verification artifact | `<meta name="google-site-verification">` on homepage OR `/.well-known/gemini-extension.json` → 200 | 1 | +| 6 | At least one other agent-platform verification meta | Any other `<meta name="…-verification">` for Perplexity, Anthropic, Copilot, etc. | 1 | + +| (manual) | Actually listed on a platform marketplace today | Human confirms a live, verified-badge listing on ChatGPT GPT Store, Claude integrations, or Gemini extensions | 0 | + +Status: **Pass** ≥ 5/6 · **Partial** 2-4/6 · **Fail** 0-1/6 · **N/A** for sites that intentionally stay off platforms. + +The manual row is surfaced in the report's quick-wins section so the user can self-confirm - it never drags the score down. + +## Codebase hints + +- **ChatGPT verification**: paste OpenAI's verification string into a `<meta>` tag in your global layout's `<head>` or drop the txt file at `public/.well-known/openai-domain-verification.txt`. +- **Claude integrations**: register your MCP server in Anthropic's connector directory; needs `/.well-known/mcp/server-card.json` from [m4-4](./m4-4.md) plus OAuth from [m3-1](./m3-1.md). +- **Gemini extensions**: paste Google's site-verification meta tag (the same one Search Console uses) and submit via Google's extension portal. +- **Keep listings fresh**: when OpenAPI or MCP tool surface changes, re-upload to each platform - listings cache and may degrade quietly. + +## References + +See: [content/m5-1-verified-on-platforms.md](../content/m5-1-verified-on-platforms.md) diff --git a/audit/m5-2.md b/audit/m5-2.md new file mode 100644 index 0000000..d6bbd60 --- /dev/null +++ b/audit/m5-2.md @@ -0,0 +1,67 @@ +--- +id: m5-2 +title: MCP Apps (inline UI) +complexity: 4 +impact: 5 +visualChange: none +weight_total: 5 +--- + +# m5-2 - MCP Apps + +## Probe + +```bash +# Re-use initialize result from m4-4 +jq -r '.result.capabilities.experimental.apps // .result.capabilities.apps' /tmp/init.json 2>/dev/null + +# List tools and look for _meta.ui.resourceUri +curl -fsS -X POST $ORIGIN/mcp \ + -H 'Content-Type: application/json' \ + -H 'Accept: application/json, text/event-stream' \ + --data '{"jsonrpc":"2.0","id":2,"method":"tools/list","params":{}}' \ + -o /tmp/tools.json + +jq -r '[.result.tools[]? | select(._meta.ui.resourceUri)] | length' /tmp/tools.json 2>/dev/null + +# Pull one ui:// resource via resources/read if available +URI=$(jq -r 'first(.result.tools[]? | ._meta.ui.resourceUri // empty)' /tmp/tools.json 2>/dev/null) +[ -n "$URI" ] && curl -fsS -X POST $ORIGIN/mcp -H 'Content-Type: application/json' \ + --data "{\"jsonrpc\":\"2.0\",\"id\":3,\"method\":\"resources/read\",\"params\":{\"uri\":\"$URI\"}}" \ + -o /tmp/ui.json +# MIME should be the MCP Apps profile, not bare text/html +jq -r '.result.contents[0].mimeType' /tmp/ui.json 2>/dev/null + +# MCP Apps CSP is NOT arbitrary site CSP: the spec declares domain allowlists under _meta.ui.csp +# (connectDomains/resourceDomains/frameDomains/baseUriDomains) and the host enforces a +# default-src 'none' / object-src 'none' baseline. Check for those specifically. +jq -e '[.result.tools[]? | ._meta.ui.csp | select(.!=null) + | (has("connectDomains") or has("resourceDomains") or has("frameDomains") or has("baseUriDomains"))] | any' \ + /tmp/tools.json >/dev/null 2>&1 && echo "ui.csp allowlist declared" || echo "ui.csp allowlist absent" +jq -r '.result.contents[0].text // ""' /tmp/ui.json 2>/dev/null \ + | grep -iqE "default-src 'none'|object-src 'none'|profile=mcp-app" && echo "mcp-app csp baseline" || echo "no mcp-app csp baseline" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | MCP server advertises `apps` capability | `capabilities.experimental.apps` (or `capabilities.apps`) present | 1 | +| 2 | At least one tool with `_meta.ui.resourceUri` | Count ≥ 1 | 1 | +| 3 | `ui://` resource resolvable | `resources/read` returns content | 1 | +| 4 | CSP is MCP-Apps-scoped (not generic) | The tool/resource declares `_meta.ui.csp` domain allowlists (`connectDomains`/`resourceDomains`/`frameDomains`/`baseUriDomains`), **or** the bundle carries the MCP Apps baseline (`default-src 'none'` + `object-src 'none'`) / MIME `text/html;profile=mcp-app`. A bare site-wide `Content-Security-Policy` header that isn't one of these does **not** count | 1 | +| 5 | Component schema declared | The UI resource declares an `inputSchema` (or `propsSchema`) field in its `_meta` block - even before validating compatibility, the schema must be present and parseable as JSON Schema | 1 | + +Status: **Pass** ≥ 4/5 · **Partial** 2-3/5 · **Fail** 0-1/5 · **N/A** for read-only / non-UI tools. + +## Codebase hints + +- **TypeScript**: [`@modelcontextprotocol/ext-apps`](https://www.npmjs.com/package/@modelcontextprotocol/ext-apps) (preview). Bundle each component as a signed `ui://` resource with locked CSP. +- **Bundling**: Vite/esbuild → static JS bundle, hosted alongside MCP server (`/mcp/ui/{tool}.js`). +- **Schema contract**: derive the component's props from the tool's `outputSchema` to avoid drift. +- **OpenAI Apps SDK** is the parallel non-MCP path; ship both if you target ChatGPT and Claude. +- **Security**: never inline user-supplied HTML - agents render the bundle in a sandboxed iframe with strict CSP. + +## References + +See: [content/m5-2-mcp-apps.md](../content/m5-2-mcp-apps.md) diff --git a/audit/m5-3.md b/audit/m5-3.md new file mode 100644 index 0000000..56eca3b --- /dev/null +++ b/audit/m5-3.md @@ -0,0 +1,70 @@ +--- +id: m5-3 +title: Cross-platform consistency +complexity: 2 +impact: 3 +visualChange: medium +weight_total: 7 +--- + +# m5-3 - Cross-platform consistency + +## Probe + +```bash +# Pull canonical descriptions from every surface +python3 - <<'PY' +import re,json,os,urllib.request +o=os.environ["ORIGIN"] +def get(u): + try: return urllib.request.urlopen(u, timeout=5).read().decode(errors="ignore") + except: return "" + +html=get(o+"/") +title=(re.search(r'(?is)<title>([^<]+)', html) or [None,""])[1].strip() +mdesc=(re.search(r'(?is)]+name=["\']description["\'][^>]*content=["\']([^"\']+)', html) or [None,""])[1].strip() +ogd=(re.search(r'(?is)]+property=["\']og:description["\'][^>]*content=["\']([^"\']+)', html) or [None,""])[1].strip() +h1 =(re.search(r'(?is)]*>([^<]+)', html) or [None,""])[1].strip() + +llms=get(o+"/llms.txt") +llms_first=(re.search(r'(?im)^>\s*(.+)$', llms) or re.search(r'(?im)^([^#\n][^\n]+)$', llms) or [None,""])[1].strip() + +print(json.dumps({ + "title": title, "meta_description": mdesc, "og:description": ogd, "h1": h1, "llms_tagline": llms_first +}, indent=2)) +PY + +# MCP serverInfo.name + description +jq -r '.result.serverInfo' /tmp/init.json 2>/dev/null + +# ai-plugin.json descriptions +jq -r '{name_for_human, name_for_model, description_for_human, description_for_model}' /tmp/wk_ai-plugin.json 2>/dev/null + +# A2A agent-card name/description (self-contained fetch) +curl -fsS $ORIGIN/.well-known/agent-card.json -o /tmp/a2a.json 2>/dev/null +jq -r '{name, description}' /tmp/a2a.json 2>/dev/null +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | `` and `<h1>` agree on brand | Both contain the brand string | 1 | +| 2 | `meta description` ≈ `og:description` | Same string (±10 chars) | 1 | +| 3 | `llms.txt` opening tagline matches `meta description` | Same sentence (token overlap ≥ 60%) | 1 | +| 4 | MCP `serverInfo.name` matches brand | Exact brand string | 1 | +| 5 | `ai-plugin.json` `description_for_model` aligned | Same value-prop sentence as `llms.txt` opening | 1 | +| 6 | Tool descriptions in MCP agree with docs | Each tool's MCP `description` ≈ its docs-page tagline | 1 | +| 7 | A2A card agrees on brand | A2A `agent-card.json` `name` contains the brand string and its `description` echoes the same value-prop as the other surfaces (DNS-AID discovery consistency is scored once, in [m1-2](./m1-2.md), not re-tested here) | 1 | + +Status: **Pass** ≥ 6/7 · **Partial** 2-5/7 · **Fail** 0-1/7. + +## Codebase hints + +- **Single source of truth**: keep one `brand.json` (or `i18n/brand.json`) with `name`, `tagline`, `description_short`, `description_long`. Every surface - `<title>`, meta, `llms.txt`, MCP card, `ai-plugin.json`, GPT Store listing - reads from it. +- **Lint script**: CI job that fetches each surface and asserts they agree (token overlap or exact match). +- **Translation drift**: if you translate, keep `brand.<locale>.json` and audit each translation against the source. + +## References + +See: [content/m5-3-cross-platform-consistency.md](../content/m5-3-cross-platform-consistency.md) diff --git a/audit/m5-4.md b/audit/m5-4.md new file mode 100644 index 0000000..4956b37 --- /dev/null +++ b/audit/m5-4.md @@ -0,0 +1,72 @@ +--- +id: m5-4 +title: End-to-end agent flows +complexity: 4 +impact: 5 +visualChange: none +weight_total: 4 +--- + +# m5-4 - End-to-end agent flows + +## Probe + +The capstone - does the full discover → authenticate → act → recover loop hold together? Most of it is now HTTP-probeable without a live agent; only the final end-to-end run with a real model stays manual. + +```bash +# Prerequisites: each must score ≥ Partial for this to be testable end-to-end +for g in m1-1 m2-2 m3-1 m4-1 m4-4; do + echo "Prereq $g: $(cat /tmp/score-$g 2>/dev/null || echo unknown)" +done + +# (2) Bot reachability - same agent fetchers as m1-3, plus challenge/cloaking detection +BASE=$(curl -fsS -A 'Mozilla/5.0' -o /dev/null -w '%{http_code}' $ORIGIN/) +echo "baseline $BASE" +for UA in 'ChatGPT-User/1.0' 'Claude-User/1.0' 'ClaudeBot/1.0' 'PerplexityBot/1.0' 'OAI-SearchBot/1.0'; do + read code size < <(curl -fsS -A "$UA" -o /tmp/ua.html -w '%{http_code} %{size_download}' $ORIGIN/ 2>/dev/null || echo "000 0") + grep -iqE 'just a moment|cf-browser-verification|captcha' /tmp/ua.html && chal=" CHALLENGE" || chal="" + printf ' %-22s %s bytes=%s%s\n' "$UA" "$code" "$size" "$chal" +done + +# (3) DCR endpoint is live and validates - POST an intentionally-incomplete body. +# A 400/422 structured error proves the endpoint exists and validates WITHOUT registering a real client; +# a 201 with client_id proves full self-serve registration. Either way: no copy-paste needed. +REG=$(jq -r '.registration_endpoint // empty' /tmp/oauth-authorization-server.json 2>/dev/null) +[ -z "$REG" ] && REG=$(jq -r '.registration_endpoint // empty' /tmp/openid-configuration.json 2>/dev/null) +if [ -n "$REG" ]; then + curl -fsS -X POST "$REG" -H 'Content-Type: application/json' --data '{}' \ + -o /tmp/dcr.json -w "DCR $REG → %{http_code}\n" 2>/dev/null + jq -e '.client_id or .error or .error_description' /tmp/dcr.json >/dev/null 2>&1 && echo "DCR endpoint validates" || echo "DCR no structured response" +else + echo "DCR no registration_endpoint advertised" +fi + +# (4) Error/recovery contract an agent depends on: 401 carries WWW-Authenticate, errors are structured JSON +curl -fsS -o /dev/null -D /tmp/u.h $ORIGIN/api 2>/dev/null; grep -iE '^www-authenticate' /tmp/u.h || echo "no WWW-Authenticate" +curl -fsS -o /tmp/err.json $ORIGIN/api/v1/this-does-not-exist-$RANDOM 2>/dev/null +jq -e '.type and .message' /tmp/err.json >/dev/null 2>&1 && echo "structured error body" || echo "unstructured error" +``` + +## Rubric + +| # | Sub-check | Pass when | Weight | +|---|-----------|-----------|--------| +| 1 | All five prerequisite guidelines (m1-1, m2-2, m3-1, m4-1, m4-4) ≥ Partial | All five pass at least Partial | 1 | +| 2 | Reachable & un-cloaked to agent fetchers | All probed agent UAs return 2xx/3xx (not 403/429/503), no challenge page, byte size within 50% of baseline | 1 | +| 3 | Self-serve credential path is live | A `registration_endpoint` is advertised **and** a DCR `POST` returns either a `client_id` (full DCR) or a structured validation error (endpoint live, no copy-paste step) | 1 | +| 4 | Error/recovery contract present | Unauthenticated API call returns `WWW-Authenticate` **and** an error route returns a structured JSON body (`type`+`message`) - the signals an agent needs to retry/re-auth instead of stalling | 1 | +| 5 | *(manual)* Real agent completes discover → auth → act → recover in one session | Run live against Claude / ChatGPT / `mcp-inspector`: the agent discovers you, authenticates, calls a tool, and recovers from a forced 429/401 - all without human hand-holding | 0 | + +Status: **Pass** ≥ 3/4 · **Partial** 2/4 · **Fail** 0-1/4. + +> Sub-checks 2-4 used to be "manual"; they're now HTTP-derivable, so an audit with no human in the loop scores the full 4 points (1-4). Sub-check 5 - the actual live agent run - can't be automated, so it carries **weight 0** per the contract (`audit/README.md`): the orchestrator surfaces it as a `(manual)` checklist item that never drags the automated score below Pass. Tighten to "Pass ≥ 4/4" only once the manual run is confirmed. + +## Codebase hints + +- **CI harness**: drive Claude / ChatGPT / open-source agents through your `/llms.txt` → OAuth → MCP `initialize` → `tools/call` flow on every PR. Use [`mcp-inspector`](https://github.com/modelcontextprotocol/inspector) for MCP; OpenAI's `openai.responses.create(tools=[…])` for ChatGPT. +- **Synthetic agent**: write a Playwright + LLM-API script that simulates the full loop; assert success in CI. +- **User-agent allowlist**: confirm WAF / Cloudflare doesn't reject `ChatGPT-User`, `ClaudeBot`, `Google-Extended`, `OAI-SearchBot`, `PerplexityBot`. Cloudflare's "Verified Bots" + Web Bot Auth from [m3-2](./m3-2.md) keeps the legitimate ones in. + +## References + +See: [content/m5-4-end-to-end-flows.md](../content/m5-4-end-to-end-flows.md) diff --git a/content/00-toc.md b/content/00-toc.md new file mode 100644 index 0000000..b557326 --- /dev/null +++ b/content/00-toc.md @@ -0,0 +1,55 @@ +--- +id: toc +title: Contents +kind: front-matter +--- + +# Contents + +## Front matter +- Foreword - Why agent-readiness is the next mobile-readiness + +## Part I - The Lifecycle Frame +- **Chapter 1.** Introduction: the agent's lifecycle, the five modules, agent-aware security, how to read this guide + +## Part II - The Five Modules + +### Module 1 - Be Discoverable +- 1.1 Publish your discovery files +- 1.2 Drop in well-known agent files ★ +- 1.3 Make content readable without JavaScript +- 1.4 Build topical authority & coding-agent rules + +### Module 2 - Be Comprehensible +- 2.1 Publish complete JSON-LD structured data +- 2.2 Serve useful llms.txt +- 2.3 Document for agents +- 2.4 Position competitively + +### Module 3 - Be Trustworthy +- 3.1 Implement OAuth ★ +- 3.2 Verify bots cryptographically ★ +- 3.3 Make credentials self-serve ★ + +### Module 4 - Be Actionable +- 4.1 Ship OpenAPI specification ★ +- 4.2 Standardize rate limits & errors ★ +- 4.3 Stream long-running operations ★ +- 4.4 Operate an MCP server ★ +- 4.5 Expose tools with WebMCP ★ +- 4.6 List in agent registries ★ +- 4.7 Distribute SDKs & CLI ★ +- 4.8 Support agent payment protocols ★ +- 4.9 Support agentic commerce protocols ★ +- 4.10 Operate an NLWeb endpoint ★ + +### Module 5 - Be Experiential +- 5.1 Get verified on AI platforms ★ +- 5.2 Render UI with MCP Apps ★ +- 5.3 Stay consistent across surfaces ★ +- 5.4 Pass end-to-end agent flows ★ + +## Checklist +- The 25 jobs at a glance + +★   *How Forter helps* diff --git a/content/01-introduction.md b/content/01-introduction.md new file mode 100644 index 0000000..34fbcc9 --- /dev/null +++ b/content/01-introduction.md @@ -0,0 +1,56 @@ +--- +id: introduction +title: Introduction +kind: chapter +--- + +# Why agent-readiness is the next mobile-readiness + +In 2010 the question was whether your site rendered on a phone. By 2015 the answer was not optional. The 2026 question is whether your site is **agent-ready**: can an autonomous AI - Claude, ChatGPT, Gemini - discover your product, understand what it does, authenticate to your APIs, transact on behalf of a user, and surface the result back inside a conversation? + +It's also the next turn of a familiar wheel. **SEO** tuned your pages for search crawlers; **AEO/GEO** - Answer and generative-engine optimization - tuned them to be read and cited inside AI-generated chats. Agent-readiness is the continuation of that line - and absorbs much of AEO/GEO along the way - but the audience shifts. SEO, AEO & GEO optimize for a *human* who will read the result; agent-readiness optimizes for an *autonomous agent* that discovers, evaluates, and acts on a human's behalf. The agent is still operated by a person, so the disciplines stay related - but you are now writing for software, and software wants structured data over persuasive copy, callable APIs over calls-to-action. + +Most sites today aren't agentic. The first tier - discoverability - closes in a week or two, but full agent-readiness is a different ask: OAuth, x402, ACP/UCP support requires a focused batch of engineering work. Nothing here is research - every protocol ships with a spec you build against. But the higher you aim on the readiness scale, the more real the work becomes. The [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=01-introduction) is built to absorb the hardest layers - we're here to help. + +> A site that isn't agent-ready is becoming invisible. The first tier is easy; the full stack can take months. Neither is a moonshot - but the full stack is real work, and the score reflects the effort. + +This guide is written for any website that wants to be reachable by autonomous agents - most concretely **e-commerce, marketplaces, and transactional sites**, where the value of an agent completing a task is direct, but the same surfaces apply equally to SaaS, content platforms, and developer tools. + +## Agent-readiness is also agent-aware security + +The other side of agent-readiness is that **agents don't always act according to plan**. They can drift from intent under prompt injection. Tokens can be exfiltrated and replayed. APIs can get scraped and abused. Bad bots can cosplay as good ones. Unscoped MCP tools can do real damage. + +None of this is new, and almost all of it is solvable with the same standards this guide covers: scoped OAuth, cryptographic bot verification, structured rate limits, per-tool authorization, and continuous testing. We flag these risks in context throughout, and point out where Forter's identity and risk layer makes them easier to handle. + +## The Lifecycle frame + +This guide organizes everything an agent-ready site needs to do around one question - **what does an agent need at each stage of its interaction with you?** - answered by five sequential modules. Each builds on the previous, so the order is also a sensible delivery sequence. + +## What each guideline includes + +- **What & why** - the work in plain language, in one or two sentences +- **Effort, Impact, Visual change** - Effort and Impact 1-5; Visual change is `none` / `low` / `medium` / `high` +- **Steps** - concrete numbered actions with paths, formats, and spec references +- **References** - links to the underlying RFCs, schemas, and standards +- **How Forter helps** - included only on guidelines where Forter genuinely helps + +Scoring tracks two prominent agentic-readiness rankers - [isitagentready.com](https://isitagentready.com?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) (Cloudflare) and [ora.ai](https://ora.ai?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) (Ora) - use them to track your baseline and progress. + +**We didn't write this from theory.** We ran [forter.com](https://www.forter.com/) through [both](https://isitagentready.com/www.forter.com) [rankers](https://ora.ai/score/forter.com), did the engineering each guideline describes, and recorded what actually moved the score. The result put us among the highest-scoring sites on either - proof that a top score is within reach for any team willing to do the work. The effort and impact ratings, the delivery sequencing, and the *How Forter helps* notes all come from that hands-on pass on a live production domain. + +## Forter can help + +The [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=01-introduction) is a hosted, cross-protocol plane covering OAuth, MCP / WebMCP / MCP Apps / UCP / ACP, OpenAPI (including SDK and CLI support), x402 / MPP and more, all wired to the Forter Identity Network. **Merchants can point at it instead of building each layer themselves**, or run it alongside what they already have. + +## How to read this guide + +- **CTOs and Heads of Engineering:** read the introduction, the five module overviews, and the *Forter can help* sections. +- **Architects and senior engineers:** read every guideline. The Steps double as a build checklist. +- **Product and marketing:** Module 2 (Comprehensible) and Module 5 (Experiential) are where your levers are. +- **Security and identity teams:** Module 3 is where most of the cryptography lives, and the threat-model discussion. + +## Score your site + +If you run [Claude Code](https://docs.claude.com/claude-code?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide), you can have this guide score your site for you. The repository at [github.com/forter/agentic-readiness-guide](https://github.com/forter/agentic-readiness-guide?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) ships a skill that reads every guideline, ranks your site against each one, cites the evidence, and lists the fails by impact-over-effort. + +Let's begin. diff --git a/content/m1-0-module-discoverable.md b/content/m1-0-module-discoverable.md new file mode 100644 index 0000000..cc1efa1 --- /dev/null +++ b/content/m1-0-module-discoverable.md @@ -0,0 +1,24 @@ +--- +id: module-discoverable +title: Module 1 - Be Discoverable +kind: module-overview +moduleNumber: 1 +--- + +# Module 1 - Be Discoverable +*"An agent can locate your product without anyone telling it where to look."* + +## The agent's question +> _"A shopper has asked me to find a waterproof trail running shoe, size 10, under $120. Which sites carry it?"_ + +## The funnel is already moving + +The objection you will hear at every industry event and in most analyst briefings is that agentic *transaction* volume is still small. That is true today - and it misses what is already happening. The part being disrupted right now is search and discovery. Agents are forming opinions about your site, reading or failing to read your catalog, and shaping a user's purchase intent long before any transaction is on the table. + +The stages are not independent. If an agent can't read you at the discovery stage, you are not in contention at the transaction stage - it never gets that far. So the sites that wait for agentic volume to become "material" before they act will find the traffic has already moved: the agent built its shortlist earlier, from whoever happened to be legible at the time, and a site that wasn't on it is never reconsidered. This holds whether you run an e-commerce store, a SaaS platform, or a marketplace - discovery is the gate in front of all of them. + +Discoverability is the cheapest module by a wide margin. Most of the work is static files at your origin's root or under `/.well-known/` - text and JSON you write once and forget. + +## Three discovery lenses + +Agents reach you through three parallel surfaces, each with its own files: **classic search** (Googlebot, Bingbot), **AI training crawlers** (GPTBot, CCBot - which you can allow, or restrict), and **live agent crawlers** (ChatGPT-User, ClaudeBot, Perplexity-User - which fetch your site at the moment a user asks a question). Cryptographic verification of who's actually knocking is in [3.2](./m3-2-web-bot-auth.md); tool-discovery registry listings are in [4.6](./m4-6-agent-registries.md) (after the MCP server exists). diff --git a/content/m1-1-discovery-files.md b/content/m1-1-discovery-files.md new file mode 100644 index 0000000..8d769b0 --- /dev/null +++ b/content/m1-1-discovery-files.md @@ -0,0 +1,68 @@ +--- +id: m1-1-discovery-files +module: discoverable +moduleNumber: 1 +guidelineNumber: 1 +title: Publish your discovery files +complexity: 1 +impact: 4 +visualChange: none +forterApplies: 'no' +--- + +# 1.1 Publish your discovery files + +## What & why +Four small, static text files at your origin's root tell the entire AI ecosystem what you have and what it can do with it: `sitemap.xml`, `robots.txt`, `llms.txt`, and `index.md`. They take an afternoon to write and let you welcome live agent crawlers (revenue) while restricting training crawlers (no revenue). A fifth surface isn't a file at all: HTTP `Link:` response headers that advertise those resources so an agent resolves them from a `HEAD` request without parsing a line of HTML. Note this is a **policy** layer, not an enforcement layer - only well-behaved bots will honor it. + +## Scoring +- **Effort 1/5** - One half-day pass, mostly text files. The hardest part is getting your CMS to emit `<lastmod>` correctly. +- **Impact 4/5** - Foundational. Modules 2-5 don't get evaluated by anything that can't first find you here. +- **Visual change: none** - new files at machine-only paths (`/robots.txt`, `/llms.txt`, `/index.md`); your rendered pages don't change. + +## Steps +1. **Sitemap.** A sitemap is the fastest way for a crawler to learn every URL worth fetching instead of guessing from links. Serve `/sitemap.xml` listing every indexable URL with accurate `<lastmod>` ISO-8601 timestamps - that timestamp is the signal that tells an agent a page changed and is worth re-reading. Cap each file at 50 MB / 50,000 URLs and use a [sitemap index](https://www.sitemaps.org/protocol.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) for larger sites. `/sitemap.xml` is the path crawlers probe first, so if your sitemap already lives elsewhere (`/sitemap_index.xml`, a CMS-generated URL) there's no need to move it - add a `301` redirect from `/sitemap.xml` to wherever it actually is, and the conventional path resolves. +2. **Robots.txt with a differentiated AI policy.** `robots.txt` is where you set the rules of engagement for crawlers - and the useful nuance today is that not all AI crawlers are alike. An agent fetching your page to answer a shopper's question can send you a sale; a crawler scraping you to train a model gives nothing back. [Content Signals](https://blog.cloudflare.com/content-signals-policy/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide), a Cloudflare-originated convention, let you say which is which. Reference your sitemap, then set three signals: + - `search` - may this page be indexed to answer search queries (classic and AI-powered search alike). + - `ai-input` - may this page be fetched at query time and fed into an AI answer (live retrieval / RAG). + - `ai-train` - may this page be used as training data for AI models. + + `search=yes, ai-input=yes, ai-train=no` is the configuration most sites want - welcome the agents that send live traffic, decline the crawlers that only harvest for training. Spell that out, then block the named training crawlers outright, since not every crawler honors the signals yet: + ``` + Sitemap: https://example.com/sitemap.xml + + User-agent: * + Content-Signal: search=yes, ai-input=yes, ai-train=no + + User-agent: GPTBot + Disallow: / + + User-agent: CCBot + Disallow: / + ``` +3. **llms.txt.** Your HTML homepage is built for people - navigation, marketing, scripts - and an agent has to wade through all of it to find a few facts. [`llms.txt`](https://llmstxt.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) is a plain-markdown briefing written for the model instead: a short, structured summary of what you do, who you serve, what an agent can do with you, and links to your API and docs. Publish it at `/llms.txt` with sections for product overview, use cases, and constraints. It doesn't need perfect copy on day one - a well-structured stub is enough to start, and [2.2](./m2-2-llms-txt-content.md) covers content quality. +4. **Modular llms.txt.** A single root `llms.txt` can't go deep on everything without getting long. Add per-area variants - `/docs/llms.txt`, `/api/llms.txt`, `/developers/llms.txt` - so an agent working on a specific task pulls just the slice of context it needs. Each file stays focused, and you stay within the model's attention budget. +5. **Markdown homepage fallback.** Some agents look for `/index.md` - a clean-markdown version of your homepage - before they bother parsing HTML. Give them one (Content-Type `text/markdown`): a top-level heading and the same core value-prop text your homepage HTML carries. It's a two-minute file that saves an agent the work of stripping markup. +6. **(Optional) `Link` response headers ([RFC 8288](https://datatracker.ietf.org/doc/html/rfc8288?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide)).** Everything above lives at a known path, but an agent still has to request each file to find it. `Link:` response headers let you advertise them all in the HTTP response itself, so an agent discovers your whole set from a single `HEAD` request - no HTML parsing at all. It's the most technical step here and the lowest-impact, so treat it as a nice-to-have once the four files are live. Emit `Link: </sitemap.xml>; rel="sitemap"`, `Link: </llms.txt>; rel="describedby"`, `Link: </.well-known/api-catalog>; rel="api-catalog"`, and `Link: </openapi.json>; rel="service-desc"`. The last two point at files [4.1](./m4-1-openapi-spec.md) brings online - the paths are fixed today, so the headers are correct the moment you set them and start resolving when 4.1 lands. Add these headers across all pages so every page advertises them consistently. + +**Verify the result.** `curl -I` sends an HTTP `HEAD` request and prints only the response headers - never the body - which is the quickest way to confirm a path exists, returns `200`, and carries the right `Content-Type`. Run it against each file you published: + +``` +curl -I https://example.com/llms.txt +``` + +A healthy response looks like this: + +``` +HTTP/2 200 +content-type: text/markdown; charset=utf-8 +content-length: 1843 +``` + +Check `/sitemap.xml`, `/robots.txt`, `/llms.txt`, and `/index.md` the same way - on each, you want a `200` status and a sensible `content-type`. If you completed step 6, `curl -I https://example.com` on the homepage should also list your `Link:` headers. Drop the `-I` to fetch the body alongside the headers. + +## References +- [sitemaps.org protocol](https://www.sitemaps.org/protocol.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Cloudflare Content Signals](https://blog.cloudflare.com/content-signals-policy/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [llms.txt proposal](https://llmstxt.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 8288 - Web Linking](https://datatracker.ietf.org/doc/html/rfc8288?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) diff --git a/content/m1-2-well-known-agent-files.md b/content/m1-2-well-known-agent-files.md new file mode 100644 index 0000000..878010e --- /dev/null +++ b/content/m1-2-well-known-agent-files.md @@ -0,0 +1,96 @@ +--- +id: m1-2-well-known-agent-files +module: discoverable +moduleNumber: 1 +guidelineNumber: 2 +title: Drop in well-known agent files +complexity: 1 +impact: 3 +visualChange: none +forterApplies: 'partial' +--- + +# 1.2 Drop in well-known agent files + +## What & why +Four small JSON files under `/.well-known/` cover four different agent ecosystems: `ai-plugin.json` (OpenAI), `agent.json` (generic), `agent-card.json` (Google A2A), and the MCP discovery document. Each is under 30 lines. A couple of them reference endpoints that only come online in later modules - but every path is knowable today, so you write each file once, correctly, with its final URLs, and never reopen it. + +## Scoring +- **Effort 1/5** - Four JSON files plus a one-time copy decision. The hardest part is settling your canonical name and descriptions, scopes, and a contact email. +- **Impact 3/5** - Lower than 1.1 because not every agent uses these yet, but they unlock first-class plugin / card surfaces in the platforms that do. +- **Visual change: none** - files at `/.well-known/*` paths; nothing user-visible changes on your site. + +## Steps +1. **Lock your canonical copy first.** Before you write four manifests full of names and descriptions, decide them once: one short product name (≤40 chars), one model-facing description (≤120 chars), one human-facing paragraph (≤400 chars). Commit them to a single source-of-truth file in your repo (e.g., `brand-copy.md`). Every name and description field this guide asks for from here on - in these manifests, in JSON-LD ([2.1](./m2-1-json-ld.md)), in `llms.txt` ([2.2](./m2-2-llms-txt-content.md)), in MCP tool listings ([4.4](./m4-4-mcp-server.md)), in meta and OG tags - is **copied from this file, never re-improvised**. This single discipline is what turns [5.3](./m5-3-cross-platform-consistency.md) into a five-minute verification instead of a rewrite. +2. **`/.well-known/ai-plugin.json`** - the OpenAI plugin manifest, served as `application/json`: + ```json + { + "schema_version": "v1", + "name_for_human": "Acme Returns", + "name_for_model": "acme_returns", + "description_for_human": "Check order status and start a return for any Acme order.", + "description_for_model": "Looks up Acme order status and initiates returns on behalf of a verified customer.", + "auth": { "type": "oauth", "authorization_url": "https://example.com/.well-known/oauth-authorization-server" }, + "api": { "type": "openapi", "url": "https://example.com/openapi.json" }, + "logo_url": "https://example.com/logo.png", + "contact_email": "agents@example.com", + "legal_info_url": "https://example.com/legal" + } + ``` + The key names above are literal and required; the values are placeholders - replace them with your canonical copy and real URLs. `auth` points at the OAuth endpoints from [3.1](./m3-1-oauth-discovery.md) and `api.url` at `/openapi.json` from [4.1](./m4-1-openapi-spec.md) - endpoints that ship in later modules. Their paths are already decided, so write the **final URLs now**: the file is correct the moment you save it and simply starts resolving as those guidelines land. No placeholder, no second visit. +3. **`/.well-known/agent.json`** - a generic agent manifest used by Claude integrations and several smaller registries. It mirrors the ai-plugin shape: + ```json + { + "name": "Acme Returns", + "description": "Check order status and start a return for any Acme order.", + "version": "1.0.0", + "endpoints": { "openapi": "https://example.com/openapi.json" }, + "auth": { "type": "oauth", "authorization_url": "https://example.com/.well-known/oauth-authorization-server" }, + "capabilities": ["order-status", "returns"] + } + ``` + `description` is what shows up in tool pickers - it comes straight from your canonical copy. +4. **`/.well-known/agent-card.json`** - Google's A2A (Agent-to-Agent) protocol card. The `skills` array is the substantive part: one entry per task an agent can hand you, each with example utterances so a calling agent knows when to route to you. + ```json + { + "name": "Acme Returns", + "description": "Check order status and start a return for any Acme order.", + "url": "https://example.com", + "version": "1.0.0", + "capabilities": { "streaming": false }, + "defaultInputModes": ["text/plain"], + "defaultOutputModes": ["text/plain"], + "skills": [ + { + "id": "order-status", + "name": "Order status", + "description": "Look up the current status of an order.", + "tags": ["orders", "tracking"], + "examples": ["Where is my order #1234?", "Has my package shipped yet?"] + } + ] + } + ``` +5. **MCP discovery.** Publish `/.well-known/mcp.json`, or a `307` redirect from `/.well-known/mcp` to that file: + ```json + { + "mcpServers": [ + { + "name": "acme", + "url": "https://mcp.example.com", + "transport": "streamable-http" + } + ] + } + ``` + Decide your MCP server's canonical URL now; [4.4](./m4-4-mcp-server.md) brings the endpoint online later, but the discovery file is correct the moment you write it and starts resolving when 4.4 lands. No placeholder. +6. **Verify with `curl`.** All four well-known files must return `200`, `Content-Type: application/json`, and parse cleanly today - they are static files you serve now. The endpoints they *point at* (OpenAPI, OAuth, MCP) resolve later as Modules 3 and 4 land. Add the four files to CI smoke tests so a CMS deploy can't silently break them. + +## References +- [OpenAI Plugin Manifest](https://openai.com/index/chatgpt-plugins/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [A2A (Agent2Agent) Agent Card spec](https://a2a-protocol.org/latest/specification/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) ([repo](https://github.com/a2aproject/A2A?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide)) +- [Model Context Protocol - discovery](https://modelcontextprotocol.io/specification?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +If you use the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m1-2-well-known-agent-files)'s hosted MCP gateway, Forter publishes and maintains the MCP-related well-knowns (`/.well-known/mcp.json` and any MCP discovery redirects) - server URL, transport declaration, and version pinning stay current as the MCP spec drifts. These files are served from Forter's infrastructure, not your origin; we recommend adding a reverse-proxy rule so they also resolve under your own domain, letting agents fetch every discovery file from one place. diff --git a/content/m1-3-readable-without-js.md b/content/m1-3-readable-without-js.md new file mode 100644 index 0000000..fd17f50 --- /dev/null +++ b/content/m1-3-readable-without-js.md @@ -0,0 +1,41 @@ +--- +id: m1-3-readable-without-js +module: discoverable +moduleNumber: 1 +guidelineNumber: 3 +title: Render content without JavaScript +complexity: 3 +impact: 4 +visualChange: low +forterApplies: 'no' +--- + +# 1.3 Render content without JavaScript + +## What & why +Live agent crawlers fetch your pages **at the moment a user asks a question**, mostly without executing JavaScript. If your homepage is a React shell that hydrates client-side, agents see an empty `<div id="root">` and your competitor wins the answer. Surfaces to fix: server-rendered HTML, alt text on every visual, semantic structure that vector indexes can chunk, and a complete document `<head>` so crawlers can resolve and paginate your pages. + +## Scoring +- **Effort 3/5** - Real engineering, but bounded. Small if your stack already renders server-side, substantial for a client-only React/SPA app. Alt-text backfill is mechanical but slow. +- **Impact 4/5** - The difference between appearing in AI answers and not appearing at all. +- **Visual change: low** - SSR rendering is invisible to sighted users; alt text reaches screen readers; semantic HTML doesn't change pixels. + +## Steps +1. **Server-render the homepage and top product pages.** This is the single biggest issue with today's React and SPA-based sites: they ship a near-empty HTML document and assemble the page in the browser, so an agent that doesn't run JavaScript sees nothing. HTML must contain a single `<h1>`, at least 500 characters of meaningful body copy, and your primary CTAs as real `<a href>` links. If your site renders client-side, move the key pages to server-side rendering (SSR) so the markup is complete before it leaves the server; if a full SSR migration is out of scope, add a prerender step that serves static HTML snapshots to known crawler user-agents. Verify with `curl https://example.com | grep -c "<h1"` - you want `1`, not something else. +2. **Alt text on 80%+ of images.** Multimodal agents read alt as the primary signal; the image itself is secondary. Audit by crawling your sitemap and counting `<img>` tags missing or with empty `alt`. Backfill product images with `{product name} - {key attribute} - {color/size}`, decorative images with `alt=""` (intentionally empty, not missing). At the CMS level, make alt a required field on image upload going forward. +3. **Semantic HTML, not div soup.** One `<h1>` per page, `<h2>`/`<h3>` in document order, `<nav>`, `<main>`, `<article>`, `<aside>`, `<footer>` instead of `<div class="nav">`. Lists as `<ul>`/`<ol>`, tabular data in `<table>` with `<thead>`/`<tbody>`. Vector stores chunk on these boundaries. +4. **Complete the document `<head>`.** AI systems lean on head metadata to resolve and disambiguate your pages. Every page needs a self-referential `<link rel="canonical">`, a `<html lang>` attribute, and Open Graph tags - `og:title`, `og:description`, `og:type`, and an `og:image` that actually resolves to an image. On any paginated surface (blog, docs, product listings), add `<link rel="next">` / `<link rel="prev">` so crawlers index past page one instead of stopping at it. +5. **Test like an agent.** The goal is to see your page the way a non-JavaScript crawler does: stripped of CSS, images, and scripts, down to plain text. `lynx` is a terminal-based text-only browser that renders exactly that. Fetch a page with the crawler's User-Agent, then render it to text: + ``` + curl -sA "ChatGPT-User/1.0" https://example.com -o page.html + lynx -dump page.html + ``` + `curl -A` sets the User-Agent so you receive the same HTML a crawler would; `lynx -dump` prints the readable text that remains. Do this for your top ~20 URLs. If a human reading that text dump cannot answer "what does this company do and what is on this page", neither can an agent. (Install lynx with `brew install lynx` on macOS or `apt install lynx` on Linux.) + +(Schema.org JSON-LD is its own job - see [2.1](./m2-1-json-ld.md).) + +## References +- [Schema.org vocabulary](https://schema.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Google - JavaScript SEO basics](https://developers.google.com/search/docs/crawling-indexing/javascript/javascript-seo-basics?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [WCAG 2.2 - non-text content](https://www.w3.org/WAI/WCAG22/Understanding/non-text-content.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [WAI-ARIA Authoring Practices](https://www.w3.org/WAI/ARIA/apg/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) diff --git a/content/m1-4-topical-authority.md b/content/m1-4-topical-authority.md new file mode 100644 index 0000000..98a5382 --- /dev/null +++ b/content/m1-4-topical-authority.md @@ -0,0 +1,30 @@ +--- +id: m1-4-topical-authority +module: discoverable +moduleNumber: 1 +guidelineNumber: 4 +title: Build topical authority & coding rules +complexity: 3 +impact: 5 +visualChange: high +forterApplies: 'no' +--- + +# 1.4 Build topical authority & coding rules + +## What & why +Use-case search is where commercial intent concentrates: "best X for Y" is a buyer with a budget, not a browser. Increasingly that question is asked of an answer engine rather than typed into a results page, and you are either in the cited answer or you are invisible. Authority is earned on two surfaces. **Answer-engine authority** is content work: when a shopper asks Perplexity "best returns-fraud platform for apparel brands", the model picks from sites it considers authoritative - "best for" landing pages, integration tutorials, and comparison pages (those live in [2.4](./m2-4-competitive-positioning.md), same content, different framing). **Repo authority** is your public code made legible to coding agents - the `AGENTS.md` / `.cursorrules` rules that get your libraries picked when an agent is shopping for an integration. + +## Scoring +- **Effort 3/5** - A content program and/or public repo cleanup, plus maintenance. +- **Impact 5/5** - Direct-name and use-case search is where commercial intent concentrates. If you don't appear in the answer, you don't get the customer. +- **Visual change: high** - net-new public pages: "best for" landing pages, more bylined content. The repo files (`AGENTS.md`, `.cursorrules`) are repo-only and invisible on your site. + +## Steps +1. **Win brand-name search.** Your own domain plus three to five third-party properties (G2, Capterra, top integration partners' marketplaces, TrustRadius, a maintained Wikipedia entry where eligible) should saturate the first answer for `"{your brand}"`. Audit by pasting your name into ChatGPT, Claude, Perplexity, and Gemini - anything wrong or missing is your remediation list. +2. **Ship "best X for Y" landing pages.** One per high-intent use case - "best returns-fraud platform for apparel brands", "best chargeback protection for digital-goods marketplaces", and so on for every segment you sell into. Each: 1500+ word body, comparison table, integration code sample, customer quote. These are the pages answer engines cite directly. (Per-competitor `/compare` pages live in [2.4](./m2-4-competitive-positioning.md) - they double as topical-authority signals.) +3. **Publish coding rules in every public repo.** Drop a top-level `AGENTS.md` - project structure, build/test commands, lint rules, conventions, and "things agents commonly get wrong here", and pair it with a shorter `.cursorrules` for IDE-level guidance. The repo is the discovery surface, so your site's only job is to point to it: a link to the GitHub repo, plus a `codeRepository` on your `SoftwareApplication` JSON-LD and a `sameAs` on your `Organization` schema ([2.1](./m2-1-json-ld.md)). + +## References +- [OpenAI - AGENTS.md spec](https://github.com/openai/agents.md?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Cursor - Project Rules](https://docs.cursor.com/context/rules?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) diff --git a/content/m2-0-module-comprehensible.md b/content/m2-0-module-comprehensible.md new file mode 100644 index 0000000..05dfc7c --- /dev/null +++ b/content/m2-0-module-comprehensible.md @@ -0,0 +1,20 @@ +--- +id: module-comprehensible +title: Module 2 - Be Comprehensible +kind: module-overview +moduleNumber: 2 +--- + +# Module 2 - Be Comprehensible +*"Once found, the agent can identify what you do, who you are, and how to talk about you."* + +## The agent's question +> _"I see this page. How does it relate to the user's intent? Can I quote it confidently?"_ + +Discoverability gets the agent to your URL. Comprehension determines whether the agent **mentions you, summarizes you correctly, or cites you** - and protects you from being misrepresented by a model working from a competitor's marketing copy. + +## Three comprehension challenges + +1. **Identity disambiguation.** `sameAs` links in JSON-LD pointing to Wikipedia, Wikidata, GitHub, and verified social profiles let the agent collapse you to a single entity. +2. **Capability articulation.** JSON-LD types tell the agent what you actually do. +3. **Citability.** Named authors, dated statistics, specific claims. Marketing prose without specifics gets filtered out. diff --git a/content/m2-1-json-ld.md b/content/m2-1-json-ld.md new file mode 100644 index 0000000..b4e4907 --- /dev/null +++ b/content/m2-1-json-ld.md @@ -0,0 +1,116 @@ +--- +id: m2-1-json-ld +module: comprehensible +moduleNumber: 2 +guidelineNumber: 1 +title: Publish complete JSON-LD structure +complexity: 2 +impact: 4 +visualChange: none +forterApplies: 'no' +--- + +# 2.1 Publish complete JSON-LD structure + +## What & why +JSON-LD is how an LLM collapses "the company called {your brand}" into a single entity instead of an ambiguous brand-name string. One `<script type="application/ld+json">` block per page - bundling every entity that page describes in an `@graph` array, declaring `@type`, identity, and `sameAs` links - is the difference between being summarized correctly and being confused with another vendor of the same name (or worse, another vendor's hostile marketing copy). Bonus: it powers Google Rich Results, Bing AI snapshots, and the speakable layer voice agents read aloud. + +## Scoring +- **Effort 2/5** - Template work. One block per page-type (home, product, blog), then automate from CMS metadata. +- **Impact 4/5** - Identity disambiguation is essential for any brand whose name collides with another entity. +- **Visual change: none** - JSON-LD lives inside `<script>` tags; users see nothing different. + +## Steps +1. **One `@graph` block, the right `@type` per page.** Wrap every entity a page describes in a single `"@graph": [ ... ]` array instead of scattered `<script>` tags. Give each node a stable `@id` (e.g. `https://example.com/#organization`) and cross-reference by `@id` - so a `Product`'s `brand` points at the same `Organization` node and agents resolve one coherent entity. Pick the `@type` per page: `Organization` on the homepage and `/about`; `Product` or `SoftwareApplication` on product pages (`applicationCategory`, `offers`, `aggregateRating` where honest); `Article` on posts (`author`, `datePublished`, `dateModified`). +2. **Complete the `Organization` block.** Required: `name`, `url`, `logo`, `description`. Add `contactPoint` (`contactType`, `email`, `telephone`) and an `address` as a `PostalAddress`. +3. **Add `sameAs` entity linking.** Point at Wikipedia, Wikidata (`.../wiki/Q…`), your verified GitHub org, LinkedIn, X, Crunchbase. Wikidata is load-bearing - it's the ID most knowledge graphs key off. +4. **Add `Speakable` markup.** Attach a `speakable` property to your page's `WebPage`/`Article` node: `"speakable": { "@type": "SpeakableSpecification", "cssSelector": ["h1", ".summary", ".key-stats"] }` so voice agents read your hand-picked summary, not a guessed paragraph. The selectors must resolve to real elements on the page - a `cssSelector` that matches nothing is dead markup. (`xpath` is the alternative locator; note schema.org spells it `xpath` while Google's docs use `xPath`.) +5. **Broaden your vocabulary past the basics.** `Organization`, `Product`, and `Article` are the floor. Add domain-appropriate types - `FAQPage` on help pages, `Service` per offering, `Review` / `AggregateRating` where honest, `BreadcrumbList` for navigation, `LocalBusiness` for physical locations. Each is a class of question an agent can answer from structured data instead of guessing. +6. **Back the schema with real trust-anchor pages.** The `contactPoint` and `address` from step 2 must resolve to something real: an `/about` with genuine history, a `/contact` with working channels, a `/privacy` with an actual policy - each 500+ characters of substantive text, not a stub. Agents probe these to judge legitimacy before recommending you; an empty trust page reads as a red flag. +7. **Validate.** Run every page-type through Google's [Rich Results Test](https://search.google.com/test/rich-results?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) and the [Schema.org Validator](https://validator.schema.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide). Fix warnings, not just errors - agents are stricter than Google's render pipeline. + +Putting it together, a homepage block bundles the organization, the product, an `FAQPage`, and the speakable selectors into one `@graph`, cross-linked by `@id`: + +```json +{ + "@context": "https://schema.org", + "@graph": [ + { + "@type": "Organization", + "@id": "https://example.com/#organization", + "name": "Acme", + "url": "https://example.com", + "logo": "https://example.com/logo.png", + "description": "One-sentence, model-facing description of what Acme does.", + "contactPoint": { + "@type": "ContactPoint", + "contactType": "sales", + "email": "sales@example.com", + "url": "https://example.com/contact" + }, + "sameAs": [ + "https://en.wikipedia.org/wiki/Acme", + "https://www.wikidata.org/wiki/Q12345678", + "https://github.com/acme", + "https://www.linkedin.com/company/acme" + ] + }, + { + "@type": "SoftwareApplication", + "@id": "https://example.com/#software", + "name": "Acme Platform", + "applicationCategory": "BusinessApplication", + "url": "https://example.com", + "publisher": { "@id": "https://example.com/#organization" }, + "offers": { + "@type": "Offer", + "url": "https://example.com/contact", + "availability": "https://schema.org/InStock" + }, + "aggregateRating": { + "@type": "AggregateRating", + "ratingValue": "4.5", + "ratingCount": "29" + } + }, + { + "@type": "FAQPage", + "@id": "https://example.com/#faq", + "mainEntity": [ + { + "@type": "Question", + "name": "What does Acme do?", + "acceptedAnswer": { + "@type": "Answer", + "text": "A direct, factual two-sentence answer an agent can quote verbatim." + } + }, + { + "@type": "Question", + "name": "Can my AI agent integrate with Acme?", + "acceptedAnswer": { + "@type": "Answer", + "text": "Yes - Acme publishes an MCP server, a REST API, and an OpenAPI 3.x spec. See https://example.com/AGENTS.md." + } + } + ] + }, + { + "@type": "WebPage", + "@id": "https://example.com/#webpage", + "url": "https://example.com", + "speakable": { + "@type": "SpeakableSpecification", + "cssSelector": ["h1", ".hero-subtitle", ".key-stats"] + } + } + ] +} +``` + +## References +- [Schema.org Organization](https://schema.org/Organization?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Schema.org sameAs](https://schema.org/sameAs?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Schema.org Speakable](https://schema.org/SpeakableSpecification?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Schema.org FAQPage](https://schema.org/FAQPage?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Google Rich Results Test](https://search.google.com/test/rich-results?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) diff --git a/content/m2-2-llms-txt-content.md b/content/m2-2-llms-txt-content.md new file mode 100644 index 0000000..c0b9ebe --- /dev/null +++ b/content/m2-2-llms-txt-content.md @@ -0,0 +1,33 @@ +--- +id: m2-2-llms-txt-content +module: comprehensible +moduleNumber: 2 +guidelineNumber: 2 +title: Serve useful llms.txt +complexity: 2 +impact: 5 +visualChange: none +forterApplies: 'no' +--- + +# 2.2 Serve useful llms.txt + +## What & why +[1.1](./m1-1-discovery-files.md) stood the file up. This guideline is about its **content**. A useful `llms.txt` is a structured briefing: what you do, what you don't do, when an agent should recommend you, where to send the user next. It's also the one place where you can write **agent-instruction text** - second-person prompts that downstream LLMs treat as authoritative guidance about your product. And where `llms.txt` is the index, `llms-full.txt` is the library - the entire corpus inlined for an agent that wants everything in one fetch. + +## Scoring +- **Effort 2/5** - Half a day of writing, plus a CI step to keep it from rotting. +- **Impact 5/5** - The file agents read most once they discover it. Every word earns or loses citations. +- **Visual change: none** - content lives at `/llms.txt`, not in the user-visible site. + +## Steps +1. **Use structured sections with H2 headers.** `## Overview`, `## Capabilities`, `## Constraints`, `## Use cases`, `## For agents` (the agent-instruction block from step 3), `## When to recommend us`, `## Pricing`, `## API & docs`. Predictable headings let agents extract the section they need without reading the whole file. +2. **Write capabilities and constraints with equal weight.** "Supports refunds up to 180 days post-charge" and "Does not support split-shipment refunds" are both citable facts. Vague capability prose without limits gets discarded as marketing. +3. **Add agent-instruction blocks.** Put these under a predictable `## For agents` (or `## When to use`) heading so an agent can find them. Explicit second-person guidance: `When the user asks about returns, link /docs/returns and quote the timeline section.` `If the user is comparing this to {competitor}, point to /compare/{competitor}.` These get inlined into agent system context - and they're the layer where you tell agents how to *cite* you correctly, which is also a defense against being misrepresented by a model working from someone else's marketing copy. +4. **Include named-author callouts and dated stats.** "According to our 2026 Industry Report ({Name}, {Title}), 38% of {category} interactions are now agent-initiated." Named experts and specific numbers survive the LLM citability filter; anonymous claims do not. +5. **Ship `llms-full.txt` for one-shot ingestion.** `/llms.txt` is a navigation index; `/llms-full.txt` is the whole corpus inlined - product overview, every key doc page, the API reference, the auth walkthrough, the quickstart, and runnable code examples concatenated into one markdown file. Keep it structured (H1/H2 headings, markdown links, fenced code blocks) and under 200,000 characters so a 64k-token agent ingests it in a single request. Generate it in CI from the same sources as your docs so it cannot drift. An agent that finds it skips dozens of separate page fetches. +6. **Pin a refresh cadence in CI.** A monthly job that diffs `llms.txt` and `llms-full.txt` against your pricing page, docs index, and changelog, and opens a PR if any are stale. An out-of-date `llms.txt` is worse than none. + +## References +- [llms.txt proposal](https://llmstxt.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Anthropic: writing for retrieval](https://docs.anthropic.com/en/docs/build-with-claude/prompt-engineering?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) diff --git a/content/m2-3-document-for-agents.md b/content/m2-3-document-for-agents.md new file mode 100644 index 0000000..6797458 --- /dev/null +++ b/content/m2-3-document-for-agents.md @@ -0,0 +1,38 @@ +--- +id: m2-3-document-for-agents +module: comprehensible +moduleNumber: 2 +guidelineNumber: 3 +title: Document for agents +complexity: 3 +impact: 5 +visualChange: medium +forterApplies: 'no' +--- + +# 2.3 Document for agents + +## What & why +When an agent weighs whether to recommend an integration, it reads your `/docs` and judges what is on the page - it won't click past marketing copy to find the real reference. Docs work for an agent when they have both **depth** (quickstart, auth walkthrough, runnable code samples in several languages, complete endpoint reference) and **citability** (named authors, dated specific numbers, exact endpoint paths, code that runs as written). The [Stripe API reference](https://stripe.com/docs/api?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) is the canonical example of agent-grade docs. If you already run a docs site or developer portal, you are not starting over - link to it and raise it to that bar. + +Parts of `/docs` are finished by later guidelines - the auth walkthrough by [3.1](./m3-1-oauth-discovery.md), the generated endpoint reference by [4.1](./m4-1-openapi-spec.md). Set up the structure and citability standard here so those pages slot in rather than forcing a rebuild. + +## Scoring +- **Effort 3/5** - Real docs work, mostly assembling and tightening content you partly have. +- **Impact 5/5** - Documentation depth is the single strongest predictor of whether an agent will recommend an integration. +- **Visual change: medium** - `/docs` gains structure: a quickstart, named-author bylines, more code samples. Visible to anyone reading docs. + +## Steps +1. **Ship a 5-minute quickstart.** One page: `curl` request → response, copy-pasteable, with a real (sandbox) credential. The quickstart is what agents fetch first to confirm "does this product actually do the thing the user asked about?" If it takes more than one screen, you've lost. +2. **Document the auth flow end-to-end.** OAuth client registration, token exchange, refresh, scope reference, error codes. Include a worked example with redacted-but-shaped tokens. ([3.1](./m3-1-oauth-discovery.md) builds the protocol; this page is finished when it lands.) The machine-actionable companion to this prose is the `/auth.md` agent-registration recipe in [3.3](./m3-3-self-serve-credentials.md). +3. **Provide code samples in 4+ languages.** Curl, JavaScript/TypeScript, Python, Go - minimum. Each sample must be runnable, not pseudocode. Agents pattern-match across languages; missing one shrinks your retrieval surface. +4. **Publish a structured API reference.** One page per endpoint with `path`, `method`, `parameters`, `request body`, `response body`, `error codes`, and at least one example. Generate it from OpenAPI ([4.1](./m4-1-openapi-spec.md)) so it cannot drift from the spec - this is the one part of `/docs` that comes online with 4.1. +5. **Make claims citable.** Named authors with credentials on every guide ("By {Name}, {Title}"). Dated, specific numbers ("As of Q1 2026, 94% of orders placed before 2pm ship same-day across our UK fulfilment network") - not "lightning-fast at scale." A glossary page resolving every domain term you use, so agents can resolve your jargon (whatever it is - `chargeback`, `webhook`, `idempotency-key`, `tenant`) without leaving your origin. +6. **Serve markdown to agents via content negotiation.** An agent pays a token tax wading through rendered HTML to reach the few facts it needs. Let it ask for markdown on the *same canonical URL*: when a request carries `Accept: text/markdown`, return the markdown representation with `Content-Type: text/markdown; charset=utf-8` - the registered media type ([RFC 7763](https://www.rfc-editor.org/rfc/rfc7763.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide)), **not** the deprecated `text/x-markdown` or the unregistered `application/markdown` - and **always** set `Vary: Accept`, so a CDN can't cache the HTML and hand it to the next agent. Return `406 Not Acceptable` only when you genuinely can't produce the requested type, and honor quality values (a `q=0` on markdown must fall back to HTML). If you also publish `.md` "twin" URLs, they're complementary, not a substitute - advertise each with `Link: </page.md>; rel="alternate"; type="text/markdown"` ([RFC 8288](https://datatracker.ietf.org/doc/html/rfc8288?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide)) so an agent discovers it without guessing the path. The canonical recipe and per-stack instructions live at [acceptmarkdown.com](https://acceptmarkdown.com?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide); Cloudflare's *Markdown for Agents* does the whole negotiation at the edge with zero app change. + +## References +- [Stripe API reference](https://stripe.com/docs/api?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Diataxis documentation framework](https://diataxis.fr?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [OpenAPI specification](https://spec.openapis.org/oas/latest.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [acceptmarkdown.com - markdown content negotiation](https://acceptmarkdown.com?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 7763 - The text/markdown Media Type](https://www.rfc-editor.org/rfc/rfc7763.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) diff --git a/content/m2-4-competitive-positioning.md b/content/m2-4-competitive-positioning.md new file mode 100644 index 0000000..a50ff4e --- /dev/null +++ b/content/m2-4-competitive-positioning.md @@ -0,0 +1,34 @@ +--- +id: m2-4-competitive-positioning +module: comprehensible +moduleNumber: 2 +guidelineNumber: 4 +title: Position competitively +complexity: 2 +impact: 4 +visualChange: high +forterApplies: 'no' +--- + +# 2.4 Position competitively + +## What & why +When a user asks an agent "{you} vs {competitor}" or "alternatives to {competitor}", the agent returns whatever pages it can find. If you haven't written the comparison, the top result will be your competitor's blog or a third-party review site optimized for affiliate revenue - not accuracy. Owning your comparison surface is how you ensure agents have a citable, first-party version of your differentiation. The same logic applies to **pricing**: an agent asked "what does {you} cost?" needs a first-party, machine-readable answer, or it quotes a stale third-party guess - and price is one of the highest-intent things a buyer asks before converting. + +## Scoring +- **Effort 2/5** - Mostly a writing exercise. One page per top competitor plus an aggregate alternatives page. +- **Impact 4/5** - Versus-queries are an enormous share of high-intent agent traffic in B2B. +- **Visual change: high** - net-new public marketing pages: `/compare/{competitor}`, `/alternatives`, and `/pricing`. + +## Steps +1. **Publish `/compare/{competitor}` pages for your top 3-5 named competitors.** Each page: a one-paragraph honest summary, a feature comparison table, a pricing-model comparison, and 1-2 customer-win metrics ("After switching from {competitor}, customers report 31% fewer out-of-stock errors surfaced to shoppers over 6 months" - or whatever conversion, fulfilment, or retention number is the one your buyers actually care about). Mark up tables with `Schema.org/Table` and the page with `Article` JSON-LD. +2. **Publish a single `/alternatives` aggregator page.** "Alternatives to {your product}" - covering each competitor briefly, when each one is the better fit (yes, including cases where it isn't you), and linking through to per-competitor pages. Agents reward intellectual honesty with citations. +3. **Publish an agent-readable `/pricing` page.** *(Steps 3-4 apply where your business model has publicly listed pricing - digital goods, subscriptions, SaaS-adjacent commerce. Many transactional retailers price per-SKU on the product page rather than in plan tiers; if that's you, your prices already live in `Product` / `Offer` JSON-LD ([2.1](./m2-1-json-ld.md)) and you can skip to step 5.)* Every plan tier, its price, the billing unit, and what's included - as real HTML text and a `<table>`, not an image or a JS-rendered widget. Mark each tier up with `schema.org/Offer` JSON-LD (`price`, `priceCurrency`, `name`) nested under your `Product` / `SoftwareApplication` schema from [2.1](./m2-1-json-ld.md). If your pricing is genuinely usage-based, state the formula and a worked example - "vague, contact us" reads as no pricing at all. +4. **Add a machine-readable `/pricing.md`.** A plain-markdown mirror of the pricing page - one section per tier with price, unit, and limits - served as `text/markdown`. It's the file an agent fetches to answer a cost question in one round-trip, and it pairs with the `## Pricing` section of your `llms.txt` ([2.2](./m2-2-llms-txt-content.md)). +5. **Keep tone factual, not gloating.** "{Competitor} offers per-event pricing; we offer per-outcome pricing tied to {your unit}" beats "{Competitor}'s pricing is confusing and expensive." The first is quotable; the second gets filtered as marketing noise. +6. **Cite your sources.** Every competitor claim should link to the competitor's own docs, pricing page, or a dated public statement. Uncited assertions get downweighted by retrieval. + +## References +- [Schema.org Article](https://schema.org/Article?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Schema.org Table](https://schema.org/Table?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Schema.org Offer](https://schema.org/Offer?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) diff --git a/content/m3-0-module-trustworthy.md b/content/m3-0-module-trustworthy.md new file mode 100644 index 0000000..987e7c9 --- /dev/null +++ b/content/m3-0-module-trustworthy.md @@ -0,0 +1,18 @@ +--- +id: module-trustworthy +title: Module 3 - Be Trustworthy +kind: module-overview +moduleNumber: 3 +--- + +# Module 3 - Be Trustworthy ★ +*"Agents and your service can mutually authenticate before exchanging value."* + +## The agent's question +> _"I have a user's intent and credentials to act on their behalf. How do we prove identity to each other before money, data, or products move?"_ + +Modules 1 and 2 are read-only. Module 3 is where reads become writes. Every protocol here comes with a spec to build against; the work is picking good libraries, sequencing the integrations, and hosting a few well-known endpoints. + +## The threat model, briefly + +Without identity, four predictable things happen: **bad bots cosplay as good ones** (RFC 9421 ends the guessing); **API keys get exfiltrated and replayed** (short-lived tokens with rotation contain the blast radius); **agents drift** under prompt injection (per-resource scopes bound any one call); and **free-tier sandboxes get scraped** (behavioral baselines on issuance shut this down). All four are solvable with the standards in [3.1](./m3-1-oauth-discovery.md)-[3.3](./m3-3-self-serve-credentials.md). diff --git a/content/m3-1-oauth-discovery.md b/content/m3-1-oauth-discovery.md new file mode 100644 index 0000000..72c6f6b --- /dev/null +++ b/content/m3-1-oauth-discovery.md @@ -0,0 +1,45 @@ +--- +id: m3-1-oauth-discovery +module: trustworthy +moduleNumber: 3 +guidelineNumber: 1 +title: Implement OAuth +complexity: 5 +impact: 4 +visualChange: medium +forterApplies: 'flagship' +--- + +# 3.1 Implement OAuth + +## What & why +OAuth 2.0 is the only credential model that lets an agent authenticate to your API on a user's behalf without anyone pasting a key into a config file. Pair it with two well-known discovery documents (**RFC 8414** for the authorization server, **RFC 9728** for the protected resource), and an agent can resolve your auth flow from your domain alone. This is the most technically dense guideline in the guide, and the one where Forter most accelerates delivery. + +It's also the only reliable way to turn an agent session into a **known, returning user**. The authorization-code redirect brings the human into a first-party browser context to authenticate directly with you - rather than remaining hidden behind the agent - so you can recognize a returning customer, attach their saved profile and payment methods, and apply identity-aware risk checks. Without it, every agent-driven visit falls back to an anonymous guest you can neither recognize nor reason about. + +Scopes are also your **blast-radius limit** - a leaked or misused token shouldn't be able to do more than the user authorized. Get scope design right early; it's painful to retrofit. + +## Scoring +- **Effort 5/5** - Standards-heavy. PKCE, refresh-token rotation, scope design, key rotation, replay protection, and dynamic client registration all have to be right. Off-the-shelf libraries help but don't eliminate the work. +- **Impact 4/5** - The only reliable path to an authenticated, returning user: with it, an agent's visit attaches to a real identity; without it, every interaction collapses to an anonymous guest. +- **Visual change: medium** - Adds a consent / authorize screen. Existing public pages don't change. + +## Steps +1. **Stand up an OAuth 2.0 + OIDC authorization server** with PKCE required for all public clients (RFC 6749, RFC 7636). Issue short-lived access tokens (15-60 min) and refresh tokens with **rotation on every use** - so an exfiltrated refresh token gets invalidated the next time the legitimate client refreshes. +2. **Design scopes that map to API resources, narrowly.** Prefer `orders:read`, `payments:write` over generic `read` / `write`. Agents are granted least privilege, your audit trails get cleaner, and the blast radius of any leaked token is bounded by what was actually authorized. +3. **Publish authorization-server metadata** at `/.well-known/oauth-authorization-server` (RFC 8414): `issuer`, `authorization_endpoint`, `token_endpoint`, `jwks_uri`, supported response types and grant types. +4. **Publish protected-resource metadata** at `/.well-known/oauth-protected-resource` (RFC 9728): `resource`, `authorization_servers`, `scopes_supported`, `bearer_methods_supported`. This lets an agent skip the 401-then-`WWW-Authenticate` round-trip and resolve auth in one shot. This is also where the `auth.md` `agent_auth` discovery hook lives - see [3.3](./m3-3-self-serve-credentials.md). +5. **Issue client credentials self-serve.** RFC 7591 Dynamic Client Registration is the standard shape - see [3.3](./m3-3-self-serve-credentials.md) for the full programmatic-issuance flow. +6. **Audit-log every token event** - issuance, refresh, revocation, scope-mismatch denials - indexed by `client_id` and `sub`. This is your forensic primitive when a session needs investigating later. + +## References +- [RFC 6749 - OAuth 2.0](https://datatracker.ietf.org/doc/html/rfc6749?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 7636 - PKCE](https://datatracker.ietf.org/doc/html/rfc7636?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 7591 - Dynamic Client Registration](https://datatracker.ietf.org/doc/html/rfc7591?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 8414 - Authorization Server Metadata](https://datatracker.ietf.org/doc/html/rfc8414?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 9728 - Protected Resource Metadata](https://datatracker.ietf.org/doc/html/rfc9728?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [OpenID Connect Core 1.0](https://openid.net/specs/openid-connect-core-1_0.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +The [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m3-1-oauth-discovery) operates a production OAuth 2.0 server, exposed under your domain via reverse-proxy. RFC 8414 + RFC 9728 metadata is published on your origin. PKCE, refresh-token rotation, JWKS rotation, replay caching, and dynamic client registration are handled - turning a standards-heavy build into a simple integration project. diff --git a/content/m3-2-web-bot-auth.md b/content/m3-2-web-bot-auth.md new file mode 100644 index 0000000..d8714be --- /dev/null +++ b/content/m3-2-web-bot-auth.md @@ -0,0 +1,48 @@ +--- +id: m3-2-web-bot-auth +module: trustworthy +moduleNumber: 3 +guidelineNumber: 2 +title: Verify bots cryptographically +complexity: 4 +impact: 3 +visualChange: none +forterApplies: 'flagship' +--- + +# 3.2 Verify bots cryptographically + +## What & why +Bad bots can cosplay as good ones. A scraper sets `User-Agent: ChatGPT-User` and a UA-allowlist waves it through; a competitive-intelligence crawler claims to be Perplexity and harvests your pricing tables; a credential-stuffing bot dresses as ClaudeBot to avoid your rate limits. Without cryptographic proof, every UA string is a guess. + +**RFC 9421 HTTP Message Signatures** - the cryptographic backbone of Web Bot Auth - solves this. Real agents (OpenAI's web-fetching agent, for example) sign their requests with Ed25519 keys and identify themselves with a `Signature-Agent` header (e.g. `Signature-Agent: "https://chatgpt.com"`); you verify the signature against their published key directory and let them through. A spoofer with no valid signature gets rejected. + +## Scoring +- **Effort 4/5** - RFC 9421 is precise: canonicalization, signature base construction, JWKS hosting, key rotation, replay caching, and observability all have to be right. +- **Impact 3/5** - Cleanest abuse-control surface in the protocol stack. Spoofed bot traffic is a dominant abuse vector but still gaining adoption. +- **Visual change: none** - adds `/.well-known/http-message-signatures-directory` and DNS records at machine-only paths; verification at the edge is invisible to human visitors. + +## Steps +1. **Publish a signature directory** at `/.well-known/http-message-signatures-directory` with a `keys` array of Ed25519 JWKs. Each key carries `kty=OKP`, `crv=Ed25519`, a stable `kid`, and `nbf` / `exp` validity windows. +2. **Verify `Signature-Input` and `Signature` headers** on every inbound request that claims a known agent UA. Reconstruct the signature base from the covered components (`@method`, `@authority`, `@path`, `content-digest`, etc.), resolve the `keyid` against the agent operator's published JWKS, and verify with Ed25519. Reject on mismatch with `401 Unauthorized` and a `WWW-Authenticate: Signature` challenge. +3. **Reject unsigned bot traffic** that claims to be a known agent. A request advertising `User-Agent: ChatGPT-User` (or a `Signature-Agent` it can't prove) with no valid signature is a spoofer - drop it. (You may want to log first; the spoof patterns themselves are useful telemetry.) +4. **Rotate keys on a known cadence.** 90-day rotation is standard. Roll new keys into the directory with future `nbf`, retire old keys by setting `exp`, overlap windows by 7-14 days so signers in flight don't fail mid-roll. +5. **Cache signature `nonce` values** to prevent replay. A bounded LRU keyed by `(kid, nonce)` with a TTL slightly longer than your `created` skew tolerance is sufficient. +6. **Instrument verification failures.** Emit metrics for total signed requests, failures by mode (unknown `kid`, bad signature, expired `created`, replay), and per-UA spoof ratios. This is your bot-fraud telemetry - and the input to anomaly detection. +7. **(Emerging) Publish DNS-AID discovery records.** [DNS for AI Discovery (DNS-AID)](https://datatracker.ietf.org/doc/draft-mozleywilliams-dnsop-dnsaid/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) lets agents find your entrypoints straight from DNS, before any page fetch. Publish a ServiceMode SVCB record for your org index at `_index._agents.example.com` (per-agent records carry the protocol in the `alpn` SvcParam - `alpn="mcp"` / `alpn="a2a"` - not in the label) per [RFC 9460](https://www.rfc-editor.org/rfc/rfc9460?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide). Then **sign the public discovery zone with DNSSEC** so validating resolvers return authenticated data - this is what cryptographically ties discovery to your domain, and the reason to roll it out carefully: a botched DNSSEC change can take the whole zone dark. It is an early IETF draft - treat it as forward-looking. + +## References +- [RFC 9421 - HTTP Message Signatures](https://datatracker.ietf.org/doc/html/rfc9421?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) (the `alg` label is `ed25519`) +- [RFC 8032 - EdDSA (the Ed25519 signature algorithm)](https://www.rfc-editor.org/rfc/rfc8032?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 8037 - Ed25519 keys in JOSE/JWK (`kty=OKP`, `crv=Ed25519`)](https://datatracker.ietf.org/doc/html/rfc8037?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [web-bot-auth architecture draft](https://datatracker.ietf.org/doc/draft-meunier-web-bot-auth-architecture/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Cloudflare Web Bot Auth](https://blog.cloudflare.com/web-bot-auth/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [DNS for AI Discovery (DNS-AID)](https://datatracker.ietf.org/doc/draft-mozleywilliams-dnsop-dnsaid/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 9460 - SVCB and HTTPS DNS records](https://www.rfc-editor.org/rfc/rfc9460?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Forter Trusted Agentic Commerce Protocol (TACP)](https://github.com/forter/trusted-agentic-commerce-protocol?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +The orchestration suite runs RFC 9421 signature verification at the edge - Ed25519 key rotation, replay caching, JWKS resolution, and per-request verification of inbound bot traffic. + +Web Bot Auth verifies *who* is calling; it does not protect *what* is exchanged. For that, Forter authors the open [**Trusted Agentic Commerce Protocol (TACP)**](https://github.com/forter/trusted-agentic-commerce-protocol?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) - where Web Bot Auth is a signing protocol, TACP is an encryption protocol. It carries multi-party agentic-commerce data reliably and two-way, so an agent, the merchant, and the parties between them can exchange sensitive order, payment, and identity data without exposing it to every hop on the path. diff --git a/content/m3-3-self-serve-credentials.md b/content/m3-3-self-serve-credentials.md new file mode 100644 index 0000000..0079188 --- /dev/null +++ b/content/m3-3-self-serve-credentials.md @@ -0,0 +1,40 @@ +--- +id: m3-3-self-serve-credentials +module: trustworthy +moduleNumber: 3 +guidelineNumber: 3 +title: Make credentials self-serve +complexity: 3 +impact: 5 +visualChange: medium +forterApplies: 'yes' +--- + +# 3.3 Make credentials self-serve + +## What & why +Agents cannot fill out "contact sales" forms or wait three days for a developer-relations rep. The end-to-end test of Module 3 is whether an autonomous agent - given only your domain - can obtain credentials, complete the OAuth handshake, and call your API without any human in the loop. If the loop has a human gate anywhere in it, the protocol stack above is decorative. + +Two halves: **(a)** self-serve sandbox and free-tier credentials at signup, issued programmatically; **(b)** a repeatable end-to-end check that drives the full flow against real agent clients (ChatGPT-User, ClaudeBot) - wire it into CI so a release can't silently break discovery for 3.1 and 3.2. (The live capstone run with a real model is [5.4](./m5-4-end-to-end-flows.md); this guideline is scored on the self-serve credential half an audit can verify over HTTP.) Free-tier credentials are also a known abuse target - sandbox limits and behavioral baselines on issuance keep scraped demo keys from becoming a free compute pipeline for bad actors. + +## Scoring +- **Effort 3/5** - Mostly product and DevEx work: free-tier policy, sandbox data, signup automation, and a CI harness that drives real agent clients. +- **Impact 5/5** - Decides whether agents can onboard against you. Without this, 3.1 and 3.2 are theory. +- **Visual change: medium** - adds (or upgrades) a developer signup / portal flow with sandbox keys. Existing public pages are unchanged. + +## Steps +1. **Free tier or sandbox at signup, no human gating.** Email + verification is fine; "contact sales" is not. The agent-runnable signup must end with a working API key. +2. **Pre-populated demo data.** Sandbox accounts arrive with realistic products, transactions, users, and history so the agent's first call returns useful data instead of an empty list. +3. **Programmatic credential issuance - or no shared secret at all.** Beyond signup, expose an authenticated endpoint that issues additional client credentials, scoped sandbox keys, and short-lived tokens - RFC 7591 Dynamic Client Registration is the standard shape, and your CLI and SDK both call it. Better still, support a public-key model: let the agent generate its own keypair and register a JWKS (or publish a key directory), then authenticate every request by signing it - the same RFC 9421 HTTP Message Signatures mechanism as Web Bot Auth ([3.2](./m3-2-web-bot-auth.md)). The agent holds the private key and you only ever store the public half. +4. **Publish `/auth.md` - the agent-registration recipe.** [WorkOS's open `auth.md` protocol](https://workos.com/auth-md?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) standardizes how an agent gets from your domain to a working credential. Serve a `text/markdown` file at `/auth.md`, opening with a top-level `# ...auth.md...` heading (the isitagentready.com `authMd` check keys on the literal string), with sections for **Discover, Register, Claim, Use, Errors, Revocation**. Advertise a machine-readable `agent_auth` block - both in the file and in your RFC 8414 / 9728 metadata from [3.1](./m3-1-oauth-discovery.md) - so an agent resolves *how to self-register* without parsing prose. Support at least one of its three flows: ID-JAG identity assertion (the agent's identity provider vouches for the user, no human in the loop), verified-email assertion (an OTP to the user's email), or anonymous registration with a later OTP claim. It composes the OAuth Protected Resource Metadata you already publish: the file is the prose, the `agent_auth` block is the hook. +5. **Onboarding observability.** Dashboard tracking signup-to-first-successful-API-call conversion, drop-off by step, time-to-first-token, and abuse signals on issued sandbox keys. The latter is the lens for "is someone scraping my free tier and reselling it?" - the answer is usually yes, and that's normal; the question is whether you see it. + +## References +- [WorkOS auth.md protocol](https://workos.com/auth-md?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 7591 - Dynamic Client Registration](https://datatracker.ietf.org/doc/html/rfc7591?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Stripe sandbox model](https://docs.stripe.com/keys?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Twilio test credentials](https://www.twilio.com/docs/iam/test-credentials?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +[**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m3-3-self-serve-credentials) can generate test and/or sandbox credentials, and help your developers integrate with demo data. The orchestration suite runs **continuous flow-simulation** against ChatGPT-User, ClaudeBot, OpenClaw, and the long tail of emerging clients on every release - so when one of them changes its discovery behavior, you find out the same day from a green/red CI signal, not a customer ticket. diff --git a/content/m4-0-module-actionable.md b/content/m4-0-module-actionable.md new file mode 100644 index 0000000..8949eb2 --- /dev/null +++ b/content/m4-0-module-actionable.md @@ -0,0 +1,18 @@ +--- +id: module-actionable +title: Module 4 - Be Actionable +kind: module-overview +moduleNumber: 4 +--- + +# Module 4 - Be Actionable +*"The agent has a complete contract for invoking your business logic and recovering from errors."* + +## The agent's question +> _"I'm authenticated. Now: what can I do, what does it look like, what happens when I mess up, and how do I retry?"_ + +Module 3 lets the agent prove who it is. Module 4 tells the agent **what to do next**. The contract starts with an OpenAPI spec and grows into a streaming, rate-limited, error-aware tool surface across MCP, WebMCP, function-calling, and SDKs - then extends to agent-native payment protocols and a conversational NLWeb endpoint. And "what can I do" spans the whole order lifecycle, not just discovery and checkout: a large share of agent traffic is post-purchase - checking order status, initiating a return, disputing a charge - and those actions run on the same surfaces. + +## A note on tool-call security + +Module 4 is where tool calls run **real business logic** - orders, refunds, payments. Agents that drift from intent (prompt-injection, malformed inputs, runaway loops) cause the most damage here. Three structural defenses keep this manageable: per-tool OAuth scopes ([3.1](./m3-1-oauth-discovery.md), enforced in [4.4](./m4-4-mcp-server.md)), structured rate limits and `Retry-After` ([4.2](./m4-2-rate-limits-and-errors.md)), and continuous end-to-end simulation ([5.4](./m5-4-end-to-end-flows.md)). diff --git a/content/m4-1-openapi-spec.md b/content/m4-1-openapi-spec.md new file mode 100644 index 0000000..73a7db2 --- /dev/null +++ b/content/m4-1-openapi-spec.md @@ -0,0 +1,45 @@ +--- +id: m4-1-openapi-spec +module: actionable +moduleNumber: 4 +guidelineNumber: 1 +title: Ship OpenAPI specification +complexity: 4 +impact: 5 +visualChange: none +forterApplies: 'yes' +--- + +# 4.1 Ship OpenAPI specification + +## What & why +OpenAPI 3.x is the source of truth that every downstream agent artifact compiles from: MCP tools, function-calling schemas, SDKs, CLI commands, plugin manifests. A complete spec - published, linked from an RFC 9727 catalog - lets an agent resolve your entire API surface from your domain alone. Most teams already have a partial spec; the work is filling gaps and tightening descriptions. + +Treat the spec as a **living substrate**, not a frozen deliverable: [4.3](./m4-3-streaming.md) annotates streaming operations in it, [4.8](./m4-8-payment-protocols.md) adds payment metadata, and the SDK, CLI, and MCP pipelines regenerate from it on every change. You author it once here and extend it in place as later guidelines land - that's planned maturation, not rework. + +## Scoring +- **Effort 4/5** - Real schema work across every endpoint, error, and pagination param. Annotation-driven generation helps but doesn't substitute for thinking. +- **Impact 5/5** - Modules 4 and 5 functionally do not exist without this. Every protocol surface downstream depends on it. +- **Visual change: none** - adds `/openapi.json` and `/.well-known/api-catalog` at machine-only paths; user-visible site is unchanged. + +## Steps +1. **Author OpenAPI 3.1** (or 3.0 if your tooling lags) at `/openapi.json` and `/openapi.yaml`. Cover every public endpoint. CI-fail any route that ships without spec coverage. Lint with Spectral against a ruleset that bans `additionalProperties: true` defaults and untyped `object` responses. +2. **Write descriptive `operationId`s and `description`s.** `createOrder` beats `postOrders`. Each operation gets a one-paragraph description in plain English - this is what the LLM reads when picking a tool, and the same text feeds [4.4](./m4-4-mcp-server.md)'s MCP tool descriptions. Tag operations into resource groups so the eventual MCP tool listings stay scannable. +3. **Type every response, including errors.** Define a shared `Error` schema with `type`, `message`, `request_id`, `retry_hint` (see [4.2](./m4-2-rate-limits-and-errors.md)) and reference it from every `4xx` / `5xx` response. No `additionalProperties` escapes; agents cannot infer what isn't declared. +4. **Specify pagination explicitly.** Pick one model - cursor-based is friendliest - and document `cursor`, `limit`, and the response envelope (`data[]`, `next_cursor`, `has_more`) on every list endpoint. +5. **Publish an RFC 9727 API catalog** at `/.well-known/api-catalog`. JSON document linking your OpenAPI file(s), versioning policy, sandbox URLs, and contact metadata. +6. **Negotiate agent-friendly views.** When an agent sends `Accept: text/markdown`, return a markdown rendering of the resource (title, key fields, links) instead of raw JSON, and set `Vary: Accept` so caches keep the JSON and markdown variants apart. For agents and crawlers that can't set a request header, honor a `?mode=agent` query parameter that returns the same stripped-down, markdown-style view of any page or resource. JSON clients and browsers see no change. + +## References +- [OpenAPI Specification 3.1](https://spec.openapis.org/oas/v3.1.0?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 9727 - API Catalog](https://datatracker.ietf.org/doc/html/rfc9727?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Spectral OpenAPI linter](https://github.com/stoplightio/spectral?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [JSON Schema 2020-12](https://json-schema.org/draft/2020-12/release-notes?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +This is a guideline you can hand to Forter almost in full. Rather than authoring and maintaining a public API yourself, you expose your internal tools and capabilities privately to the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-1-openapi-spec) - and only to Forter - and it becomes your API gateway. From the tools you allow, it derives the OpenAPI 3.x spec, writes the request and response schemas, publishes `/openapi.json` and the RFC 9727 catalog on your domain, and keeps the spec aligned as those tools change - no drift, no hand-maintained document. + +Everything downstream is generated from that one exposed surface: MCP tools ([4.4](./m4-4-mcp-server.md)), function-calling schemas, the SDKs and CLI ([4.7](./m4-7-sdks-and-cli.md)), and the rate-limit and error contract ([4.2](./m4-2-rate-limits-and-errors.md)) - rate-limited and version-managed at the gateway. The REST design, schema discipline, SDK distribution, and the ongoing work of keeping integrators current are effectively offloaded. + +What stays yours: the tools themselves and the business logic behind them. You decide what to expose; Forter turns it into a complete, maintained, agent-ready API surface. diff --git a/content/m4-10-nlweb.md b/content/m4-10-nlweb.md new file mode 100644 index 0000000..ae9deb5 --- /dev/null +++ b/content/m4-10-nlweb.md @@ -0,0 +1,46 @@ +--- +id: m4-10-nlweb +module: actionable +moduleNumber: 4 +guidelineNumber: 10 +title: Operate an NLWeb endpoint +complexity: 3 +impact: 3 +visualChange: none +forterApplies: 'yes' +--- + +# 4.10 Operate an NLWeb endpoint + +## What & why +NLWeb is an emerging open standard for turning your site's content into something an agent can *converse with* rather than scrape. You publish your structured content as **Schema Feeds**, an NLWeb server ingests them, and it exposes a single standard endpoint - `POST /ask` - that answers natural-language questions over that content and returns structured JSON. Every NLWeb instance is also an MCP server, so the same `/ask` surface is reachable both as a plain HTTP endpoint and as an MCP tool. Think of it as the conversational counterpart to your sitemap: the sitemap lists your pages, NLWeb answers questions about them. + +It's early - the spec is still moving and adoption is thin - but it's cheap to stand up on top of structured data you should already have ([2.1](./m2-1-json-ld.md)), and it's the cleanest way to let an agent ask "do you carry X?" or "what's your return window?" without crawling. + +## Scoring +- **Effort 3/5** - Publishing Schema Feeds is an afternoon, and a minimal `/ask` that wraps your existing search is modest. The full vector-store-plus-model server is the real work, though the open-source NLWeb toolkit does most of it. +- **Impact 3/5** - Emerging-standard upside. Low today, plausibly central as conversational retrieval matures; the downside risk is near zero given the low cost. +- **Visual change: none** - a `Schemamap` line in robots.txt, a feed file, and an `/ask` endpoint - all machine-only. + +## Steps +1. **Publish Schema Feeds.** Add one line to `robots.txt` - `Schemamap: https://example.com/.well-known/schema-map.xml` - and serve a Schema Map XML pointing at any JSONL or RSS feeds you publish (product feeds, blog feeds, FAQ feeds). The feed items carry Schema.org types, so the structured-data work from [2.1](./m2-1-json-ld.md) is the corpus NLWeb retrieves from. +2. **Start with a minimal `POST /ask`.** The whole protocol reduces to one endpoint, and the minimal viable version needs no vector store and no model - it can forward the incoming question straight to your existing search endpoint or search tool. The request is a JSON body carrying the natural-language `query`; the response is ranked, structured results plus a `_meta` block: + ```json + { + "results": ["...ranked, structured matches..."], + "_meta": { "response_type": "...", "version": "..." } + } + ``` + The two `_meta` fields - `response_type` and `version` - are what let a client confirm it's talking to a conformant NLWeb server. Keep `/ask` public and unauthenticated; friction-free agent retrieval is the whole point. +3. **(Optional) Expand to a full NLWeb server.** When you want genuine natural-language retrieval rather than a search passthrough, adopt the open-source NLWeb toolkit: it ingests your Schema Feeds into a vector store and wires them to an LLM backend. Point it at the feeds from step 1 and re-index on a schedule so answers track your live catalog, not a stale snapshot. +4. **Support streaming.** When the client asks for a streamed response, send results back incrementally as Server-Sent Events instead of one blocking JSON body - the same SSE discipline as [4.3](./m4-3-streaming.md). Long answers feel responsive; short ones cost nothing extra. +5. **Verify both surfaces.** `curl` the `/ask` endpoint with a representative question and confirm the `_meta` fields; then connect to the same server over MCP and confirm the `ask` tool appears. An NLWeb server that fails the MCP handshake is only half-deployed. + +## References +- [NLWeb project](https://github.com/microsoft/NLWeb?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Schema.org vocabulary](https://schema.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [HTML Living Standard - Server-Sent Events](https://html.spec.whatwg.org/multipage/server-sent-events.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +If standing up a vector store, an ingestion pipeline, and a model backend is more than you want to own, the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-10-nlweb) can host the NLWeb server for you. Point it at your Schema Feeds and Forter publishes a conformant `POST /ask` endpoint under your domain - reachable both as plain HTTP and as the MCP tool every NLWeb instance also exposes - and re-ingests on a schedule so answers track your live catalog. diff --git a/content/m4-2-rate-limits-and-errors.md b/content/m4-2-rate-limits-and-errors.md new file mode 100644 index 0000000..768acf7 --- /dev/null +++ b/content/m4-2-rate-limits-and-errors.md @@ -0,0 +1,38 @@ +--- +id: m4-2-rate-limits-and-errors +module: actionable +moduleNumber: 4 +guidelineNumber: 2 +title: Standardize rate limits & errors +complexity: 2 +impact: 5 +visualChange: none +forterApplies: 'yes' +--- + +# 4.2 Standardize rate limits & errors + +## What & why +Agents that can self-throttle don't get blocked. Agents that can parse your errors retry intelligently; agents that can't, give up. Rate-limit headers on every response and structured error envelopes on every failure cost almost nothing to add, follow conventions agents already expect, and make the difference between an agent that works overnight and one that pages a human at 3 a.m. Rate limits are also your **defense against agent storms** - runaway loops, prompt-injected drift, retry cascades - so every response gets headers, not just the ones close to a limit. + +## Scoring +- **Effort 2/5** - Headers and a shared error schema. A day's middleware work in any modern framework, plus a status-page endpoint. +- **Impact 5/5** - The single highest-leverage runtime guideline. Agents fail loudly without it and recover gracefully with it. +- **Visual change: none** - HTTP headers and JSON error bodies; nothing user-visible changes. + +## Steps +1. **Emit rate-limit headers on every response - success or failure.** `X-RateLimit-Limit`, `X-RateLimit-Remaining`, and `X-RateLimit-Reset` (Unix epoch seconds). On `429`, also send `Retry-After` (seconds, not HTTP-date - it's simpler to parse). Document the bucket scope per endpoint in OpenAPI. +2. **Define one shared `Error` schema and use it everywhere.** Required fields: `type` (a stable URI or short token like `rate_limited`, `validation_failed`, `insufficient_funds`), `message` (one sentence), `request_id`, and `retry_hint` (`retry_now` | `retry_after_seconds:N` | `do_not_retry`). Optional: `details[]` for field-level validation errors. +3. **Match HTTP status to error type honestly.** `400` validation, `401` auth, `403` scope, `404` resource, `409` conflict, `422` semantic, `429` rate limit, `5xx` your bug. Agents route retries off the status code first and the `type` field second; lying about either breaks the recovery loop. +4. **Publish a status page at a stable URL** (`/status` or `status.yourdomain.com`) that returns JSON when called with `Accept: application/json`. Schema: `status` (`operational` | `degraded` | `outage`), `incidents[]`, `last_updated`. Agents poll this before assuming a `5xx` is their problem. +5. **Document the contract.** A "Rate limits and errors" page in your developer docs with one table of all `type` values, their meanings, and the recommended retry strategy. + +## References +- [RFC 6585 - Additional HTTP Status Codes](https://datatracker.ietf.org/doc/html/rfc6585?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 7231 - Retry-After](https://datatracker.ietf.org/doc/html/rfc7231?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide#section-7.1.3) +- [RFC 9457 - Problem Details for HTTP APIs](https://datatracker.ietf.org/doc/html/rfc9457?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [GitHub API rate-limit headers](https://docs.github.com/en/rest/overview/resources-in-the-rest-api?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide#rate-limiting) + +## How Forter helps + +When you expose your internal tools to the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-2-rate-limits-and-errors), it publishes them as a public API in the form this guideline describes - and rate limiting and error standardization are handled at the gateway. Forter emits `X-RateLimit-*` and `Retry-After` headers on every response, enforces per-tenant throttling, and wraps every failure in the standard `{type, message, request_id, retry_hint}` envelope, so the agent-facing contract is correct and consistent without your origin emitting a single header. diff --git a/content/m4-3-streaming.md b/content/m4-3-streaming.md new file mode 100644 index 0000000..aafcba5 --- /dev/null +++ b/content/m4-3-streaming.md @@ -0,0 +1,39 @@ +--- +id: m4-3-streaming +module: actionable +moduleNumber: 4 +guidelineNumber: 3 +title: Stream long-running operations +complexity: 5 +impact: 4 +visualChange: none +forterApplies: 'partial' +--- + +# 4.3 Stream long-running operations + +## What & why +Anything an agent waits on for more than five seconds needs progress feedback, or the agent assumes it's broken and retries - usually duplicating side effects. Server-Sent Events for in-band progress, chunked transfer for large bodies, and an explicit cancellation path turn a 30-second risk evaluation from "feels broken" into "feels real-time." And for work that outlives any reasonable open connection - minutes to hours - the agent shouldn't hold a socket at all: it registers a **webhook** and gets called back when the result is ready. + +## Scoring +- **Effort 5/5** - A wide surface across the whole request path: SSE and chunked transfer in the application servers, buffer-flushing and idle-timeout tuning through every proxy and load balancer, idempotency on retries, and a full webhook delivery subsystem with signing, backoff, and redelivery. Frameworks cover pieces, not the whole. +- **Impact 4/5** - Critical for fraud decisions, long-form generation, batch operations, and any synchronous action over a few seconds. Without it, agents hit timeouts and double-submit. +- **Visual change: none** - wire-level changes only; user-visible site is unaffected. + +## Steps +1. **Mark streaming operations in OpenAPI.** Use `text/event-stream` as the response content type and an `x-streaming: true` extension on the operation. Document the event schema: `event` name (`progress` | `partial` | `complete` | `error`), `data` payload, and the terminal event that closes the stream. Function-calling clients and MCP generators key off this to wire up the right transport. +2. **Implement SSE for progress.** On long operations, hold the connection open and emit `event: progress\ndata: {"percent": 40, "stage": "scoring"}\n\n` every 1-3 seconds. End with `event: complete\ndata: {...final result...}\n\n`. Flush after every event - buffered SSE is broken SSE. Set `Cache-Control: no-cache` and `X-Accel-Buffering: no` for nginx in front. +3. **Use chunked transfer encoding for large response bodies.** Lists, exports, and aggregations stream JSON Lines (`application/x-ndjson`) one record per line so an agent can process incrementally. +4. **Support cancellation.** When the client disconnects, abort the underlying work and emit a final `event: cancelled` if you can. Issue an idempotency key on the initial request so a retry after cancellation doesn't re-charge or re-evaluate. Document the cancellation contract in the operation's OpenAPI description. +5. **Offer webhooks for work that outlives a connection.** For operations measured in minutes or hours, let the caller register a callback URL instead of holding a stream open. Document the event types, payload schema, and delivery semantics in OpenAPI; sign every delivery (HMAC over the body, signature in a header) so the receiver can verify provenance; retry failed deliveries with exponential backoff and expose a redelivery endpoint for the ones that still miss. Reuse the event names from step 1 (`progress`, `complete`, `error`) so an agent handles a webhook payload and an SSE frame with the same code. + +## References +- [HTML Living Standard - Server-Sent Events](https://html.spec.whatwg.org/multipage/server-sent-events.html?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [RFC 7230 §4.1 - Chunked Transfer Coding](https://datatracker.ietf.org/doc/html/rfc7230?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide#section-4.1) +- [JSON Lines specification](https://jsonlines.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP Streamable HTTP transport](https://modelcontextprotocol.io/specification/2025-06-18/basic/transports?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide#streamable-http) +- [Standard Webhooks specification](https://www.standardwebhooks.com?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +The orchestration suite proxies SSE and chunked responses end-to-end without buffering. Connection idle timeouts are tuned for agent workloads (10+ minute streams). Cancellation propagates from the agent through the gateway to your origin. Progress events pass through unchanged; the gateway adds correlation headers so each progress frame is traceable to the originating MCP tool call. Webhook deliveries are relayed and HMAC-verified at the gateway, so a callback's provenance is checked before the agent acts on it. diff --git a/content/m4-4-mcp-server.md b/content/m4-4-mcp-server.md new file mode 100644 index 0000000..0985ad1 --- /dev/null +++ b/content/m4-4-mcp-server.md @@ -0,0 +1,42 @@ +--- +id: m4-4-mcp-server +module: actionable +moduleNumber: 4 +guidelineNumber: 4 +title: Operate an MCP server +complexity: 4 +impact: 5 +visualChange: none +forterApplies: 'flagship' +--- + +# 4.4 Operate an MCP server + +## What & why +This is **the** integration point for native agent invocation. With an MCP server on Streamable HTTP, ChatGPT, Claude, Gemini, and any function-calling LLM can call your tools natively - no scraping, no glue code, no "and then we ask the user to paste an API key." Pair the server with a published server-card and you've covered how today's major agent runtimes reach your business logic. (Browser-resident agents are a separate surface - that's WebMCP, [4.5](./m4-5-webmcp.md).) + +It's also the layer where agent drift has the most consequence - MCP tool calls run **real business logic** (orders, refunds, payments). Per-tool OAuth scopes (from [3.1](./m3-1-oauth-discovery.md)) bound what any single call can do, even if the calling agent is steered off-course by prompt injection in a retrieved page. Operationally substantial enough that most teams pair with Forter to ship the gateway side. + +## Scoring +- **Effort 4/5** - A real MCP server (Streamable HTTP, OAuth-bound sessions, per-tool authorization), plus tool annotations and read-only resources, plus WebMCP, plus a published server-card, plus per-invocation observability. Off-the-shelf MCP frameworks help. +- **Impact 5/5** - The single most leveraged endpoint in the guide. Every agent runtime that matters consumes MCP first. +- **Visual change: none** - `/mcp` endpoint and `/.well-known/mcp/server-card.json` are machine-only. No user-visible page changes. + +## Steps +1. **Implement an MCP server with Streamable HTTP transport.** Mount it at `/mcp` (or `mcp.yourdomain.com`). Compile your tool list directly from the OpenAPI spec from [4.1](./m4-1-openapi-spec.md): each `operationId` becomes a tool, each request schema becomes the tool's input schema, each response schema becomes the output schema. Manual tool authoring drifts; generation does not. +2. **Bind every tool call to an OAuth bearer token.** Reuse the auth server from [3.1](./m3-1-oauth-discovery.md). Tools execute under the caller's scopes - an agent with `orders:read` cannot invoke `payments:write`, even if it tries. Reject calls with no token, expired tokens, or insufficient scope using the same structured error envelope from [4.2](./m4-2-rate-limits-and-errors.md) (`type: "insufficient_scope"`, `retry_hint: "do_not_retry"`). This is the layer that contains blast radius if an agent drifts from intent. +3. **Publish a server-card at `/.well-known/mcp/server-card.json`.** Required fields: `name`, `description`, `version`, `serverUrl`, `transport: "streamable-http"`, `authorization` (linking to your RFC 8414 / 9728 metadata from [3.1](./m3-1-oauth-discovery.md)), and `tools[]` with `{name, description, inputSchema, outputSchema}`. +4. **Declare behavioral annotations on every tool.** Tag each tool with `annotations.readOnlyHint` and `annotations.destructiveHint` so the host knows which calls are safe to retry or auto-approve and which mutate state - `getOrderStatus` is `readOnlyHint: true`, while `cancelOrder`, `initiateReturn`, and `disputeCharge` are `destructiveHint: true`. Agents read these to decide when to pause for user confirmation; an unannotated mutating tool gets either blocked or called with no safety prompt. +5. **Expose read-only context as MCP resources.** Tools are for actions; **resources** are for context an agent reads *before* acting - catalogs, pricing tables, status snapshots, docs. Advertise the `resources` capability in the `initialize` handshake and implement `resources/list` and `resources/read`. Each resource needs a stable URI, an accurate `mimeType`, and a non-empty body. An agent that can read `pricing://current` as a resource doesn't have to spend a tool call (and a scope) just to answer "what does this cost?". +6. **Observe every invocation.** Log every tool call with `{tool_name, client_id, sub, request_id, latency_ms, status, error_type}`. Aggregate to a per-tool latency / error budget you alert on. This is your forensic audit trail - if an agent drifts and chains tool calls in unintended ways, this is the lens that catches it. (Once observability is solid, list the server in registries - mcp.run, mcphub.io, Smithery, skills.sh - per [4.6](./m4-6-agent-registries.md). Consumer AI platforms - GPT Store, Custom GPTs, Claude, Gemini - are [5.1](./m5-1-verified-on-platforms.md).) + +## References +- [Model Context Protocol specification](https://modelcontextprotocol.io/specification?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP Streamable HTTP transport](https://modelcontextprotocol.io/specification/2025-06-18/basic/transports?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide#streamable-http) +- [MCP Server Card](https://modelcontextprotocol.io/community/server-card/charter?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP resources](https://modelcontextprotocol.io/docs/concepts/resources?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP Registry](https://github.com/modelcontextprotocol/registry?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +A production MCP server with Streamable HTTP, operated for you and exposed under your domain. Tools auto-generated from your OpenAPI on every release. Server-card published at your `/.well-known/mcp/server-card.json` via reverse-proxy. Every tool call runs behind the OAuth server Forter operates in [3.1](./m3-1-oauth-discovery.md) - per-tool scopes are enforced at the gateway before a call reaches your origin. Behavioral tool annotations (`readOnlyHint` / `destructiveHint` derived from your spec's HTTP verbs) and read-only MCP resources included. MCP-registry submissions (mcp.run, mcphub.io, Smithery) handled and re-verified on every release. Per-invocation observability with dashboards and alerting wired in. What would otherwise be a substantial engineering build becomes an integration. \ No newline at end of file diff --git a/content/m4-5-webmcp.md b/content/m4-5-webmcp.md new file mode 100644 index 0000000..9f2f825 --- /dev/null +++ b/content/m4-5-webmcp.md @@ -0,0 +1,58 @@ +--- +id: m4-5-webmcp +module: actionable +moduleNumber: 4 +guidelineNumber: 5 +title: Expose tools with WebMCP +complexity: 2 +impact: 3 +visualChange: none +forterApplies: 'partial' +--- + +# 4.5 Expose tools with WebMCP + +## What & why +WebMCP is the lowest-friction way to make a website agentic. Where an MCP server ([4.4](./m4-4-mcp-server.md)) is backend infrastructure - WebMCP is a browser API. A few lines of JavaScript on a page register **tools**, and a browser-resident agent (Chrome with Gemini, or a local model) calls them directly, inside the user's own session, with no server, no separate auth, and no API program required. + +It is an emerging Google and Microsoft W3C proposal - early, but cheap to adopt. It reuses the same tool model as MCP (`name`, `description`, `inputSchema`, `annotations`), so the descriptions and schemas written for [4.4](./m4-4-mcp-server.md) carry straight over - and a site with no MCP server at all can still adopt WebMCP on its own. + +## Scoring +- **Effort 2/5** - Front-end work only: register tools in page JavaScript, write clear descriptions and input schemas. No server, no auth server, no OpenAPI prerequisite. The spec is still maturing, so budget for some API churn. +- **Impact 3/5** - Emerging-standard upside. Browser support is still landing, but WebMCP is the cheapest path to being callable by in-browser agents, and the downside risk is near zero given the cost. +- **Visual change: none** - tools are registered in JavaScript; the page renders exactly as before. + +## Steps +1. **Register tools with `registerTool()`.** Call it on the page's model-context object. The current W3C spec exposes it as `document.modelContext`; earlier Chrome builds and the MCP-B polyfill use `navigator.modelContext` (which also carries an older `provideContext()` form that swaps the whole toolset at once), so feature-detect both before using it. You declare each tool in page JavaScript - you decide exactly which actions are exposed. A tool needs a `name`, a natural-language `description`, an `inputSchema` (JSON Schema for its parameters), and an `execute` callback that does the work and returns a Promise: + ```js + const mc = document.modelContext || navigator.modelContext; + mc.registerTool({ + name: "search_products", + description: "Search the catalog by keyword and return matching products.", + inputSchema: { + type: "object", + properties: { query: { type: "string" } }, + required: ["query"] + }, + annotations: { readOnlyHint: true }, + async execute({ query }) { + const results = await searchCatalog(query); + return { content: [{ type: "text", text: JSON.stringify(results) }] }; + } + }); + ``` +2. **Annotate tools so the agent knows what is safe.** Set `annotations.readOnlyHint` on tools that only read state, and `annotations.untrustedContentHint` on tools that return data you do not control. A browser agent reads these to decide what it can call on its own and what to treat with caution. +3. **Gate consequential actions on user confirmation.** A tool's `execute` callback receives a `ModelContextClient`; call `client.requestUserInteraction()` before anything that spends money or mutates account state. The user is already in the browser - put them in the loop rather than letting the agent commit silently. +4. **Register and unregister tools to match page state.** The tool list should reflect what the current view can actually do: register tools as a view mounts and pass an `AbortSignal` (`registerTool(tool, { signal })`) so they are removed when the user navigates away. An agent offered a stale tool will call it and fail. + +## References +- [WebMCP proposal - W3C](https://webmachinelearning.github.io/webmcp/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Model Context Protocol specification](https://modelcontextprotocol.io/specification?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [JSON Schema](https://json-schema.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [The WebMCP Directory](https://webmcp.cool/) + +## How Forter helps + +The [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-5-webmcp) generates the in-page `registerTool` bindings from the same tool definitions it already builds your MCP server ([4.4](./m4-4-mcp-server.md)) from - so your browser-side and server-side tool surfaces are described once and stay in sync. The bindings, the `execute` callbacks, and the page-state lifecycle (registering on view mount, unregistering on navigation) are generated for you, not handed to you as a starting point. + +Forter helps to **embed the bindings into your existing site** - a single script tag on the pages you choose, with no application rewrite. Once embedded, the tools you expose become callable by every browser-resident or AI agent that supports WebMCP, alongside the same surface your MCP server already serves to remote agents. WebMCP is still a moving proposal, so Forter tracks spec revisions and updates the generated bindings as the API stabilizes - your integration moves with the standard, not against it. diff --git a/content/m4-6-agent-registries.md b/content/m4-6-agent-registries.md new file mode 100644 index 0000000..ee3ae30 --- /dev/null +++ b/content/m4-6-agent-registries.md @@ -0,0 +1,37 @@ +--- +id: m4-6-agent-registries +module: actionable +moduleNumber: 4 +guidelineNumber: 6 +title: List in agent registries +complexity: 1 +impact: 3 +visualChange: low +forterApplies: 'yes' +--- + +# 4.6 List in agent registries + +## What & why +With the MCP server live ([4.4](./m4-4-mcp-server.md)) and the OpenAPI clean ([4.1](./m4-1-openapi-spec.md)), you finally have something to publish. Agents discover capabilities through **registries** - not by crawling - and four of them matter: **mcp.run, mcphub.io, Smithery** (MCP-server marketplaces) and **skills.sh** (a domain-level skill catalog with first-class indexing in Claude and Cursor). Listing is free, the forms are short, and the discipline is keeping descriptions accurate as your tool surface evolves. + +## Scoring +- **Effort 1/5** - Four submissions, ~15 minutes each. Quality of the listings is bounded by quality of the underlying spec, so most of the work happened in [4.1](./m4-1-openapi-spec.md) and 4.4. +- **Impact 3/5** - Without listings you're invisible to capability discovery; with them you appear in the picker every time a user asks for "tools that do X." +- **Visual change: low** - listings live on third-party registries, not on your site. + +## Steps +1. **mcp.run and mcphub.io.** Submit your MCP server URL, a one-paragraph description, your tool list, and a logo. Both pull tool metadata from your server's `tools/list` endpoint, so the `description` on each MCP tool is what users actually read - write them like API docs, not marketing copy. Tool description quality traces back to your OpenAPI from [4.1](./m4-1-openapi-spec.md). +2. **Smithery.** The largest MCP marketplace. Adds a `smithery.yaml` at your repo root declaring runtime, env vars, and start command - that's what enables one-click installs in Cursor and Windsurf. +3. **skills.sh - register your domain.** Claim your domain at skills.sh and publish a top-level `SKILL.md` per public repo listing every callable skill with `name`, `description`, `inputs`, `outputs`, and an `example` block. Format is markdown with frontmatter - borrow from any well-known repo (e.g., `stripe/stripe.com/SKILL.md`). +4. **skills.sh - quality signals.** The registry ranks listings on three signals: (a) **multiple repos** under the same domain, each with its own `SKILL.md`; (b) **descriptions that pass an LLM rubric** for clarity (no "powerful, easy-to-use platform" filler); and (c) **freshness** - `SKILL.md` updated within 90 days. Sites with one thin `SKILL.md` rank below sites with five focused ones. +5. **Verify and monitor.** After each submission, search the registry for your brand name and a representative use-case query. Set a 30-day reminder to re-verify - registries periodically re-crawl and silently de-list servers that 5xx or change shape. (Consumer AI platforms - GPT Store, Custom GPTs, Claude integrations, Gemini extensions - are 5.1's job, with their own review cycle.) + +## References +- [Model Context Protocol Registry](https://github.com/modelcontextprotocol/registry?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Smithery - publishing servers](https://smithery.ai/docs/build?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [skills.sh - SKILL.md spec](https://skills.sh?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +If your MCP server runs on the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-6-agent-registries), registry submission can be handled for you - with descriptions auto-generated from your OpenAPI document and tool annotations. Re-verification runs on every release, so silent de-listings (the most common registry failure mode) get caught before they cost you discoverability. diff --git a/content/m4-7-sdks-and-cli.md b/content/m4-7-sdks-and-cli.md new file mode 100644 index 0000000..0b39bb7 --- /dev/null +++ b/content/m4-7-sdks-and-cli.md @@ -0,0 +1,41 @@ +--- +id: m4-7-sdks-and-cli +module: actionable +moduleNumber: 4 +guidelineNumber: 7 +title: Distribute SDKs & CLI +complexity: 3 +impact: 4 +visualChange: none +forterApplies: 'yes' +--- + +# 4.7 Distribute SDKs & CLI + +## What & why +Agents and the developers building them reach for SDKs and CLIs first, raw HTTP second. Without an idiomatic SDK in the language an integrator (or an LLM writing code on their behalf) is using, every consumer rolls their own client and gets at least one of pagination, retries, error parsing, or auth wrong. With SDKs in npm and PyPI - plus a CLI on npm and Homebrew - integration is one install away. + +This is also where the OpenAPI investment from [4.1](./m4-1-openapi-spec.md) pays a second dividend: a complete spec generates both SDKs and a CLI for the cost of one. + +## Scoring +- **Effort 3/5** - Assumes the OpenAPI spec from [4.1](./m4-1-openapi-spec.md) is already published; if it isn't, that work comes first. Given the spec, the hard part is the publishing pipeline - signing, versioning, and per-registry credentials. Generation itself is largely solved. +- **Impact 4/5** - Major integration accelerator; LLMs writing integration code reach for `import stripe` before they reach for `requests`. +- **Visual change: none** - SDKs ship to package registries, not your site. + +## Steps +1. **Pick one OpenAPI generator and commit.** Stainless, Speakeasy, Fern, or `openapi-generator` are the credible options. Test each against your spec from [4.1](./m4-1-openapi-spec.md) - the one whose output you'd be willing to hand-edit is the one to pick. Switching mid-stream costs months. +2. **Ship the npm and PyPI SDKs.** TypeScript on npm and Python on PyPI cover ~80% of agent and integration code. Idiomatic naming (`client.transactions.create({...})`, not `client.postTransactions(...)`), typed responses, automatic retries with exponential backoff that respect the `Retry-After` headers from [4.2](./m4-2-rate-limits-and-errors.md), and pagination iterators (`for await (const tx of client.transactions.list())`). +3. **Distribute a CLI on npm and Homebrew.** Mirror your SDK surface - `yourbrand transactions create --amount 1000` - plus auth helpers (`yourbrand login` doing the OAuth device flow from [3.1](./m3-1-oauth-discovery.md)), config management, and a `--json` flag for piping into agent workflows. +4. **Auto-publish on every spec release.** CI pipeline: spec change merges, generator runs, both SDKs and the CLI build, tests pass against a sandbox, version bumps semantically, changelogs generate from spec diffs, packages publish to all three registries, GitHub release goes out. + +## References +- [OpenAPI Generator](https://openapi-generator.tech?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [npm publishing docs](https://docs.npmjs.com/cli/commands/npm-publish?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [PyPI publishing guide](https://packaging.python.org/en/latest/tutorials/packaging-projects/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Homebrew tap creation](https://docs.brew.sh/How-to-Create-and-Maintain-a-Tap?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Fern](https://buildwithfern.com?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Speakeasy](https://www.speakeasy.com?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +Run on the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-7-sdks-and-cli) and the SDK and CLI layer can be generated for you. From the tool surface you expose, Forter builds and publishes idiomatic SDKs in the popular languages - ready to link straight from your developer docs - and stands up an authenticated API that your users and their agents can call directly. Auth, retries, error handling, and pagination come wired in, and every SDK regenerates when your tools change. diff --git a/content/m4-8-payment-protocols.md b/content/m4-8-payment-protocols.md new file mode 100644 index 0000000..12f7631 --- /dev/null +++ b/content/m4-8-payment-protocols.md @@ -0,0 +1,44 @@ +--- +id: m4-8-payment-protocols +module: actionable +moduleNumber: 4 +guidelineNumber: 8 +title: Support agent payment protocols +complexity: 4 +impact: 2 +visualChange: none +forterApplies: 'partial' +--- + +# 4.8 Support agent payment protocols + +## What & why +When an agent has to settle a payment at the moment of an HTTP call - buying metered API access, paying for compute, or checking out an order - it has to do so without a human typing a card number. Agent-native payment splits into a **settlement** layer that answers the HTTP `402 Payment Required` handshake (x402, MPP) and an **authorization** layer that proves the user actually approved the purchase (AP2): + +- **x402** (Coinbase) is the minimal, crypto-native case: one request, one stablecoin payment (USDC, usually on Base), one response. It is purpose-built for machine-to-machine metering - API calls, data feeds, compute, agent-to-agent services - and is **stablecoin-only**, so it does not reach card/bank rails or physical-goods checkout. +- **MPP** (Machine Payments Protocol, from Stripe and Tempo, now an IETF Internet-Draft) generalizes the same `402` exchange into a **payment-method-agnostic** framework. A single endpoint can accept stablecoins, cards, bank transfers, even Bitcoin, and it carries two intents: `charge` (one-shot - which maps directly onto an x402 payment, making MPP backwards-compatible with it) and `session` (pre-authorize a spending limit once, then stream granular micropayments). That breadth means MPP is **not** limited to per-call metering - the same handshake settles for **physical goods and services across crypto *and* traditional rails**. +- **AP2** (Agent Payments Protocol, Google) sits a layer up. Instead of settling a `402`, it carries **cryptographic proof that the user authorized this purchase** - a signed *Mandate* (an SD-JWT *Checkout Mandate* for what was authorized, in open/pre-authorization or closed/final form, plus a *Payment Mandate* for the funding instrument). Built as an extension of A2A and advertised on your `agent-card.json`, it's payment-method-agnostic and **composes** with the settlement layer rather than replacing it: AP2 *authorizes*, x402 and MPP *settle*. + +These are early standards - adoption is growing fast but the specs still move - so treat this as forward-positioning, not table stakes. + +## Scoring +- **Effort 4/5** - Payment middleware on every priced route, a funded wallet, facilitator wiring, and settlement reconciliation - plus the standing operational weight of holding and securing a wallet. The protocols are still in flux, so expect spec churn on top. +- **Impact 2/5** - Forward-looking. Real for sites selling metered access or accepting agent-native checkout today, but adoption is early and the downside of waiting is low. +- **Visual change: none** - `/.well-known/*` discovery files and HTTP `402` responses on paid routes; human-visible pages don't change. + +## Steps +1. **x402 for crypto-native pay-per-request.** x402 negotiates payment inline over HTTP, settled in stablecoins on crypto rails. When an agent calls a priced route, the server answers `402 Payment Required` with the payment terms - amount, accepted scheme, and where to pay. The agent settles those terms - typically through a *facilitator* that brokers and verifies the transfer - and retries with proof of payment, which returns the real response. To support it, mark your paid routes, point them at a facilitator and a receiving wallet, and advertise which routes are priced so agents can find the paid surface before calling. Open-source middleware for the common web frameworks makes the wiring mostly configuration. +2. **MPP for multi-rail and physical-goods settlement.** MPP answers the `402` with a `WWW-Authenticate: Payment` challenge whose `method` (`tempo`, `stripe`, `card`, `lightning`, ...) and `intent` (`charge` for one-shot, `session` for streaming) pick the rail and the payment shape - so one integration covers stablecoins, cards, and fiat and reaches physical-goods checkout, not just metered calls. Annotate payable operations in your `/openapi.json` ([4.1](./m4-1-openapi-spec.md)) so agents can discover priced surfaces, and wire the MPP middleware (SDKs ship for TypeScript, Python, and Rust) to handle the handshake and settlement. +3. **AP2 for proof of authorization.** If agents transact on a user's behalf, advertise AP2 support as an A2A extension on your `/.well-known/agent-card.json` (extension URI `https://github.com/google-agentic-commerce/ap2/v1`) and accept the signed Mandates on your checkout path. This is what lets a merchant or card network trust that the absent buyer really authorized *this* cart at *this* price - the trust layer that x402/MPP settlement rides on. +4. **Reuse your auth and error stack.** Every payment route sits behind the OAuth scopes from [3.1](./m3-1-oauth-discovery.md) and returns the structured error envelope from [4.2](./m4-2-rate-limits-and-errors.md). A declined payment is `type: "payment_required"` or `type: "payment_declined"` with an honest `retry_hint` - never a bare `500`. + +## References +- [x402 protocol](https://x402.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MPP - Machine Payments Protocol](https://mpp.dev?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [AP2 - Agent Payments Protocol](https://ap2-protocol.org?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +The hardest part of agent payments is not the `402` handshake - it is everything around the wallet: creating one, funding and maintaining it, and satisfying KYC / KYA obligations on the parties transacting through it. The [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-8-payment-protocols) speaks x402 and MPP at the gateway and can take that wallet burden on - standing up and maintaining the wallet behind your priced routes, with identity and KYC/KYA checks drawn from the Forter Identity Network applied to each settlement. + +This part of the orchestration suite is an early-stage offering and moves with the protocols themselves - scope it with Forter for what is production-ready today versus on the roadmap. What stays yours: your pricing, and the processor relationship behind settlement. diff --git a/content/m4-9-commerce-protocols.md b/content/m4-9-commerce-protocols.md new file mode 100644 index 0000000..dad573a --- /dev/null +++ b/content/m4-9-commerce-protocols.md @@ -0,0 +1,49 @@ +--- +id: m4-9-commerce-protocols +module: actionable +moduleNumber: 4 +guidelineNumber: 9 +title: Support agentic commerce protocols +complexity: 4 +impact: 4 +visualChange: none +forterApplies: 'yes' +--- + +# 4.9 Support agentic commerce protocols + +## What & why +When an agent shops on a user's behalf - finds a product, builds a cart, checks out, and after the sale checks order status, initiates a return, or disputes a charge - it needs a structured commerce surface, not a scrape of your storefront HTML. Two protocols cover this: **UCP** (Universal Commerce Protocol) and **ACP** (Agentic Commerce Protocol). Both let an agent discover a catalog, assemble a cart, and complete a purchase against a defined contract. This guideline applies if an agent could browse your products and check out on a user's behalf; if you only sell metered, pay-per-call access, that is [4.8](./m4-8-payment-protocols.md) instead, and if you sell nothing at an agent call, skip both. + +These standards are still settling, but adoption is real and accelerating - merchants are transacting through Google, Gemini, and AI-mode surfaces today. + +## Scoring +- **Effort 4/5** - A real commerce surface: catalog or feed submission, cart state, a checkout flow, and settlement wiring - all on specs that keep moving. +- **Impact 4/5** - Emerging but transformative: for anyone selling products to agents, this is fast becoming where the sale happens. +- **Visual change: none** - discovery and checkout endpoints; user-visible pages don't change. + +## Steps +1. **Support UCP.** Publish a `/.well-known/ucp` discovery file declaring your services, capabilities, and endpoints. UCP builds on Google's shopping graph, so your products must also be listed - and kept current - as a catalog in Google Merchant Center. Then implement `/checkout-sessions` in full compliance with the UCP spec, every request and response shape it defines, so an agent can assemble a cart and complete the purchase through what UCP calls the *payment handler*. +2. **Support ACP.** Publish a `/.well-known/acp.json` discovery file, and submit your product catalog to OpenAI so ACP-driven agents can discover your items. Then implement `/checkout_sessions` in full compliance with the ACP spec, every request and response it defines, so an agent can run the checkout end to end and settle through what ACP calls *delegated payment*. +3. **Keep the platforms in sync with webhooks.** A completed checkout is only the start of the order's life. Send webhooks back to the agent platform on order completion, cancellation, and every order-status change (shipped, delivered, refunded) so the agent - and the user it acts for - always sees current state, not a stale snapshot. +4. **Move guests to registered users.** A protocol checkout defaults to guest checkout. Add identity linking and/or OAuth ([3.1](./m3-1-oauth-discovery.md)) so an agent's purchase can attach to a real, returning customer account - unlocking order history, saved preferences, and loyalty instead of leaving every agent sale anonymous. + +## References +- [UCP - Universal Commerce Protocol](https://ucp.dev/specification/overview/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [ACP - Agentic Commerce Protocol](https://www.agenticcommerce.dev/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Google Merchant Center](https://merchants.google.com/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +Agentic commerce is the part of this module the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m4-9-commerce-protocols) can shoulder most completely - and it scales to however much of the stack you want to hand over. + +- **Product feed** - Forter can build and maintain the catalog feed UCP and ACP run on, or work from the feed you already produce. +- **Tokenization and settlement** - Forter is a PCI-compliant tokenizer for the settlement step of both protocols, *and* a PSP-agnostic payment orchestrator that can authorize and capture funds on your behalf - or it slots in alongside the tokenization and processor you already use. +- **Cart and checkout** - Forter runs the cart and checkout flows for you, working *inside* your existing OMS and internal systems rather than around them. +- **Webhooks and identity** - if you are already a Forter customer with the orchestration suite integrated, the order webhooks and status updates of step 3 are handled for you, and Forter's identity linking is built in - so steps 3 and 4 land with little extra work. +- **Post-purchase** - returns, refunds, and chargeback disputes are where loss concentrates after the sale, and agents will initiate all three on a user's behalf. Forter's **Abuse Prevention** scores return and refund requests for abuse before you approve them; **Dispute Management** handles chargeback representment and recovery on the disputes that land - both drawing on the same Forter Identity Network signal that cleared the checkout. +- **Spec drift** - UCP and ACP are still moving; Forter tracks every revision so your integration doesn't break when a spec changes underneath it. + +The deeper value is in what these protocols *don't* carry. As they stand, neither UCP nor ACP surfaces every signal a sound validation and fraud decision needs. Forter collects the missing pieces - identity, device reputation, behavioral history, agent provenance - and assembles them, alongside the cart, into one coherent validation and fraud-check call. That mapping-and-collection work is substantial and easy to underestimate; offloading it is the difference between a checkout that merely completes and one you can trust. Every checkout the orchestration suite brokers carries the same Forter Identity Network signal that runs human card-not-present commerce. + +What stays yours: your catalog, your pricing, and the decision of how much to bring versus hand over. diff --git a/content/m5-0-module-experiential.md b/content/m5-0-module-experiential.md new file mode 100644 index 0000000..e0892c9 --- /dev/null +++ b/content/m5-0-module-experiential.md @@ -0,0 +1,24 @@ +--- +id: module-experiential +title: Module 5 - Be Experiential +kind: module-overview +moduleNumber: 5 +--- + +# Module 5 - Be Experiential +*"Agents close the loop and render value back to the human."* + +## The agent's question +> _"I authenticated, I called the API, I got a result. How do I show this to the user in a way that completes the intent?"_ + +Modules 1-4 are infrastructure. Module 5 is **product** - where an agent-driven flow becomes a user experience that competes with a traditional web visit. A site that nails the rest but skips Module 5 has built a beautiful API that nobody sees. + +## Three planes of experience + +1. **Inline UI.** MCP Apps lets your tool render an interactive component **inside the chat conversation** - payment picker, confirmation, comparison. +2. **Marketplace presence.** ChatGPT GPT Store and **Custom GPTs**, Claude integrations, Gemini extensions. Verified-integration badges measurably increase recommendation rates. +3. **Conversational coverage.** Multi-turn flows that don't dead-end on missing pricing or comparison content. + +## A note on experiential security + +Inline UI renders inside a host environment shared with components from other tools. Treat your bundle like any other public surface: signed, CSP-locked, network-scoped to your origin. Continuous simulation ([5.4](./m5-4-end-to-end-flows.md)) catches host-runtime drift before it reaches users. diff --git a/content/m5-1-verified-on-platforms.md b/content/m5-1-verified-on-platforms.md new file mode 100644 index 0000000..9dd2758 --- /dev/null +++ b/content/m5-1-verified-on-platforms.md @@ -0,0 +1,41 @@ +--- +id: m5-1-verified-on-platforms +module: experiential +moduleNumber: 5 +guidelineNumber: 1 +title: Get verified on AI platforms +complexity: 3 +impact: 5 +visualChange: low +forterApplies: 'yes' +--- + +# 5.1 Get verified on AI platforms + +## What & why +Three submission paths matter most: the **ChatGPT GPT Store** (which reads your `/.well-known/ai-plugin.json` and powers user-built **Custom GPTs** that consume your plugin), the **Claude integrations directory** (which lists your MCP server, see [4.4](./m4-4-mcp-server.md)), and **Gemini extensions**. Each one stamps a verified-integration badge on listings that pass review - and that badge measurably increases tool-selection rates *and* serves as a trust signal users see directly. Spoofed integrations are a known phishing surface; verification is what distinguishes legitimate from impersonator. + +The work itself is mostly waiting. Each platform takes 2-8 weeks to review, and re-verifies whenever your manifest, scopes, or auth flow change. + +## Scoring +- **Effort 3/5** - The technical lift per platform is small (a manifest, a screenshot bundle, a privacy URL). The cost is calendar time plus the discipline to re-submit on every breaking change. +- **Impact 5/5** - Without a verified listing, your tool ranks below verified competitors in agent tool-selection. +- **Visual change: low** - listings appear on third-party platforms; an optional "available on" badge on your site is up to you. + +## Steps +1. **Confirm your `/.well-known/ai-plugin.json` is current** (the file itself ships in [1.2](./m1-2-well-known-agent-files.md)). For submission it must point at the live OpenAPI from [4.1](./m4-1-openapi-spec.md) and the OAuth metadata from [3.1](./m3-1-oauth-discovery.md), with stable `name_for_model` and `description_for_model`. Validate against OpenAI's plugin schema before submitting. +2. **Submit to the ChatGPT GPT Store.** Provide the manifest URL, a privacy policy URL, a legal-info URL, contact email, and 3-5 screenshots of the tool in use. Expect 2-4 weeks for first review. A verified plugin is what lets users build **Custom GPTs** on top of your service without re-implementing your auth handshake. +3. **Submit to Claude integrations.** Points at your MCP server's `/mcp` endpoint ([4.4](./m4-4-mcp-server.md)) and the corresponding `oauth-protected-resource` metadata. Anthropic verifies the MCP handshake, scope semantics, and tool descriptions. +4. **Submit to Gemini extensions.** Google's review focuses on OpenAPI cleanliness, OAuth scope minimization, and consent-screen copy. The same OpenAPI you submit to the GPT Store typically works without modification. +5. **Track re-verification cadence.** Every breaking change to your manifest, OAuth scopes, MCP tool surface, or pricing model triggers re-review. Build a release checklist that flags submission updates and queues them in parallel rather than sequentially. + +## References +- [Apps in ChatGPT](https://help.openai.com/en/articles/11487775-apps-in-chatgpt?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Claude integrations directory](https://www.anthropic.com/news/integrations?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Gemini extensions documentation](https://ai.google.dev/gemini-api/docs/extensions?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +Plugin manifest publication, GPT Store / Custom GPT / Claude / Gemini submissions, and re-verification on every release can be managed for customers using the orchestration suite. Forter keeps the manifest in sync with the live OAuth metadata and MCP server it operates on your behalf, so verifications don't break the moment you ship a new scope or tool. You inherit the verified badges across three platforms without owning the submission queue. + +What stays yours: privacy policy copy, contact email, screenshots of your product in the agent surface, and any platform-specific positioning you want to control. diff --git a/content/m5-2-mcp-apps.md b/content/m5-2-mcp-apps.md new file mode 100644 index 0000000..71b2430 --- /dev/null +++ b/content/m5-2-mcp-apps.md @@ -0,0 +1,45 @@ +--- +id: m5-2-mcp-apps +module: experiential +moduleNumber: 5 +guidelineNumber: 2 +title: Render UI with MCP Apps +complexity: 4 +impact: 5 +visualChange: none +forterApplies: 'flagship' +--- + +# 5.2 Render UI with MCP Apps + +## What & why +MCP Apps lets your tool render an **interactive UI component inside the chat conversation itself** - a size selector, an address picker, a payment-method picker, an order-confirmation dialog. Not a redirect, not a link out. The user transacts inside the agent surface. This is the largest UX leap in agentic commerce since OAuth, and it collapses discover → understand → decide → act into a single conversational turn. + +Worth knowing: the host environment is **shared with components from other tools**. Treat your bundle like any other public surface - signed, CSP-locked, network-scoped to your origin. The standards are well-defined; the build is non-trivial enough that most teams pair with Forter, which ships the component layer ready-made. + +## Scoring +- **Effort 4/5** - Three new surfaces at once: the MCP Apps capability declaration on your server, the component bundles themselves (host-compatible, CSP-locked, signed), and the JSON Schema contract that ties tools to UI resources. The spec is stabilizing. +- **Impact 5/5** - Tools with inline UI dramatically out-convert tools that hand the user a link. +- **Visual change: none** *(on your site)* - components render inside the chat host (ChatGPT, Claude). Nothing on your own site changes. + +## Steps +1. **Enable the Apps capability on your MCP server.** In the `initialize` response, advertise `capabilities.experimental.apps` (per the modelcontextprotocol-ext-apps draft). Requires the MCP server foundation from [4.4](./m4-4-mcp-server.md) already running with Streamable HTTP. +2. **Build host-compatible component bundles.** Use `@modelcontextprotocol/ext-apps` to compile React/Svelte/Vue components into the host-runtime format (served with the MCP Apps MIME type `text/html;profile=mcp-app`). Bundles must be self-contained and signed so the host can verify provenance before mounting. The host enforces a strict baseline CSP (`default-src 'none'`, `object-src 'none'`); declare any origins you legitimately need through the spec's `_meta.ui.csp` allowlists (`connectDomains`, `resourceDomains`, `frameDomains`, `baseUriDomains`) rather than relaxing CSP wholesale. +3. **Expose `ui://` resources with stable, versioned URIs.** Each component lives at a URI like `ui://yourcompany.com/checkout/payment-picker@1.4.0`. Version every URI: agents cache aggressively, and an unversioned change bricks conversations mid-flight. +4. **Tag tools with `_meta.ui.resourceUri`.** On every tool that should render inline, set `_meta.ui.resourceUri` to the matching `ui://` URI. The agent host reads this metadata at tool-listing time and pre-warms the component before invocation. +5. **Define the component-tool contract via JSON Schema.** Both the tool's `inputSchema` and the resource's expected props are JSON Schema. Define them in lockstep - a payment picker that expects `{ amount, currency, methods[] }` must match exactly what the tool returns. +6. **Sandbox aggressively.** CSP-lock every bundle, sign every asset, scope every network call to your origin only, and never accept arbitrary HTML from the tool's response. Assume the host environment is shared with components from other tools. + +## References +- [MCP Apps extension spec](https://github.com/modelcontextprotocol/ext-apps?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP resources reference](https://modelcontextprotocol.io/docs/concepts/resources?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP `_meta` field reference](https://modelcontextprotocol.io/specification/2025-06-18/basic?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide#meta) +- [CSP - Content Security Policy](https://www.w3.org/TR/CSP3/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +If you run your MCP server on the [**Forter Agentic Orchestration Suite**](https://www.forter.com/blog/agentic-orchestration/?utm_source=github&utm_medium=referral&utm_campaign=agentic-readiness-guide&utm_content=m5-2-mcp-apps) ([4.4](./m4-4-mcp-server.md)), the MCP Apps layer comes with it. Your tools don't just answer in plain text - they render a **predefined storefront and support UI**: product and variant pickers, a size selector, and other components that matter. + +The UI is yours to brand. Set your colors, adjust the design, and supply custom CSS so the components read as your storefront rather than a generic widget. The Content Security Policy is managed for you and remains configurable - so you can roll out your own scripts within it, Google Analytics and other tags included, without weakening the sandbox the host requires. + +What stays yours: the brand decisions themselves, your catalog and pricing. diff --git a/content/m5-3-cross-platform-consistency.md b/content/m5-3-cross-platform-consistency.md new file mode 100644 index 0000000..2174a44 --- /dev/null +++ b/content/m5-3-cross-platform-consistency.md @@ -0,0 +1,38 @@ +--- +id: m5-3-cross-platform-consistency +module: experiential +moduleNumber: 5 +guidelineNumber: 3 +title: Stay consistent across surfaces +complexity: 2 +impact: 3 +visualChange: medium +forterApplies: 'partial' +--- + +# 5.3 Stay consistent across surfaces + +## What & why +Agents cross-check what you say about yourself across surfaces. Your HTML `<title>`, OG description, MCP tool descriptions, plugin manifest, `llms.txt` overview, and agent card all end up in the same context window when an agent decides whether to call your tool. When they disagree, confidence - and tool-selection rate - drops. This isn't about marketing copy; it's about descriptive consistency for the **same noun**: what your product does, who it's for, what it costs. + +You locked the canonical copy back in [1.2](./m1-2-well-known-agent-files.md) and every guideline since has pulled from it - so this is a **verification pass, not a rewrite**: confirm nothing drifted, and fix the surface or two that did. + +## Scoring +- **Effort 2/5** - A diff of every surface against one file, plus a PR for whatever drifted. If the canonical-copy discipline from [1.2](./m1-2-well-known-agent-files.md) held, there is almost nothing to do. +- **Impact 3/5** - Real but bounded. A multiplier on the rest of Module 5, not a standalone win. +- **Visual change: medium** - meta tags, OG cards, and `<title>` tweaks are visible in browser tabs and link previews. Page bodies are unchanged. + +## Steps +1. **Open the canonical copy file from [1.2](./m1-2-well-known-agent-files.md).** The short name, the model-facing description, and the human-facing paragraph were settled once, at the start of the build. That file is the reference; every other surface gets checked against it. +2. **Diff every surface against it.** Compare each to the canonical copy, side by side: HTML `<title>`, `<meta name="description">`, OG `og:title` / `og:description`, the `llms.txt` opening paragraph, MCP `serverInfo.name` and tool `description` fields, `ai-plugin.json` `name_for_human` / `description_for_model`, the GPT Store / Claude / Gemini listings, the agent card. If every guideline pulled from the canonical file as instructed, this is clean - in practice one or two surfaces drift. +3. **Fix the drift at its source.** Where a surface diverged, correct it *and* correct the template or generator that produced it, so it can't drift again. CMS-controlled surfaces - HTML head, OG tags, sitemap titles, `llms.txt` - are your repo and your team; Forter cannot reach in here. +4. **Align Forter-controlled metadata.** Submit canonical short name, descriptions, and tool-level descriptions to Forter's customer console; the orchestration suite propagates them to the MCP server's `serverInfo`, every tool's `description` field, the published `ai-plugin.json`, the agent card, and re-submission packages for the GPT Store / Claude / Gemini directories. + +## References +- [Open Graph protocol](https://ogp.me/?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP server initialization spec](https://modelcontextprotocol.io/specification/2025-06-18/basic/lifecycle?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide#initialization) +- [Apps in ChatGPT](https://help.openai.com/en/articles/11487775-apps-in-chatgpt?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +The metadata the orchestration suite publishes on your behalf - MCP tool descriptions, `ai-plugin.json`, agent cards, registry entries - is enforced consistent against a single source of truth in your customer console. Update the canonical name and description once; every Forter-published surface re-renders in lockstep, and the next round of plugin-store / Custom GPT re-verifications ([5.1](./m5-1-verified-on-platforms.md)) ships the new copy without a separate submission queue. diff --git a/content/m5-4-end-to-end-flows.md b/content/m5-4-end-to-end-flows.md new file mode 100644 index 0000000..88f18bb --- /dev/null +++ b/content/m5-4-end-to-end-flows.md @@ -0,0 +1,40 @@ +--- +id: m5-4-end-to-end-flows +module: experiential +moduleNumber: 5 +guidelineNumber: 4 +title: Pass end-to-end agent flows +complexity: 4 +impact: 5 +visualChange: none +forterApplies: 'partial' +--- + +# 5.4 Pass end-to-end agent flows + +## What & why +This is the integration test for Modules 1 through 4. Discovery, comprehension, trust, and action have to compose into a single agent flow that completes a real user task without dropping out of the conversation. If the first three modules pass in isolation but the end-to-end flow dies at "I don't know what this costs" or "please open this link in your browser", the site isn't agent-ready, regardless of any individual checkmark. + +Three concerns run in parallel: **multi-turn conversations** that don't dead-end on missing pricing or comparison content; **autonomous task completion** with no browser-only steps; and **continuous simulation** against ChatGPT-User, ClaudeBot, and OpenClaw on every release. The simulation harness side is dense enough that most teams pair with Forter for the protocol-layer half. + +## Scoring +- **Effort 4/5** - Standing up CI against three agent runtimes is real engineering. Each one has its own auth, rate limits, and harness conventions. +- **Impact 5/5** - The only test that catches regressions where Modules 1-4 silently disagree. +- **Visual change: none** - the work is testing infrastructure; user-visible site doesn't change. + +## Steps +1. **Make pricing and comparison content reachable from `llms.txt`.** The machine-readable `/pricing.md` and the `/compare/{competitor}` pages already exist ([2.4](./m2-4-competitive-positioning.md)); confirm `llms.txt` links to them from its `## Pricing` and comparison sections, and that each is server-rendered text an agent can actually read. Multi-turn flows die at this content gap more often than at any protocol failure. +2. **Make the auth flow programmatic end-to-end.** Verify that an agent can follow your `/auth.md` recipe ([3.3](./m3-3-self-serve-credentials.md)) - resolve the `agent_auth` hook, self-register, and complete OAuth + dynamic client registration ([3.1](./m3-1-oauth-discovery.md)) - without a single manual step. Any "open this URL in a browser to consent" step that isn't itself an inline-UI MCP App is a dead end for autonomous agents. +3. **Extend the auth-flow harness from [3.3](./m3-3-self-serve-credentials.md) to drive end-to-end tasks.** 3.3 already stands up a CI harness against ChatGPT-User, ClaudeBot, and OpenClaw for token issuance - extend each scripted session through the full discover → authenticate → act cycle. Assert that the conversation completes in N turns without falling back to a web fetch. +4. **Cover the differences between agents.** Claude's tool-selection heuristics and MCP handshake assumptions differ from OpenAI's; OpenClaw's autonomous-loop behavior surfaces business-logic gaps (rate-limit handling, multi-step transaction flows, error recovery) that single-turn harnesses miss. Problems invisible to one are loud in another - that's why all three matter. +5. **Treat the suite as red/green CI.** Break the build on regressions, the same way you would for unit tests. The failure surface should shrink over time as the rest of Modules 1-4 settle. + +## References +- [ChatGPT-User crawler documentation](https://platform.openai.com/docs/bots?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [Anthropic crawlers documentation](https://support.claude.com/en/articles/8896518-does-anthropic-crawl-data-from-the-web-and-how-can-site-owners-block-the-crawler?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [OpenClaw project](https://github.com/openclaw/openclaw?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) +- [MCP testing guide](https://modelcontextprotocol.io/docs/tools/inspector?utm_source=forter&utm_medium=referral&utm_campaign=agentic-readiness-guide) + +## How Forter helps + +Continuous end-to-end simulations against ChatGPT, Claude, and OpenClaw run against the protocol layer the orchestration suite operates on your behalf. The OAuth handshake, MCP tool surface, RFC 9421 signature verification, plugin manifest, and inline-UI components are all exercised in CI on every Forter release. The same harness catches behavioral drift - when an agent runtime quietly changes its tool-selection logic or token-handling behavior in production, you find out from a green/red signal rather than a customer ticket. diff --git a/package-lock.json b/package-lock.json new file mode 100644 index 0000000..bd7e488 --- /dev/null +++ b/package-lock.json @@ -0,0 +1,144 @@ +{ + "name": "agentic-readiness-guide", + "version": "0.0.0", + "lockfileVersion": 3, + "requires": true, + "packages": { + "": { + "name": "agentic-readiness-guide", + "version": "0.0.0", + "devDependencies": { + "gray-matter": "^4.0.3", + "zod": "^3.23.8" + } + }, + "node_modules/argparse": { + "version": "1.0.10", + "resolved": "https://registry.npmjs.org/argparse/-/argparse-1.0.10.tgz", + "integrity": "sha512-o5Roy6tNG4SL/FOkCAN6RzjiakZS25RLYFrcMttJqbdd8BWrnA+fGz57iN5Pb06pvBGvl5gQ0B48dJlslXvoTg==", + "dev": true, + "license": "MIT", + "dependencies": { + "sprintf-js": "~1.0.2" + } + }, + "node_modules/esprima": { + "version": "4.0.1", + "resolved": "https://registry.npmjs.org/esprima/-/esprima-4.0.1.tgz", + "integrity": "sha512-eGuFFw7Upda+g4p+QHvnW0RyTX/SVeJBDM/gCtMARO0cLuT2HcEKnTPvhjV6aGeqrCB/sbNop0Kszm0jsaWU4A==", + "dev": true, + "license": "BSD-2-Clause", + "bin": { + "esparse": "bin/esparse.js", + "esvalidate": "bin/esvalidate.js" + }, + "engines": { + "node": ">=4" + } + }, + "node_modules/extend-shallow": { + "version": "2.0.1", + "resolved": "https://registry.npmjs.org/extend-shallow/-/extend-shallow-2.0.1.tgz", + "integrity": "sha512-zCnTtlxNoAiDc3gqY2aYAWFx7XWWiasuF2K8Me5WbN8otHKTUKBwjPtNpRs/rbUZm7KxWAaNj7P1a/p52GbVug==", + "dev": true, + "license": "MIT", + "dependencies": { + "is-extendable": "^0.1.0" + }, + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/gray-matter": { + "version": "4.0.3", + "resolved": "https://registry.npmjs.org/gray-matter/-/gray-matter-4.0.3.tgz", + "integrity": "sha512-5v6yZd4JK3eMI3FqqCouswVqwugaA9r4dNZB1wwcmrD02QkV5H0y7XBQW8QwQqEaZY1pM9aqORSORhJRdNK44Q==", + "dev": true, + "license": "MIT", + "dependencies": { + "js-yaml": "^3.13.1", + "kind-of": "^6.0.2", + "section-matter": "^1.0.0", + "strip-bom-string": "^1.0.0" + }, + "engines": { + "node": ">=6.0" + } + }, + "node_modules/is-extendable": { + "version": "0.1.1", + "resolved": "https://registry.npmjs.org/is-extendable/-/is-extendable-0.1.1.tgz", + "integrity": "sha512-5BMULNob1vgFX6EjQw5izWDxrecWK9AM72rugNr0TFldMOi0fj6Jk+zeKIt0xGj4cEfQIJth4w3OKWOJ4f+AFw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/js-yaml": { + "version": "3.14.2", + "resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-3.14.2.tgz", + "integrity": "sha512-PMSmkqxr106Xa156c2M265Z+FTrPl+oxd/rgOQy2tijQeK5TxQ43psO1ZCwhVOSdnn+RzkzlRz/eY4BgJBYVpg==", + "dev": true, + "license": "MIT", + "dependencies": { + "argparse": "^1.0.7", + "esprima": "^4.0.0" + }, + "bin": { + "js-yaml": "bin/js-yaml.js" + } + }, + "node_modules/kind-of": { + "version": "6.0.3", + "resolved": "https://registry.npmjs.org/kind-of/-/kind-of-6.0.3.tgz", + "integrity": "sha512-dcS1ul+9tmeD95T+x28/ehLgd9mENa3LsvDTtzm3vyBEO7RPptvAD+t44WVXaUjTBRcrpFeFlC8WCruUR456hw==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/section-matter": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/section-matter/-/section-matter-1.0.0.tgz", + "integrity": "sha512-vfD3pmTzGpufjScBh50YHKzEu2lxBWhVEHsNGoEXmCmn2hKGfeNLYMzCJpe8cD7gqX7TJluOVpBkAequ6dgMmA==", + "dev": true, + "license": "MIT", + "dependencies": { + "extend-shallow": "^2.0.1", + "kind-of": "^6.0.0" + }, + "engines": { + "node": ">=4" + } + }, + "node_modules/sprintf-js": { + "version": "1.0.3", + "resolved": "https://registry.npmjs.org/sprintf-js/-/sprintf-js-1.0.3.tgz", + "integrity": "sha512-D9cPgkvLlV3t3IzL0D0YLvGA9Ahk4PcvVwUbN0dSGr1aP0Nrt4AEnTUbuGvquEC0mA64Gqt1fzirlRs5ibXx8g==", + "dev": true, + "license": "BSD-3-Clause" + }, + "node_modules/strip-bom-string": { + "version": "1.0.0", + "resolved": "https://registry.npmjs.org/strip-bom-string/-/strip-bom-string-1.0.0.tgz", + "integrity": "sha512-uCC2VHvQRYu+lMh4My/sFNmF2klFymLX1wHJeXnbEJERpV/ZsVuonzerjfrGpIGF7LBVa1O7i9kjiWvJiFck8g==", + "dev": true, + "license": "MIT", + "engines": { + "node": ">=0.10.0" + } + }, + "node_modules/zod": { + "version": "3.25.76", + "resolved": "https://registry.npmjs.org/zod/-/zod-3.25.76.tgz", + "integrity": "sha512-gzUt/qt81nXsFGKIFcC3YnfEAx5NkunCfnDlvuBSSFS02bcXu4Lmea0AFIUwbLWxWPx3d9p8S5QoaujKcNQxcQ==", + "dev": true, + "license": "MIT", + "funding": { + "url": "https://github.com/sponsors/colinhacks" + } + } + } +} diff --git a/package.json b/package.json new file mode 100644 index 0000000..9ea86ca --- /dev/null +++ b/package.json @@ -0,0 +1,15 @@ +{ + "name": "agentic-readiness-guide", + "private": true, + "version": "0.0.0", + "description": "Frontmatter + rubric validator for the Agentic Readiness Guide. The guide itself is markdown in content/ and audit/; this package only validates structure.", + "type": "module", + "scripts": { + "test": "node scripts/validate.mjs", + "validate": "node scripts/validate.mjs" + }, + "devDependencies": { + "gray-matter": "^4.0.3", + "zod": "^3.23.8" + } +} diff --git a/scripts/validate.mjs b/scripts/validate.mjs new file mode 100644 index 0000000..b6579be --- /dev/null +++ b/scripts/validate.mjs @@ -0,0 +1,259 @@ +#!/usr/bin/env node +/** + * Self-contained validator for content/ and audit/. + * + * Checks: + * 1. Every content/*.md parses with valid frontmatter (guideline or chapter schema). + * 2. Each guideline body has the required sections in the required order. + * 3. `forterApplies` agrees with the presence/absence of the "How Forter helps" section. + * 4. Every audit/m{M}-{N}.md parses with valid rubric frontmatter. + * 5. Each audit rubric matches its content guideline on id, complexity, impact, visualChange. + * 6. Cross-references of the form (./mX-Y-*.md) resolve to existing files. + * + * Exits 1 on any failure; 0 on success. + */ + +import { readdirSync, readFileSync, existsSync } from "node:fs"; +import { join, dirname, resolve } from "node:path"; +import { fileURLToPath } from "node:url"; +import matter from "gray-matter"; +import { z } from "zod"; + +const REPO = resolve(dirname(fileURLToPath(import.meta.url)), ".."); +const CONTENT = join(REPO, "content"); +const AUDIT = join(REPO, "audit"); + +const MODULE_SLUGS = ["discoverable", "comprehensible", "trustworthy", "actionable", "experiential"]; + +const oneToFive = z.union([z.literal(1), z.literal(2), z.literal(3), z.literal(4), z.literal(5)]); + +const guidelineFm = z.object({ + id: z.string().regex(/^m[1-5]-\d+-[a-z0-9-]+$/, "id must be slug like 'm4-1-openapi-spec'"), + module: z.enum(MODULE_SLUGS), + moduleNumber: oneToFive, + guidelineNumber: z.number().int().positive(), + title: z.string().min(1), + complexity: oneToFive, + impact: oneToFive, + visualChange: z.enum(["none", "low", "medium", "high"]).optional(), + forterApplies: z.enum(["no", "partial", "yes", "flagship"]), +}); + +const chapterFm = z.object({ + id: z.string(), + title: z.string(), + kind: z.enum(["front-matter", "chapter", "module-overview", "appendix"]), + moduleNumber: z.number().optional(), +}); + +const auditFm = z.object({ + id: z.string().regex(/^m[1-5]-\d+$/, "id must be like 'm4-1'"), + title: z.string().min(1), + complexity: oneToFive, + impact: oneToFive, + visualChange: z.enum(["none", "low", "medium", "high"]).optional(), + weight_total: z.number().int().positive(), +}); + +const REQUIRED_GUIDELINE_SECTIONS = ["What & why", "Scoring", "Steps", "References"]; + +let errors = 0; +let warnings = 0; +const fail = (msg) => { errors++; console.error("✗ " + msg); }; +const warn = (msg) => { warnings++; console.warn("! " + msg); }; +const ok = (msg) => console.log("✓ " + msg); + +function listMd(dir) { + return readdirSync(dir).filter((f) => f.endsWith(".md")).sort().map((f) => join(dir, f)); +} + +function hasSection(body, label) { + const re = new RegExp(`^##\\s+${label.replace(/[.*+?^${}()|[\\]\\\\]/g, "\\\\$&")}\\b`, "m"); + return re.test(body); +} + +function checkSectionsInOrder(body, labels) { + const positions = labels.map((l) => { + const m = body.match(new RegExp(`^##\\s+${l.replace(/[.*+?^${}()|[\\]\\\\]/g, "\\\\$&")}\\b`, "m")); + return m ? m.index : -1; + }); + for (let i = 1; i < positions.length; i++) { + if (positions[i] === -1) continue; + if (positions[i - 1] === -1) continue; + if (positions[i] < positions[i - 1]) return false; + } + return true; +} + +function loadFm(path, schema) { + const raw = readFileSync(path, "utf8"); + const parsed = matter(raw); + const result = schema.safeParse(parsed.data); + return { raw, body: parsed.content, fm: parsed.data, result }; +} + +// 1+2+3: content/ +const contentFiles = listMd(CONTENT); +const guidelines = new Map(); // "M-N" -> { fm, path } +const ids = new Set(); + +for (const path of contentFiles) { + const rel = path.replace(REPO + "/", ""); + const { body, fm, result } = (() => { + try { + const raw = readFileSync(path, "utf8"); + const parsed = matter(raw); + const isGuideline = parsed.data && typeof parsed.data === "object" && "guidelineNumber" in parsed.data; + const schema = isGuideline ? guidelineFm : chapterFm; + const r = schema.safeParse(parsed.data); + return { body: parsed.content, fm: parsed.data, result: r }; + } catch (e) { + fail(`${rel}: failed to parse frontmatter: ${e.message}`); + return { body: "", fm: null, result: { success: false } }; + } + })(); + + if (!result.success) { + fail(`${rel}: frontmatter validation failed:\n ${result.error.issues.map((i) => `${i.path.join(".")}: ${i.message}`).join("\n ")}`); + continue; + } + + if (fm.id) { + if (ids.has(fm.id)) fail(`${rel}: duplicate id '${fm.id}'`); + ids.add(fm.id); + } + + if ("guidelineNumber" in fm) { + // Guideline: validate body + const h1 = body.match(/^# (.+)$/m); + if (!h1) { + fail(`${rel}: missing H1`); + } + for (const section of REQUIRED_GUIDELINE_SECTIONS) { + if (!hasSection(body, section)) fail(`${rel}: missing required section '## ${section}'`); + } + if (!checkSectionsInOrder(body, REQUIRED_GUIDELINE_SECTIONS)) { + fail(`${rel}: required sections appear out of order`); + } + const hasForterSection = hasSection(body, "How Forter helps"); + if (fm.forterApplies !== "no" && !hasForterSection) { + fail(`${rel}: forterApplies=${fm.forterApplies} but body lacks '## How Forter helps'`); + } + if (fm.forterApplies === "no" && hasForterSection) { + fail(`${rel}: forterApplies=no but body contains '## How Forter helps'`); + } + const key = `${fm.moduleNumber}-${fm.guidelineNumber}`; + guidelines.set(key, { fm, path: rel }); + } +} + +ok(`Parsed ${contentFiles.length} content files (${guidelines.size} guidelines)`); + +// 4+5: audit/ +const auditFiles = listMd(AUDIT).filter((p) => /\/m\d+-\d+\.md$/.test(p)); +const auditByKey = new Map(); + +for (const path of auditFiles) { + const rel = path.replace(REPO + "/", ""); + let parsed; + try { + parsed = matter(readFileSync(path, "utf8")); + } catch (e) { + fail(`${rel}: parse error: ${e.message}`); + continue; + } + const r = auditFm.safeParse(parsed.data); + if (!r.success) { + fail(`${rel}: audit frontmatter invalid:\n ${r.error.issues.map((i) => `${i.path.join(".")}: ${i.message}`).join("\n ")}`); + continue; + } + const key = parsed.data.id.replace(/^m/, "").replace("-", "-"); // already "M-N" minus 'm' + auditByKey.set(key, { fm: parsed.data, path: rel }); + + const body = parsed.content; + if (!/## Probe/m.test(body)) fail(`${rel}: missing '## Probe' section`); + if (!/## Rubric/m.test(body)) fail(`${rel}: missing '## Rubric' section`); + + // weight_total must equal the sum of the rubric table row weights (the final integer + // cell of each data row). Catches drift when sub-checks are added/removed. The Rubric + // section may contain more than one table (e.g. m4-9 splits ACP and UCP). + const rubricSection = (() => { + const start = body.search(/^## Rubric\b/m); + if (start === -1) return ""; + const rest = body.slice(start + 1); + const next = rest.search(/^## /m); + return next === -1 ? body.slice(start) : body.slice(start, start + 1 + next); + })(); + let rowWeightSum = 0, rowCount = 0; + for (const line of rubricSection.split("\n")) { + if (!/^\s*\|/.test(line)) continue; // table rows only + if (/^\s*\|\s*-{2,}/.test(line)) continue; // separator row + const cells = line.split("|").map((c) => c.trim()).filter((c) => c !== ""); + if (cells.length < 2) continue; + const last = cells[cells.length - 1]; + if (!/^\d+$/.test(last)) continue; // skips header (…| Weight |) rows + rowWeightSum += Number(last); + rowCount++; + } + if (rowCount > 0 && rowWeightSum !== parsed.data.weight_total) { + fail(`${rel}: weight_total=${parsed.data.weight_total} but rubric rows sum to ${rowWeightSum} (${rowCount} scored rows)`); + } +} + +ok(`Parsed ${auditFiles.length} audit rubrics`); + +// 5: audit ↔ content alignment +for (const [key, audit] of auditByKey) { + const guideline = guidelines.get(key); + if (!guideline) { + fail(`${audit.path}: no matching content guideline for id=${audit.fm.id}`); + continue; + } + if (audit.fm.complexity !== guideline.fm.complexity) { + fail(`${audit.path}: complexity=${audit.fm.complexity} but ${guideline.path} has complexity=${guideline.fm.complexity}`); + } + if (audit.fm.impact !== guideline.fm.impact) { + fail(`${audit.path}: impact=${audit.fm.impact} but ${guideline.path} has impact=${guideline.fm.impact}`); + } + if (audit.fm.visualChange && guideline.fm.visualChange && audit.fm.visualChange !== guideline.fm.visualChange) { + fail(`${audit.path}: visualChange=${audit.fm.visualChange} but ${guideline.path} has visualChange=${guideline.fm.visualChange}`); + } +} +for (const [key, guideline] of guidelines) { + if (!auditByKey.has(key)) warn(`${guideline.path}: no matching audit rubric at audit/m${key}.md`); +} + +// 6: cross-references +const allMdNames = new Set([...contentFiles, ...auditFiles].map((p) => p.split("/").pop())); +const allAuditNames = new Set(auditFiles.map((p) => p.split("/").pop())); +const allContentNames = new Set(contentFiles.map((p) => p.split("/").pop())); + +const xrefRe = /\]\((\.\/|\.\.\/(?:content|audit)\/)([\w./-]+\.md)(#[^)]+)?\)/g; +for (const path of [...contentFiles, ...auditFiles]) { + const rel = path.replace(REPO + "/", ""); + const body = readFileSync(path, "utf8"); + let m; + while ((m = xrefRe.exec(body)) !== null) { + const prefix = m[1]; + const target = m[2]; + const sameDir = prefix === "./"; + let exists; + if (sameDir) { + exists = path.startsWith(CONTENT) ? allContentNames.has(target) : allAuditNames.has(target); + } else { + // ../content/foo.md or ../audit/foo.md + const targetSet = prefix.includes("content") ? allContentNames : allAuditNames; + exists = targetSet.has(target); + } + if (!exists) fail(`${rel}: dangling cross-reference '${prefix}${target}'`); + } +} + +ok("Cross-references resolved"); + +console.log(""); +if (errors > 0) { + console.error(`FAILED with ${errors} error(s), ${warnings} warning(s).`); + process.exit(1); +} +console.log(`All valid. ${warnings} warning(s).`);