diff --git a/.github/workflows/smithy-sync.yml b/.github/workflows/smithy-sync.yml
index 9046df72..f1db2816 100644
--- a/.github/workflows/smithy-sync.yml
+++ b/.github/workflows/smithy-sync.yml
@@ -48,22 +48,65 @@ jobs:
echo "changed=true" >> $GITHUB_OUTPUT
fi
+ # continue-on-error, not a gate: this suite includes the published-figure
+ # gate in cmd/devcloud/coverage_test.go, which an upstream model that gains
+ # a single operation is *designed* to fail. Ending the job here would skip
+ # Create Pull Request and discard the refreshed models with the runner —
+ # so the only sync worth reviewing would be the only one nobody ever sees.
+ # The failure is not swallowed; it is re-raised below, after the PR exists.
- name: Run tests
+ id: tests
if: steps.changes.outputs.changed == 'true'
+ continue-on-error: true
run: CGO_ENABLED=0 go test ./... -v
+ # The diff is whole-tree whether upstream moved one model or ninety — a
+ # real refresh measured on 2026-09-06 changed 93 models and 134 generated
+ # files. Asking a reviewer to "review" that is asking for nothing. This
+ # reduces it to the question they actually have: which operations moved.
+ - name: Summarise the churn
+ if: steps.changes.outputs.changed == 'true'
+ run: |
+ {
+ echo "Automated weekly sync of AWS Smithy models from \`aws-sdk-go-v2\`."
+ echo
+ echo "**In-job test result: \`${{ steps.tests.outcome }}\`.**"
+ echo
+ echo "A \`failure\` is expected when upstream added or removed an"
+ echo "operation: the published-figure gate over \`docs/coverage.md\` fails"
+ echo "on purpose so a human looks at a coverage change. Re-derive the"
+ echo "figures and correct the doc in this PR — do not silence the gate."
+ echo
+ echo "This PR's own \`ci\`, \`compat\` and \`codegen-drift\` runs are the"
+ echo "gate on merging."
+ echo
+ echo "---"
+ echo
+ python3 scripts/model_churn.py --upstream
+ echo
+ echo "Summary from \`scripts/model_churn.py\`; re-derive with"
+ echo "\`python3 scripts/model_churn.py --upstream\`."
+ } > /tmp/sync-pr-body.md
+ cat /tmp/sync-pr-body.md
+ env:
+ GITHUB_TOKEN: ${{ github.token }}
+
- name: Create Pull Request
if: steps.changes.outputs.changed == 'true'
uses: peter-evans/create-pull-request@5f6978faf089d4d20b00c7766989d076bb2fc7f1 # v8
with:
commit-message: "chore: sync Smithy models and regenerate code"
title: "chore: weekly Smithy model sync"
- body: |
- Automated weekly sync of AWS Smithy models from `aws-sdk-go-v2`.
-
- This PR updates generated code in `internal/generated/` and scaffolds
- in `internal/services/` based on the latest AWS service models.
-
- Please review and merge if CI passes.
+ body-path: /tmp/sync-pr-body.md
branch: smithy-sync/weekly
delete-branch: true
+
+ # The PR now exists, so the failure can end the job. Without this the cron
+ # would report success every week no matter what upstream did — the same
+ # silent no-op scripts/download-smithy-models.sh was fixed to stop doing,
+ # traded for the one continue-on-error just removed.
+ - name: Fail if the sync's tests failed
+ if: steps.tests.outcome == 'failure'
+ run: |
+ echo "::error::Smithy sync tests failed; see the PR opened by this run."
+ exit 1
diff --git a/changes/unreleased/Added-20260906-210100.yaml b/changes/unreleased/Added-20260906-210100.yaml
new file mode 100644
index 00000000..d60cd0e9
--- /dev/null
+++ b/changes/unreleased/Added-20260906-210100.yaml
@@ -0,0 +1,5 @@
+kind: Added
+body: '`scripts/model_churn.py` summarises what a Smithy model sync actually changed — which services gained or lost operations, which models moved only documentation, and how many upstream models are not vendored here. The weekly sync PR is built from it, so a reviewer reads a change rather than a whole-tree regeneration of 194 models. Run it by hand with `python3 scripts/model_churn.py --upstream`'
+time: 2026-09-06T21:01:00.000000+09:00
+custom:
+ Issue: "147"
diff --git a/changes/unreleased/Documentation-20260906-210200.yaml b/changes/unreleased/Documentation-20260906-210200.yaml
new file mode 100644
index 00000000..31e801ff
--- /dev/null
+++ b/changes/unreleased/Documentation-20260906-210200.yaml
@@ -0,0 +1,5 @@
+kind: Documentation
+body: 'docs/coverage.md now publishes what keeping up with upstream costs, measured rather than estimated: refreshing all 194 vendored models changed 93 of them, 32 moved an operation, and none was documentation-only. The reading states its own ceiling — the 93 had been vendored 141 days earlier, and the 101 vendored the day before did not move at all, so it is an accumulated backlog and not a weekly rate. docs/contributing.md adds what a reviewer of the weekly sync PR is actually being asked to check'
+time: 2026-09-06T21:02:00.000000+09:00
+custom:
+ Issue: "147"
diff --git a/changes/unreleased/Fixed-20260906-210000.yaml b/changes/unreleased/Fixed-20260906-210000.yaml
new file mode 100644
index 00000000..6b74b9f9
--- /dev/null
+++ b/changes/unreleased/Fixed-20260906-210000.yaml
@@ -0,0 +1,5 @@
+kind: Fixed
+body: 'The weekly Smithy model sync discarded the changes worth reviewing. Its test step ended the job before the pull request was opened, and an upstream model that gains an operation is designed to fail that step — it moves the fidelity manifest and trips the published-figure gate. So a cosmetic sync produced a PR and a semantic one produced a red cron job with nothing attached. The test result is now recorded rather than gating, the PR is always opened when models moved, and the failure is re-raised afterwards so the cron does not silently go green'
+time: 2026-09-06T21:00:00.000000+09:00
+custom:
+ Issue: "147"
diff --git a/cmd/devcloud/sync_test.go b/cmd/devcloud/sync_test.go
new file mode 100644
index 00000000..60929625
--- /dev/null
+++ b/cmd/devcloud/sync_test.go
@@ -0,0 +1,211 @@
+// SPDX-License-Identifier: Apache-2.0
+
+package main
+
+import (
+ "os"
+ "path/filepath"
+ "strings"
+ "testing"
+
+ "github.com/stretchr/testify/assert"
+ "github.com/stretchr/testify/require"
+ "gopkg.in/yaml.v3"
+)
+
+// The weekly Smithy sync is the only mechanism that surfaces upstream model
+// churn, and its output is a pull request. These tests gate the shape of that
+// workflow for the same reason coverage_test.go gates docs/coverage.md: the
+// asset is not Go, but a silent regression in it is invisible until the week it
+// matters.
+
+// syncStep is the subset of a GitHub Actions step these tests read. `with`
+// values are not all strings — `delete-branch: true` is a bool — so the map is
+// typed loosely and read as text only where it is read at all.
+type syncStep struct {
+ Name string `yaml:"name"`
+ ID string `yaml:"id"`
+ Uses string `yaml:"uses"`
+ Run string `yaml:"run"`
+ If string `yaml:"if"`
+ ContinueOnError bool `yaml:"continue-on-error"`
+ With map[string]any `yaml:"with"`
+}
+
+type syncWorkflow struct {
+ Jobs map[string]struct {
+ Steps []syncStep `yaml:"steps"`
+ } `yaml:"jobs"`
+}
+
+// syncSteps returns the steps of the sync job in file order.
+func syncSteps(t *testing.T) []syncStep {
+ t.Helper()
+
+ path := filepath.Join(repoRoot(t), ".github", "workflows", "smithy-sync.yml")
+ raw, err := os.ReadFile(path)
+ require.NoError(t, err)
+
+ var wf syncWorkflow
+ require.NoError(t, yaml.Unmarshal(raw, &wf))
+ require.Len(t, wf.Jobs, 1, "smithy-sync.yml is expected to hold exactly one job")
+
+ for _, job := range wf.Jobs {
+ require.NotEmpty(t, job.Steps, "the sync job has no steps")
+ return job.Steps
+ }
+ return nil
+}
+
+// findStep returns the index of the first step matching pred, or -1.
+func findStep(steps []syncStep, pred func(syncStep) bool) int {
+ for i, s := range steps {
+ if pred(s) {
+ return i
+ }
+ }
+ return -1
+}
+
+func runsGoTest(s syncStep) bool { return strings.Contains(s.Run, "go test") }
+
+func opensPullRequest(s syncStep) bool { return strings.Contains(s.Uses, "create-pull-request") }
+
+// prBodyText returns the text that becomes the pull request body.
+//
+// create-pull-request takes the body either inline as `body` or from a file as
+// `body-path`. Both are legitimate, and which one the workflow uses is not a
+// guarantee worth pinning — what the body *says* is. So a `body-path` resolves
+// to the shell of every step that writes to that path, which is where the
+// content actually comes from.
+func prBodyText(t *testing.T, steps []syncStep) string {
+ t.Helper()
+
+ prIdx := findStep(steps, opensPullRequest)
+ require.NotEqual(t, -1, prIdx, "no step opens a pull request")
+ with := steps[prIdx].With
+
+ if body, ok := with["body"].(string); ok {
+ return body
+ }
+
+ path, ok := with["body-path"].(string)
+ require.True(t, ok, "the create-pull-request step supplies neither body nor body-path")
+
+ var b strings.Builder
+ for _, s := range steps[:prIdx] {
+ if strings.Contains(s.Run, path) {
+ b.WriteString(s.Run)
+ }
+ }
+ require.NotEmpty(t, b.String(),
+ "body-path is %q but no earlier step writes to it, so the PR body is empty", path)
+ return b.String()
+}
+
+// TestSyncOpensAPullRequestEvenWhenTestsFail is the reproducer for the defect
+// that made the sustaining cost unmeasurable.
+//
+// A step that fails ends the job, and every step after it is skipped, unless
+// either the failing step is marked continue-on-error or the later step's `if`
+// re-enables it with always(). The sync runs the full Go suite — which includes
+// the published-figure gate in coverage_test.go — before it opens the PR. An
+// upstream model that gains a single operation moves the manifest, fails that
+// gate, and takes the PR with it. The refreshed models are then discarded with
+// the runner, so the one change worth reviewing is the one nobody ever sees.
+//
+// The rule asserted here is GitHub Actions' own: the PR step must be reachable
+// from a failed test step. How that is arranged — continue-on-error on the test
+// or always() on the PR — is left to the workflow.
+func TestSyncOpensAPullRequestEvenWhenTestsFail(t *testing.T) {
+ steps := syncSteps(t)
+
+ testIdx := findStep(steps, runsGoTest)
+ require.NotEqual(t, -1, testIdx,
+ "no step runs 'go test'; if the sync stopped testing, this gate is reading the wrong thing")
+
+ prIdx := findStep(steps, opensPullRequest)
+ require.NotEqual(t, -1, prIdx, "no step opens a pull request")
+ require.Less(t, testIdx, prIdx,
+ "the test step is expected to run before the PR step; reordering them changes what this gate means")
+
+ reachable := steps[testIdx].ContinueOnError || strings.Contains(steps[prIdx].If, "always()")
+ assert.True(t, reachable,
+ "a failing test step ends the job and the PR is never opened, so the refreshed models "+
+ "are discarded with the runner. Mark the test step continue-on-error, or gate the "+
+ "PR step with always(), so a red sync still leaves a reviewable PR.")
+}
+
+// TestSyncStillFailsWhenTestsFail is the other half of the fix, and the reason
+// continue-on-error is not sufficient on its own.
+//
+// Marking the test step continue-on-error makes the PR reachable and makes the
+// job green — a weekly cron that reports success no matter what upstream did.
+// That is the same silent no-op download-smithy-models.sh was fixed to stop
+// doing, traded for the one this milestone removes. So a workflow that swallows
+// the test failure must re-raise it after the PR exists.
+func TestSyncStillFailsWhenTestsFail(t *testing.T) {
+ steps := syncSteps(t)
+
+ testIdx := findStep(steps, runsGoTest)
+ require.NotEqual(t, -1, testIdx)
+
+ if !steps[testIdx].ContinueOnError {
+ t.Skip("the test step is not continue-on-error, so its failure already fails the job")
+ }
+ require.NotEmpty(t, steps[testIdx].ID,
+ "a swallowed failure cannot be re-raised without an id to read it from")
+
+ prIdx := findStep(steps, opensPullRequest)
+ require.NotEqual(t, -1, prIdx, "no step opens a pull request")
+
+ outcome := "steps." + steps[testIdx].ID + ".outcome"
+ reraised := findStep(steps[prIdx+1:], func(s syncStep) bool {
+ return strings.Contains(s.If, outcome) && strings.Contains(s.Run, "exit 1")
+ })
+ assert.NotEqual(t, -1, reraised,
+ "the test failure is swallowed by continue-on-error and never re-raised, so the weekly "+
+ "cron reports success regardless of what upstream changed. Add a step after the PR "+
+ "that reads "+outcome+" and exits non-zero.")
+}
+
+// TestSyncPullRequestBodyReportsTheTestResult keeps the previous guarantee
+// honest. Opening a PR whose tests failed is only an improvement if the body
+// says so; a red PR that reads like a green one invites a merge rather than a
+// review.
+func TestSyncPullRequestBodyReportsTheTestResult(t *testing.T) {
+ steps := syncSteps(t)
+
+ testIdx := findStep(steps, runsGoTest)
+ require.NotEqual(t, -1, testIdx)
+ require.NotEmpty(t, steps[testIdx].ID,
+ "the test step needs an id before its result can be quoted in the PR body")
+
+ assert.Contains(t, prBodyText(t, steps), "steps."+steps[testIdx].ID,
+ "the PR body must state the test result, so a reviewer sees a red sync as red")
+}
+
+// TestSyncPullRequestBodySummarisesTheChurn is the guarantee that makes the PR
+// reviewable rather than merely present.
+//
+// The sync regenerates from all 194 models at once, so its diff is whole-tree
+// whether upstream moved one model or ninety — measured on 2026-09-06, a real
+// refresh changed 93 models and 134 generated files. "Please review" over that
+// is not an action anyone performs. scripts/model_churn.py reduces it to the
+// question a reviewer has, which operations moved, and the body must carry that
+// answer or the reviewer is back to reading the regeneration.
+func TestSyncPullRequestBodySummarisesTheChurn(t *testing.T) {
+ steps := syncSteps(t)
+
+ churnIdx := findStep(steps, func(s syncStep) bool {
+ return strings.Contains(s.Run, "model_churn.py")
+ })
+ require.NotEqual(t, -1, churnIdx,
+ "no step runs scripts/model_churn.py, so the PR cannot say which operations moved")
+
+ prIdx := findStep(steps, opensPullRequest)
+ require.Less(t, churnIdx, prIdx, "the churn summary must be produced before the PR is opened")
+
+ assert.Contains(t, prBodyText(t, steps), "model_churn",
+ "the churn summary is produced but never reaches the PR body")
+}
diff --git a/docs/contributing.md b/docs/contributing.md
index e8e86c18..a00ec7df 100644
--- a/docs/contributing.md
+++ b/docs/contributing.md
@@ -65,6 +65,29 @@ make codegen-s3
- **Generator:** `internal/codegen/generator.go` — produces Go files using templates in `internal/codegen/templates/`
- **Output:** `internal/generated/{service}/` — types, interface, serializer, deserializer, router, errors, base_provider
+### Reviewing the weekly model sync
+
+[`smithy-sync.yml`](../.github/workflows/smithy-sync.yml) refreshes all 194
+vendored models every Monday and opens a pull request. The diff is whole-tree —
+a measured refresh moved 93 models and 134 generated files — so do not try to
+read it. Read the PR body instead: it is generated by `scripts/model_churn.py`
+and lists which services gained or lost operations.
+
+Three things to check, in order:
+
+1. **Operations added or removed.** These are the only changes that alter what
+ DevCloud serves. Everything else is upstream reshaping traits or docs.
+2. **A red `ci` run on the published-figure gate.** Expected, not a defect: new
+ operations move the fidelity manifest, and `cmd/devcloud/coverage_test.go`
+ fails until [`docs/coverage.md`](coverage.md) is re-derived. Correct the
+ figures in the sync PR — never relax the gate to make it pass.
+3. **`codegen-drift` and `compat`.** These must be green on their own. A red
+ `codegen-drift` means the committed output does not match the models; a red
+ `compat` means a real behavioural regression.
+
+The in-job test result is printed at the top of the PR body. A `failure` there
+with a clean `codegen-drift` almost always means item 2.
+
## Adding a New AWS Service
1. **Add the Smithy model** — place the JSON model file in `smithy-models/`
diff --git a/docs/coverage.md b/docs/coverage.md
index 79c2fe34..7869c583 100644
--- a/docs/coverage.md
+++ b/docs/coverage.md
@@ -314,6 +314,41 @@ taken on a different day and are not a controlled comparison, so read this as
readings agree on is the shape: memory is dominated by the runtime and the
store, not by how many services are registered.
+## Keeping up with upstream
+
+The 205 services are vendored from 194 Smithy models, and AWS keeps changing
+them. A [weekly workflow](../.github/workflows/smithy-sync.yml) refreshes all of
+them and opens a pull request. What that review costs was measured once, on
+**2026-09-06**:
+
+| Reading | Value |
+|---|---|
+| Vendored models refreshed | 194 |
+| Models that changed | 93 |
+| Of those, models that added or removed an operation | 32 |
+| Of those, models that changed only documentation | 0 |
+| Net change in known operations | 12,407 to 12,660 |
+| Generated files that moved | 134 |
+| Wall-clock to download 194 models | 1 min 53 s |
+
+**This is one sample, and it is not one week of churn.** The 93 models that
+changed were all vendored on 2026-04-18 — 141 days earlier. The other 101 were
+vendored on 2026-09-05, and not one of them changed. So the reading above is an
+accumulated backlog, and the only measurement at weekly scale is the second
+cohort's: 101 models, one day, zero changes. The weekly rate is still unknown,
+and a figure derived from a single 141-day sample should not be quoted as one.
+
+What the sample does settle is the *shape* of the work. None of the 93 was
+documentation-only, so no sync can be waved through on the assumption that AWS
+only reworded things. Thirty-two services gained operations — `ec2` alone gained
+46 — which moves the manifest and makes the published-figure gate below fail on
+purpose. That failure *is* the review: the numbers on this page have to be
+re-derived, by a person, before the sync can merge.
+
+The sync PR states which operations moved, so that review reads a change rather
+than a regeneration. Re-derive it with
+`python3 scripts/model_churn.py --upstream`.
+
## Reproducing these numbers
```bash
diff --git a/scripts/model_churn.py b/scripts/model_churn.py
new file mode 100644
index 00000000..890df6f0
--- /dev/null
+++ b/scripts/model_churn.py
@@ -0,0 +1,366 @@
+#!/usr/bin/env python3
+# SPDX-License-Identifier: Apache-2.0
+"""scripts/model_churn.py — summarise what changed in the vendored Smithy models.
+
+The weekly sync regenerates from all 194 committed models at once, so its pull
+request is a whole-tree diff whether upstream changed one model or ninety. A
+reviewer reading tens of thousands of lines of regenerated Go cannot tell which
+service moved, let alone whether an operation appeared or only a doc-string was
+reworded. This script answers the one question a reviewer actually has: *which
+operations moved?*
+
+What it does not do: it does not re-implement codegen, and it does not decide
+whether a change is safe. It reads operation shape names and documentation
+traits — nothing about protocols, bindings, or types. If it ever disagrees with
+codegen, codegen is right.
+
+Run from the workflow after `download-smithy-models.sh --refresh`, or by hand.
+Like demand_rank.py, a source that reads wrong exits non-zero rather than
+emitting a summary that under-reports.
+
+usage: model_churn.py [--models-dir smithy-models] [--base-ref HEAD]
+ [--upstream] [--self-check] [--out -]
+"""
+
+from __future__ import annotations
+
+import argparse
+import json
+import os
+import pathlib
+import re
+import subprocess
+import sys
+import urllib.error
+import urllib.request
+
+REPO_ROOT = pathlib.Path(__file__).resolve().parent.parent
+
+DOC_TRAIT = "smithy.api#documentation"
+
+UPSTREAM_MODELS_API = (
+ "https://api.github.com/repos/aws/aws-sdk-go-v2/contents/"
+ "codegen/sdk-codegen/aws-models?ref=main"
+)
+
+
+def die(msg: str) -> None:
+ print(f"ERROR: {msg}", file=sys.stderr)
+ sys.exit(1)
+
+
+def service_operations(model: dict) -> set[str]:
+ """The operation names a Smithy model declares.
+
+ Every operation is a top-level shape keyed `#` with
+ `"type": "operation"`. Reading the shape map rather than the service shape's
+ `operations` list catches an operation upstream defined but has not yet
+ wired into the service — a real upstream state, and one a reviewer should
+ see coming.
+ """
+ return {
+ shape_id.split("#")[-1]
+ for shape_id, shape in model.get("shapes", {}).items()
+ if shape.get("type") == "operation"
+ }
+
+
+def strip_documentation(node: object) -> object:
+ """Return node with every `smithy.api#documentation` trait removed.
+
+ Documentation traits appear on shapes, on members, and on nested structures,
+ so the removal is recursive rather than a pass over the top-level map.
+ """
+ if isinstance(node, dict):
+ return {
+ key: strip_documentation(value)
+ for key, value in node.items()
+ if key != DOC_TRAIT
+ }
+ if isinstance(node, list):
+ return [strip_documentation(item) for item in node]
+ return node
+
+
+def is_documentation_only(old: dict, new: dict) -> bool:
+ """True when the two models differ only in documentation text.
+
+ This is the reading that decides whether a sync needs a review or a glance.
+ It is deliberately strict: anything that is not a documentation trait — a
+ trait value, a member, an enum entry — makes the change non-cosmetic.
+
+ Two models that are byte-identical are not documentation-only; they are
+ unchanged, and the caller only ever passes models git already reported as
+ differing.
+ """
+ return strip_documentation(old) == strip_documentation(new)
+
+
+def _self_check() -> None:
+ """Assert the three readings above on synthetic models.
+
+ Synthetic rather than fixture files: the properties under test are about
+ shape maps and trait nesting, and a real multi-megabyte model would bury
+ them.
+ """
+
+ def op(doc: str | None = None) -> dict:
+ shape: dict = {"type": "operation"}
+ if doc is not None:
+ shape["traits"] = {DOC_TRAIT: doc}
+ return shape
+
+ ns = "com.amazonaws.things"
+
+ # 1. Operation names come out of the shape map; non-operations do not.
+ model = {
+ "shapes": {
+ f"{ns}#ListThings": op(),
+ f"{ns}#GetThing": op(),
+ f"{ns}#Thing": {"type": "structure"},
+ f"{ns}#Things": {"type": "service"},
+ }
+ }
+ assert service_operations(model) == {"ListThings", "GetThing"}, service_operations(
+ model
+ )
+
+ # 2. A model with no shapes has no operations rather than raising.
+ assert service_operations({}) == set()
+
+ # 3. A reworded doc-string is documentation-only.
+ old = {"shapes": {f"{ns}#ListThings": op("Lists things.")}}
+ new = {"shapes": {f"{ns}#ListThings": op("Lists all the things.")}}
+ assert is_documentation_only(old, new)
+
+ # 4. An added operation is not, even though its doc-string is the only new text.
+ new_with_op = {
+ "shapes": {
+ f"{ns}#ListThings": op("Lists things."),
+ f"{ns}#DeleteThing": op("Deletes a thing."),
+ }
+ }
+ assert not is_documentation_only(old, new_with_op)
+
+ # 5. Documentation nested on a member is stripped too — the common case, and
+ # the one a top-level-only strip would misreport as semantic churn.
+ deep_old = {
+ "shapes": {
+ f"{ns}#Thing": {
+ "type": "structure",
+ "members": {
+ "Name": {"target": "smithy.api#String", "traits": {DOC_TRAIT: "A."}}
+ },
+ }
+ }
+ }
+ deep_new = json.loads(json.dumps(deep_old))
+ deep_new["shapes"][f"{ns}#Thing"]["members"]["Name"]["traits"][DOC_TRAIT] = "B."
+ assert is_documentation_only(deep_old, deep_new)
+
+ # 6. A changed member target is semantic, not cosmetic.
+ deep_new["shapes"][f"{ns}#Thing"]["members"]["Name"]["target"] = (
+ "smithy.api#Integer"
+ )
+ assert not is_documentation_only(deep_old, deep_new)
+
+ print("self-check OK")
+
+
+def git(args: list[str]) -> str:
+ proc = subprocess.run(
+ ["git", *args], cwd=REPO_ROOT, capture_output=True, text=True, check=False
+ )
+ if proc.returncode != 0:
+ die(f"git {' '.join(args)} failed: {proc.stderr.strip()}")
+ return proc.stdout
+
+
+def changed_models(models_dir: str, base_ref: str) -> list[str]:
+ """Model filenames (without .json) that differ from base_ref."""
+ out = git(["diff", "--name-only", base_ref, "--", models_dir])
+ return sorted(
+ pathlib.PurePosixPath(line).stem
+ for line in out.splitlines()
+ if line.endswith(".json")
+ )
+
+
+def model_at_ref(models_dir: str, name: str, base_ref: str) -> dict:
+ raw = git(["show", f"{base_ref}:{models_dir}/{name}.json"])
+ try:
+ return json.loads(raw)
+ except json.JSONDecodeError as exc:
+ die(f"{name}.json at {base_ref} is not valid JSON: {exc}")
+ raise AssertionError("unreachable")
+
+
+def model_on_disk(models_dir: str, name: str) -> dict:
+ path = REPO_ROOT / models_dir / f"{name}.json"
+ try:
+ return json.loads(path.read_text(encoding="utf-8"))
+ except (OSError, json.JSONDecodeError) as exc:
+ die(f"{path} could not be read: {exc}")
+ raise AssertionError("unreachable")
+
+
+def upstream_not_in_tree(models_dir: str) -> tuple[int, int]:
+ """(upstream model count, count not vendored here).
+
+ Uses the one endpoint the sync's own download script already depends on. The
+ three emulator sources demand_rank.py reads stay hand-run: they are
+ third-party, they change under us, and they belong in a deliberate
+ re-derivation rather than on a cron.
+ """
+ req = urllib.request.Request(
+ UPSTREAM_MODELS_API, headers={"User-Agent": "devcloud-model-churn"}
+ )
+ token = os.environ.get("GITHUB_TOKEN")
+ if token:
+ req.add_header("Authorization", f"Bearer {token}")
+ try:
+ with urllib.request.urlopen(req, timeout=60) as resp:
+ entries = json.loads(resp.read())
+ except urllib.error.HTTPError as exc:
+ hint = (
+ " — GitHub rate limit; set GITHUB_TOKEN and retry"
+ if exc.code in (403, 429)
+ else ""
+ )
+ die(f"{UPSTREAM_MODELS_API} returned HTTP {exc.code}{hint}")
+ except urllib.error.URLError as exc:
+ die(f"{UPSTREAM_MODELS_API} unreachable: {exc.reason}")
+
+ names = {e["name"][:-5] for e in entries if e["name"].endswith(".json")}
+ if not names:
+ die("upstream model listing returned no .json files")
+ local = {p.stem for p in (REPO_ROOT / models_dir).glob("*.json")}
+ return len(names), len(names - local)
+
+
+def render(
+ rows: list[dict],
+ doc_only: list[str],
+ total: int,
+ upstream: tuple[int, int] | None,
+) -> str:
+ lines: list[str] = []
+ semantic = [r for r in rows if r["added"] or r["removed"]]
+
+ lines.append(
+ f"**{len(rows)} of {total} vendored models changed.** "
+ f"{len(semantic)} moved an operation; {len(doc_only)} are documentation-only."
+ )
+ lines.append("")
+
+ if semantic:
+ lines.append("| Service | Operations added | Operations removed |")
+ lines.append("|---|---|---|")
+ for r in semantic:
+ added = ", ".join(f"`{o}`" for o in r["added"]) or "—"
+ removed = ", ".join(f"`{o}`" for o in r["removed"]) or "—"
+ lines.append(f"| `{r['name']}` | {added} | {removed} |")
+ lines.append("")
+ lines.append(
+ "An operation added or removed moves the fidelity manifest, so the "
+ "published-figure gate over `docs/coverage.md` is expected to fail. "
+ "Re-derive the figures in this PR rather than silencing the gate."
+ )
+ else:
+ lines.append("No operation was added or removed by any changed model.")
+ lines.append("")
+
+ other = [
+ r["name"]
+ for r in rows
+ if not (r["added"] or r["removed"]) and r["name"] not in doc_only
+ ]
+ if other:
+ lines.append(
+ "Changed with the same operation set but not documentation-only "
+ "(shapes, traits or members moved): " + ", ".join(f"`{n}`" for n in other)
+ )
+ lines.append("")
+
+ if doc_only:
+ lines.append(
+ f"Documentation-only ({len(doc_only)})
"
+ )
+ lines.append("")
+ lines.append(", ".join(f"`{n}`" for n in doc_only))
+ lines.append("")
+ lines.append(" ")
+ lines.append("")
+
+ if upstream is not None:
+ total_up, missing = upstream
+ lines.append(
+ f"Upstream publishes {total_up} models; {missing} are not vendored here. "
+ "Ranked demand for those lives in `docs/demand.md`, re-derived by hand "
+ "with `python3 scripts/demand_rank.py`."
+ )
+ lines.append("")
+
+ return "\n".join(lines)
+
+
+def main() -> None:
+ parser = argparse.ArgumentParser(description=__doc__)
+ parser.add_argument("--models-dir", default="smithy-models")
+ parser.add_argument(
+ "--base-ref",
+ default="HEAD",
+ help="git ref to compare the working tree against",
+ )
+ parser.add_argument(
+ "--upstream",
+ action="store_true",
+ help="also report upstream models not vendored here",
+ )
+ parser.add_argument(
+ "--self-check",
+ action="store_true",
+ help="run the built-in assertions and exit",
+ )
+ parser.add_argument("--out", default="-", help="output path, or - for stdout")
+ args = parser.parse_args()
+
+ if args.self_check:
+ _self_check()
+ return
+
+ if not re.fullmatch(r"[\w./-]+", args.models_dir):
+ die(f"suspicious --models-dir {args.models_dir!r}")
+
+ total = len(list((REPO_ROOT / args.models_dir).glob("*.json")))
+ if not total:
+ die(f"no models under {args.models_dir}")
+
+ rows: list[dict] = []
+ doc_only: list[str] = []
+ for name in changed_models(args.models_dir, args.base_ref):
+ old = model_at_ref(args.models_dir, name, args.base_ref)
+ new = model_on_disk(args.models_dir, name)
+ before, after = service_operations(old), service_operations(new)
+ rows.append(
+ {
+ "name": name,
+ "added": sorted(after - before),
+ "removed": sorted(before - after),
+ }
+ )
+ if is_documentation_only(old, new):
+ doc_only.append(name)
+
+ upstream = upstream_not_in_tree(args.models_dir) if args.upstream else None
+ text = render(rows, doc_only, total, upstream)
+
+ if args.out == "-":
+ print(text)
+ else:
+ pathlib.Path(args.out).write_text(text, encoding="utf-8")
+ print(f"wrote {args.out}", file=sys.stderr)
+
+
+if __name__ == "__main__":
+ main()