diff --git a/.github/workflows/smoke-rpa-skills.yml b/.github/workflows/smoke-rpa-skills.yml index 39d8f64f56..8c3bef47c4 100644 --- a/.github/workflows/smoke-rpa-skills.yml +++ b/.github/workflows/smoke-rpa-skills.yml @@ -126,9 +126,18 @@ jobs: # Host coder-eval CLI = pinned wheel (coder_eval feed + ml-packages for # deps). Windows RPA tasks run under the tempdir driver, so no agent image. + # + # claude-agent-sdk 0.2.144 shipped without a win_amd64 wheel, so uv falls + # back to the sdist and nothing bundles claude.exe — every task here then + # dies at agent_crash ("Claude Code not found"). coder-eval asks for + # >=0.2.124 with no upper bound, so exclude that single release instead of + # capping: a later version that restores the wheel resolves normally. - name: Install coder-eval shell: bash - run: uv pip install --system "coder-eval==${{ steps.ceref.outputs.version }}" + run: | + uv pip install --system \ + "coder-eval==${{ steps.ceref.outputs.version }}" \ + "claude-agent-sdk!=0.2.144" - name: Configure NuGet feed for Helm packages shell: bash diff --git a/tests/.coder-eval-version b/tests/.coder-eval-version index af88ba8248..bc859cbd6d 100644 --- a/tests/.coder-eval-version +++ b/tests/.coder-eval-version @@ -1 +1 @@ -0.11.1 +0.11.2 diff --git a/tests/README.md b/tests/README.md index 9148e31780..44ae935fa1 100644 --- a/tests/README.md +++ b/tests/README.md @@ -179,6 +179,7 @@ Run-time caps live under `defaults.run_limits` (see coder_eval `RunLimits`). | `smoke-windows.yaml` | tempdir | PR-gate smoke (Windows RPA only) | 40 | 900s | 900s | | `activation.yaml` | tempdir | Skill activation classifier (benchmark) | 3 + early-stop | 360s | 120s | | `same-ground-headtohead.yaml` | docker | Campaign-only local comparison arm | 200 | 1200s | 900s | +| `flow-v2-preview.yaml` | docker | Flow v2 builder-SDK preview skills | 200 | 1200s | 900s | `same-ground-headtohead.yaml` is not a clean-checkout CI experiment. The campaign runner first builds the pinned `skills-image:sg1`, prepares isolated @@ -189,6 +190,20 @@ runner. The image build passes the package credential as exists only for the external nightly caller during migration. Regular nightly and smoke jobs continue to use `skills-image:latest`. +`flow-v2-preview.yaml` runs the three `preview/uipath-maestro-{flow,case,bpmn}` +builder-SDK skills as the ONLY skill catalog, shadowing the shipped v1 skills of +the same name, so a run measures the Flow v2 authoring path rather than a mix of +both generations. Narrowing `plugins.path` to `preview/` drops the automatic +repo-root bind mount, so the root is remounted explicitly; the image also needs +runtime npm auth for the `@uipath` scope. Login state mounts at `/.uipath`, +identical to `nightly.yaml`. Confirm that mount resolves before a full run, or +every tenant call fails as a capability problem rather than a config one: + +```bash +docker run --rm --env HOME="$HOME" -v ~/.uipath:/.uipath:rw \ + --entrypoint bash skills-codex:latest -c 'uip login status' +``` + `activation.yaml` is a different shape from the tiered configs above — it runs the agent against single-prompt rows to measure whether the right skill fires (precision/recall/F1 per skill). Rows get a small turn budget (`max_turns: 3`) with `stop_early: true`: the armed `skill_triggered` criteria (`stop_when: auto`) end a row as soon as its outcome is live-decided. A positive row pass-stops the moment the expected skill engages; a negative row fail-stops on its first engagement. A wrong-skill engagement alone does NOT end a positive row — fail-stop is deferred while the row's positive criterion is still undecided, so a positive row that only misfires runs to the cap, as do rows with no engagement. Decided rows cost ~1 turn and a late-but-correct invocation is no longer truncated. Requires coder_eval >= 0.9.1. It's an opt-in benchmark, not a smoke gate. See [`tasks/activation/README.md`](tasks/activation/README.md). For **A/B comparisons between two skill variants** (e.g. `main` vs a feature branch, or two historical commits), see [`experiments/skill-comparison-playbook.md`](experiments/skill-comparison-playbook.md) and the [`experiments/skill-comparison-template.yaml`](experiments/skill-comparison-template.yaml). The playbook covers worktree setup, SHA pinning for reproducibility, getting N>1, and interpreting divergent tasks. To automate the whole flow, use the `/skill-compare [task_selector] [n_reps]` slash command — each ref can be a branch name or a commit SHA, and `task_selector` accepts a skill name (`uipath-maestro-flow`), tag list (`tags:smoke,init`), or path globs (`paths:tasks/uipath-maestro-flow/*.yaml`). diff --git a/tests/experiments/flow-v2-preview.yaml b/tests/experiments/flow-v2-preview.yaml new file mode 100644 index 0000000000..6b9dbf9006 --- /dev/null +++ b/tests/experiments/flow-v2-preview.yaml @@ -0,0 +1,76 @@ +experiment_id: flow-v2-preview +description: >- + Flow v2: the three preview Maestro builder-SDK skills + (uipath-maestro-{flow,case,bpmn}) as the ONLY skill catalog, under the docker + driver. Shadows the shipped v1 skills of the same name by pointing the catalog + at preview/ alone, so a run measures the SDK authoring path rather than a mix + of both generations. + +defaults: + run_limits: + max_turns: 200 + task_timeout: 1200 + turn_timeout: 900 + + sandbox: + driver: docker + docker: + image: skills-codex:latest + network: bridge + env_passthrough_extra: + - SKILLS_REPO_PATH + # GitHub Packages auth for `npm install @uipath/flow-sdk`. + - NODE_AUTH_TOKEN + - UIPATH_CLI_DISABLE_VERSION_SYNC + # Parity with nightly.yaml. Inert unless a Delegate runtime sets them. + - UIPATH_CLI_ENABLE_ENV_AUTH + - UIPATH_CLI_AUTH_TOKEN + - UIPATH_CLI_ORGANIZATION_NAME + - UIPATH_CLI_ORGANIZATION_ID + - UIPATH_CLI_TENANT_NAME + - UIPATH_CLI_TENANT_ID + - E2E_PROCESS_KEY + - E2E_LONG_PROCESS_KEY + extra_mounts: + # uip login state. Same destination nightly.yaml uses; do not "fix" it. + - ~/.uipath:/.uipath:rw + # Repo root. Narrowing plugins.path to preview/ drops the automatic + # repo-root mount, and criteria shell out to + # $SKILLS_REPO_PATH/tests/tasks/**/_shared/*.py. Destination must be the + # host path, which the framework only expands from 0.11.2 on + # (UiPath/coder_eval#128); earlier versions reject it at load. + - $SKILLS_REPO_PATH:$SKILLS_REPO_PATH:ro + + agent: + permission_mode: acceptEdits + allowed_tools: ["Skill", "Bash", "Read", "Write", "Edit", "Glob", "Grep"] + plugins: + # PREVIEW ONLY. Also the source of the automatic repo-path bind mount. + - type: "local" + path: "$SKILLS_REPO_PATH/preview" + ignore_patterns: [] + + pre_run: + # Runtime npm auth for the @uipath scope. The shared image ships none: its + # build-time npmrc carries a literal token and is deleted in the same layer. + # @uipath/flow-sdk is published only to GitHub Packages, so without this + # every in-sandbox `npm install @uipath/flow-sdk` 404s against the public + # registry and each compile fails with it. Kept here rather than in + # tests/docker/Dockerfile so npm resolution is unchanged for every other + # suite. Written to $HOME, which is where npm resolves userconfig, so it + # holds whatever HOME the runner forwards. ${NODE_AUTH_TOKEN} stays literal: + # npm expands it at read time, so no token is written to disk. + - command: |- + mkdir -p "$HOME" && printf '%s\n' \ + '@uipath:registry=https://npm.pkg.github.com/' \ + '//npm.pkg.github.com/:_authToken=${NODE_AUTH_TOKEN}' \ + > "$HOME/.npmrc" + timeout: 30 + + post_run: + # Unpruned node_modules costs gigabytes across a full Maestro run. + - command: "find . -maxdepth 5 -type d \\( -name node_modules -o -name .npm-prefix -o -name .venv \\) -prune -exec rm -rf {} +" + timeout: 30 + +variants: + - variant_id: default diff --git a/tests/experiments/same-ground-headtohead.yaml b/tests/experiments/same-ground-headtohead.yaml index 1cab1996b7..892659cec8 100644 --- a/tests/experiments/same-ground-headtohead.yaml +++ b/tests/experiments/same-ground-headtohead.yaml @@ -28,10 +28,10 @@ defaults: - SKILLS_REPO_PATH - UIPATH_CLI_DISABLE_VERSION_SYNC extra_mounts: - # coder-eval forwards the host HOME into this image, so the destination - # must match that path until UiPath/coder_eval#100 makes HOME portable. - - ${SG_UIPATH_HOME}:/home/tmatup/.uipath:rw - - ${SG_EMPTY_SKILLS}:/home/tmatup/.uipath/.skills:ro + # Same login destination nightly.yaml uses. Deeper target wins, so + # .skills still shadows. + - ${SG_UIPATH_HOME}:/.uipath:rw + - ${SG_EMPTY_SKILLS}:/.uipath/.skills:ro agent: type: claude-code