diff --git a/.github/workflows/run-coder-eval.yml b/.github/workflows/run-coder-eval.yml index df40b5bb73..b0b3c7bdef 100644 --- a/.github/workflows/run-coder-eval.yml +++ b/.github/workflows/run-coder-eval.yml @@ -23,6 +23,14 @@ on: description: 'REQUIRED. Space-separated globs under tests/. E.g. tasks/uipath-agents/**/*.yaml' type: string required: true + # Per-skill configs differ only in defaults the runtime shares, so the + # file is a parameter rather than a fork of this workflow. Scope it with + # task_globs: a config's defaults apply to whatever that run selects. + experiment: + description: 'Experiment YAML under tests/. E.g. experiments/flow.yaml for the flow zero-shot config.' + type: string + required: false + default: 'experiments/nightly.yaml' # Host-level concurrency for the Linux job's `coder-eval -j`. Default 4: # ubuntu-latest is 4 vCPU, and j=20 oversubscribes ~5:1 — agents miss the # 1200s turn timeout and ERROR (false negatives, including a bindings @@ -163,8 +171,9 @@ jobs: - name: Resolve globs, split by tag, enforce large-run gate id: split env: - INPUT_GLOBS: ${{ inputs.task_globs }} - CONFIRMED: ${{ inputs.confirm_large_run }} + INPUT_GLOBS: ${{ inputs.task_globs }} + CONFIRMED: ${{ inputs.confirm_large_run }} + EXPERIMENT_YAML: ${{ inputs.experiment }} run: | set -euo pipefail if [ -z "$INPUT_GLOBS" ]; then @@ -200,6 +209,27 @@ jobs: echo "::error::Glob matches $TOTAL tasks (> 50). Tick confirm_large_run to authorize." exit 1 fi + # The flow suite's prompts carry no autonomy language; experiments/flow.yaml + # supplies it in the system prompt. Running flow tasks under any other config + # silently drops it, and running the simulated tasks under flow.yaml tells a + # task that HAS a user that nobody is present. Catch both here rather than + # discover them in the results. Word-split, not "${ARR[@]}": the arrays are + # empty on one platform and that is an unbound-variable error under `set -u`. + FLOW_N=0; SIM_N=0 + for f in ${LINUX[*]:-} ${WINDOWS[*]:-}; do + case "$f" in + *tasks/uipath-maestro-flow/interactive/*) SIM_N=$((SIM_N + 1)) ;; + *tasks/uipath-maestro-flow/*) FLOW_N=$((FLOW_N + 1)) ;; + esac + done + if [ "$FLOW_N" -gt 0 ] && [ "$EXPERIMENT_YAML" != "experiments/flow.yaml" ]; then + echo "::error::$FLOW_N flow task(s) selected under '$EXPERIMENT_YAML'. Flow task prompts carry no autonomy language; experiments/flow.yaml supplies it. Set the experiment input, and dispatch flow separately from other skills." + exit 1 + fi + if [ "$SIM_N" -gt 0 ] && [ "$EXPERIMENT_YAML" = "experiments/flow.yaml" ]; then + echo "::error::$SIM_N task(s) under uipath-maestro-flow/interactive/ have a simulated user and must not run under experiments/flow.yaml, which states no user is present. Exclude that directory from the glob." + exit 1 + fi # Space-separated; downstream jobs word-split on consumption. echo "linux_globs=${LINUX[*]:-}" >> "$GITHUB_OUTPUT" echo "windows_globs=${WINDOWS[*]:-}" >> "$GITHUB_OUTPUT" @@ -409,6 +439,7 @@ jobs: E2E_LONG_PROCESS_KEY: ${{ secrets.E2E_LONG_PROCESS_KEY }} TASK_GLOBS: ${{ needs.partition.outputs.linux_globs }} TASK_COUNT: ${{ needs.partition.outputs.linux_count }} + EXPERIMENT_YAML: ${{ inputs.experiment }} TASK_PARALLELISM: ${{ inputs.parallelism }} AGENT: ${{ inputs.agent }} AGENT_MODEL: ${{ inputs.agent_model }} @@ -465,7 +496,7 @@ jobs: # Args appended to `coder-eval run`, ONE PER LINE, each verbatim — no # word splitting, no pathname expansion. That is what the delegate `-D` # override needs: to bash its `[...]` list is a character class. - args=(-e experiments/nightly.yaml) + args=(-e "${EXPERIMENT_YAML}") # Agent selection. codex authenticates via CODEX_API_KEY/CODEX_BASE_URL, # antigravity via GEMINI_API_KEY (SDK baked into the agent image), diff --git a/skills/uipath-maestro-flow/SKILL.md b/skills/uipath-maestro-flow/SKILL.md index bd3fb65a15..2463da2399 100644 --- a/skills/uipath-maestro-flow/SKILL.md +++ b/skills/uipath-maestro-flow/SKILL.md @@ -61,7 +61,7 @@ Guide for creating, editing, validating, debugging, publishing, diagnosing, and > **Tool vocabulary.** `Edit` means in-place replacement, `Write` a full-file write, `Read`/`Glob`/`Grep` file access, `Bash` shell, and a progress list the harness task list. Map them to equivalent tools elsewhere; preserve reviewable diffs and use shell file edits only as a last resort. 1. **Use `--output json`; prefer `--output-filter` for extraction.** Filters are global and run against the `Data` envelope, so expressions start at `Data` without a `Data.` prefix. Registry search returns a flat PascalCase array (`NodeType`, `DisplayName`, `Description`, `AvailableOnTenant`), not `Data.Nodes` or lowercase fields. Example: `uip maestro flow registry search --output json --output-filter "[*].{NodeType:NodeType,DisplayName:DisplayName,Description:Description,AvailableOnTenant:AvailableOnTenant}"`. With `--local`, omit `AvailableOnTenant`. Use `python3 -c` or `jq` only after verifying shape and when JMESPath cannot express the transform. See [cli-conventions.md §3](references/shared/cli-conventions.md#3-prefer---output-filter-for-extraction). -2. **Do not run `flow debug` without explicit user consent.** It executes the flow for real (sends emails, posts messages, calls APIs). +2. **`flow debug` consent comes from the mandate.** It executes the flow for real (sends emails, posts messages, calls APIs), so run it only when the request is for a flow that *works*: the user asked for something that does X, or said make it work, get it running, iterate until it passes. Building and validating does not discharge that, and a flow that was never executed is not finished. Ask when the request stops short of a working artifact (review this, add a node, validate only); with nobody to ask, report debug as the step not run. Debug also overwrites the Studio Web solution matching the local `.uipx` `SolutionId`, so never debug a solution this run did not scaffold. 3. **Search before creating or declaring resources absent.** For named agents, API workflows, RPA processes, and similar resources: (a) pull and search the tenant registry with `uip maestro flow registry pull --force && uip maestro flow registry search "" --output json`; pull first because the cache expires after 30 minutes, login is required, and only published resources are returned; (b) search locally with `uip maestro flow registry list --local --output json` or `search "" --local` (no login; returns sibling projects in the same `.uipx` solution); an empty keyword search does not prove absence, so confirm with `list --local`; (c) scaffold, mock, or create only when both searches find no match and the user explicitly requests embedding/creation or no published resource satisfies the need. "Coded" and "low-code" describe implementation style, not inline status. Use `uipath.agent.autonomous` only when explicitly asked to embed/inline/create an agent. Use `core.logic.mock` only when the resource is neither in the solution nor published. See [rpa](references/author/plugins/rpa/impl.md) and [agent](references/author/plugins/agent/impl.md). @@ -73,7 +73,7 @@ Guide for creating, editing, validating, debugging, publishing, diagnosing, and **Two tells that you skipped the search and took the brand-name shortcut — both are build defects, not valid manual-mode HTTP:** (a) you authored a manual-mode `core.action.http.v2` node whose `url` targets a well-known SaaS API domain that has a connector (`slack.com/api/*`, `api.github.com`, `*.salesforce.com`, `graph.microsoft.com`, …); (b) you declared an `in` variable to hold that service's API token or secret (e.g. a `slackToken` holding an `xoxb-…` bot token, an `apiKey`, a bearer token). A connector-backed flow never carries the raw credential — the IS connection does. If you find yourself writing either, **stop**: run `uip maestro flow registry search ""` and `uip is connections list "" --all-folders`, then use the connector activity (or connector-mode HTTP: `authentication:"connector"` + `targetConnector` + a bound `connectionId`/`folderKey`). Manual mode is legitimate only for a service the search proves has no connector. 4. **Never invoke other skills automatically** — when a flow needs an RPA process, agent, or app, identify the gap and provide handoff instructions. Let the user decide when to switch skills. **One exception — IXP extraction with documents in hand:** when the flow needs document extraction, the user supplied sample documents, and `registry search "uipath.ixp"` shows no extractor covering them, invoke the `uipath-ixp` skill to build and deploy the model, then resume the flow ([plugins/ixp/impl.md — If the Model Does Not Exist Yet](references/author/plugins/ixp/impl.md#if-the-model-does-not-exist-yet)). Resolve the target Orchestrator folder for the deployment before invoking — from the user's request when it names one, otherwise per rule #5 (its non-interactive fallback applies) — and pass it in the handoff; the sibling stops rather than guess a folder. There is deliberately no separate consent gate on the tenant writes this creates: the project and folder deployment fulfil the extraction request itself, and the one consequential choice — where the deployment lands (deployments have no delete API) — is exactly the folder decision rule #5 just routed. Do NOT drive `uip ixp` project or deployment commands from this skill instead of invoking it — the sibling's guides carry guardrails this skill does not. If `uipath-ixp` is unavailable in the session, fall back to `core.logic.mock` plus an Open Questions entry, exactly as when no documents were supplied. -5. **Always present finite decisions as a dropdown with a final "Something else" escape hatch.** Whenever the skill needs a decision (which solution, publish vs debug vs deploy, which connector, trigger type, or resource to bind, etc.), ask with the enumerated choices plus **"Something else"** last for free-form input; never ask open-ended in chat when a finite set of sensible defaults exists. If the user picks "Something else", parse their answer and continue. No structured-question facility on the harness → ask in chat as a numbered list with "Something else" last. Non-interactively (CI/headless, no user available) → take the marked recommended option, proceed, and record the decision prominently in the final report; if none is recommended, stop and report the open decision instead of guessing. Consent gates (`flow debug`, destructive operations) are never auto-answered — in non-interactive mode, stop and report the blocked step. These fallbacks define "ask the user" / "confirm with the user" wherever this skill's references require it. +5. **Always present finite decisions as a dropdown with a final "Something else" escape hatch.** Whenever the skill needs a decision (which solution, publish vs debug vs deploy, which connector, trigger type, or resource to bind, etc.), ask with the enumerated choices plus **"Something else"** last for free-form input; never ask open-ended in chat when a finite set of sensible defaults exists. If the user picks "Something else", parse their answer and continue. No structured-question facility on the harness → ask in chat as a numbered list with "Something else" last. Non-interactively (CI/headless, no user available) → take the marked recommended option, proceed, and record the decision prominently in the final report; if none is recommended, stop and report the open decision instead of guessing. Consent gates (destructive operations, tenant writes) are never auto-answered — in non-interactive mode, stop and report the blocked step; `flow debug` is not one of them, and is governed by the mandate rule above. These fallbacks define "ask the user" / "confirm with the user" wherever this skill's references require it. diff --git a/skills/uipath-maestro-flow/references/author/brownfield.md b/skills/uipath-maestro-flow/references/author/brownfield.md index 707203f2a1..501122d6fc 100644 --- a/skills/uipath-maestro-flow/references/author/brownfield.md +++ b/skills/uipath-maestro-flow/references/author/brownfield.md @@ -81,8 +81,8 @@ Authoring ends here. For any selected option, read [operate/CAPABILITY.md](../op | Option | What it does | |---|---| -| **Publish to Studio Web** (default) | Push the solution to Studio Web so the user can visualize, edit, and publish from the browser. | -| **Debug the solution** | Execute the flow end-to-end against real systems. Confirm consent first because debug has real side effects (see the consent-before-debug rule in [SKILL.md](../../SKILL.md)). | +| **Publish to Studio Web** | Push the solution to Studio Web so the user can visualize, edit, and publish from the browser. | +| **Debug the solution** | Execute the flow end-to-end against real systems. Consent comes from the mandate, not from this menu — see the `flow debug` rule in [SKILL.md](../../SKILL.md). Selecting it here is the user asking for a run. | | **Deploy to Orchestrator** | Pack and publish directly to Orchestrator (bypasses Studio Web). Only when explicitly chosen; see [/uipath:uipath-platform](/uipath:uipath-platform). | | **Something else** | Last option. Accept free-form string input and act on it. | diff --git a/skills/uipath-maestro-flow/references/author/greenfield.md b/skills/uipath-maestro-flow/references/author/greenfield.md index 9f2d049dc1..162b44b3c0 100644 --- a/skills/uipath-maestro-flow/references/author/greenfield.md +++ b/skills/uipath-maestro-flow/references/author/greenfield.md @@ -373,8 +373,8 @@ Authoring terminates here. Each option below hands off to Operate — read [oper | Option | What it does | | --- | --- | -| **Publish to Studio Web** (default) | Push the solution to Studio Web so the user can visualize, edit, and publish from the browser. | -| **Debug the solution** | Execute the flow end-to-end against real systems. Confirm consent first — debug has real side effects (see the consent-before-debug rule in [SKILL.md](../../SKILL.md)). | +| **Publish to Studio Web** | Push the solution to Studio Web so the user can visualize, edit, and publish from the browser. | +| **Debug the solution** | Execute the flow end-to-end against real systems. Consent comes from the mandate, not from this menu — see the `flow debug` rule in [SKILL.md](../../SKILL.md). Selecting it here is the user asking for a run. | | **Deploy to Orchestrator** | Pack and publish directly to Orchestrator (bypasses Studio Web). Only when explicitly chosen — see [/uipath:uipath-platform](/uipath:uipath-platform). | | **Something else** | Last option. Accept free-form string input and act on it (e.g., "just leave it", "pack but don't publish", "upload to a different tenant"). | diff --git a/skills/uipath-maestro-flow/references/operate/run.md b/skills/uipath-maestro-flow/references/operate/run.md index 0c808e8d61..042450301a 100644 --- a/skills/uipath-maestro-flow/references/operate/run.md +++ b/skills/uipath-maestro-flow/references/operate/run.md @@ -13,7 +13,7 @@ Execute a flow on demand and monitor progress. Three modes: **debug** (controlle ## Debug — controlled end-to-end run -> **Confirm consent first.** `flow debug` executes the flow for real — sends emails, posts messages, calls APIs. See the consent-before-debug rule in [SKILL.md](../../SKILL.md). Do not run without explicit user authorization. +> **Consent comes from the mandate.** `flow debug` executes the flow for real — sends emails, posts messages, calls APIs. Run it when the request is for a flow that works; ask when the request stops at build or validate. Never debug a solution this run did not scaffold: debug overwrites the Studio Web solution matching the local `.uipx` `SolutionId`. See the `flow debug` consent rule in [SKILL.md](../../SKILL.md). ```bash UIP_LOG_LEVEL=info uip maestro flow debug --output json diff --git a/tests/Makefile b/tests/Makefile index 72d9f1a0b4..307de79b12 100644 --- a/tests/Makefile +++ b/tests/Makefile @@ -1,9 +1,13 @@ -.PHONY: help install all smoke smoke_rpa e2e tags +.PHONY: help install all smoke smoke_rpa e2e tags flow SKILLS_REPO_PATH ?= $(shell cd .. && pwd) VENV := .venv CODER_EVAL := SKILLS_REPO_PATH=$(SKILLS_REPO_PATH) $(VENV)/bin/coder-eval TASKS := $(shell find tasks -name '*.yaml' -type f) +# Flow's zero-shot config states the run is headless, which is false for the +# simulated tasks under interactive/. The exclusion is the config's contract, +# so it lives with the target rather than in a comment someone has to find. +FLOW_TASKS := $(shell find tasks/uipath-maestro-flow -name '*.yaml' -type f -not -path '*/interactive/*') TASK_PARALLELISM ?= 1 help: ## Show available commands @@ -38,6 +42,9 @@ smoke_rpa: ## Run all Windows RPA smoke tests (tempdir) e2e: ## Run all end-to-end tests $(CODER_EVAL) run $(TASKS) -e experiments/default.yaml --tags e2e -j $(TASK_PARALLELISM) -v +flow: ## Run the flow suite zero-shot (headless system prompt; excludes interactive/) + $(CODER_EVAL) run $(FLOW_TASKS) -e experiments/flow.yaml -j $(TASK_PARALLELISM) -v + tags: ## Run tests matching one or more tags: make tags TAGS="connector-feature" [EXPERIMENT=experiments/default.yaml] @if [ -z "$(TAGS)" ]; then echo "Usage: make tags TAGS=\"tag1 tag2\" [EXPERIMENT=experiments/.yaml]"; exit 2; fi @TAGS="$(TAGS)" python3 -c 'import os,re,sys,glob; \ diff --git a/tests/README.md b/tests/README.md index abddabdff2..1763d63263 100644 --- a/tests/README.md +++ b/tests/README.md @@ -52,6 +52,10 @@ make smoke_rpa # Run e2e-tagged tests under the default config make e2e +# Run the flow suite zero-shot: experiments/flow.yaml states the run is headless, +# and the glob excludes interactive/, whose tasks have a simulated user +make flow + # Run tests matching a combination of tags (AND semantics — tasks must carry all listed tags) (defaults to experiments/default.yaml): make tags TAGS="integration connector-feature" # Optionally override the experiment config @@ -176,6 +180,7 @@ Run-time caps live under `defaults.run_limits` (see coder_eval `RunLimits`). | `default.yaml` | tempdir | Devs locally, ad-hoc runs | 200 | 1200s | 900s | | `nightly.yaml` | docker | Nightly cron (`daily.sh`) | 200 | 1200s | 900s | | `smoke.yaml` | docker | PR-gate smoke (Linux) | 40 | 900s | 900s | +| `flow.yaml` | docker | `make flow` / flow dispatches — nightly's runtime plus a system prompt stating the run is headless | 200 | 1200s | 900s | | `smoke-windows.yaml` | tempdir | PR-gate smoke (Windows RPA only) | 40 | 900s | 900s | | `activation.yaml` | tempdir | Skill activation classifier (benchmark) | 3 + early-stop | 360s | 120s | | `same-ground-headtohead.yaml` | docker | Campaign-only local comparison arm | 200 | 1200s | 900s | diff --git a/tests/experiments/flow.yaml b/tests/experiments/flow.yaml new file mode 100644 index 0000000000..10e9cf2d01 --- /dev/null +++ b/tests/experiments/flow.yaml @@ -0,0 +1,88 @@ +experiment_id: skill-flow-zero-shot +description: > + Flow zero-shot config — identical runtime to nightly.yaml, with a system + prompt that states the run is headless. Pair it with a task glob that + excludes tests/tasks/uipath-maestro-flow/interactive/**, whose tasks have a + simulated user and must not be told nobody is present. + +defaults: + run_limits: + max_turns: 200 + task_timeout: 1200 + # A single turn can chain several cold-runner `uip rpa` calls (each 30-90s); + # per-turn budget tracks the slowest external call, not the suite's intent. + turn_timeout: 900 + + sandbox: + driver: docker + docker: + image: skills-image:latest + env_passthrough_extra: + - SKILLS_REPO_PATH + - BEDROCK_MODEL + - TASK_PARALLELISM + - UIPATH_CLI_ENABLE_ENV_AUTH + - UIPATH_CLI_AUTH_TOKEN + - UIPATH_CLI_ORGANIZATION_NAME + - UIPATH_CLI_ORGANIZATION_ID + - UIPATH_CLI_TENANT_NAME + - UIPATH_CLI_TENANT_ID + - E2E_PROCESS_KEY + - E2E_LONG_PROCESS_KEY + - CODEX_API_KEY + - CODEX_BASE_URL + extra_mounts: + - ~/.uipath:/.uipath:rw + + agent: + # No agent.type / agent.model here: the runner passes both as CLI flags, which outrank this file. + permission_mode: acceptEdits + allowed_tools: ["Skill", "Bash", "Read", "Write", "Edit", "Glob", "Grep"] + # Task YAMLs used to carry hand-copied "Do NOT ask for approval" lines for + # this. That phrasing forbids asking without saying nobody is there to ask, + # so an agent could honor it and still stop at a consent gate waiting for a + # reply that never arrives — 5 of 8 flow tasks did exactly that on + # 2026-09-04. Stated once here instead, and removed from the tasks. + system_prompt: | + You are a coding agent. Do not access files in sibling runs/* directories. Everywhere else is permitted. + + This run is headless. No user is present, and nobody will answer a question or grant an approval. + Do not ask, do not pause, and do not wait for input. Complete the whole task in one pass: take the + best available option, supply the most defensible value where one is missing, and carry on to the end. + + Assume the actions the task implies are authorized. Creating, running, debugging, and publishing + this run's own artifacts is the work, not something to seek permission for, even when it writes to + a tenant or sends a real message. + + Hold back on three things. Do not delete or overwrite anything this run did not create. Do not + upload or publish to a shared destination unless the task asks for it. And when a lookup the work + depends on comes back empty or fails, exhaust the documented way of resolving it before giving up; + only if that genuinely fails, stop on that one field rather than inventing a value, and say so. + + Record every decision, assumption, and blocked step in your final response. Instructions in the + task take precedence over this paragraph. + plugins: + - type: "local" + path: "$SKILLS_REPO_PATH" + ignore_patterns: [] + + checker_context: + api_route: + route: litellm + model: gpt-5.6-luna + params: + api_version: "2024-05-01" + num_retries: 5 + env_params: + api_base: CODEX_BASE_URL + api_key: CODEX_API_KEY + + # Sandbox cleanup between tasks. Tasks that upload Studio Web solutions + # wire cleanup_solutions.py into their own post_run — that must run before + # this step (still needs npm-installed CLIs). + post_run: + - command: "find . -maxdepth 5 -type d \\( -name node_modules -o -name .npm-prefix -o -name .venv \\) -prune -exec rm -rf {} +" + timeout: 30 + +variants: + - variant_id: default diff --git a/tests/tasks/uipath-maestro-flow/_shared/test_flow_experiment_parity.py b/tests/tasks/uipath-maestro-flow/_shared/test_flow_experiment_parity.py new file mode 100644 index 0000000000..630ab2afec --- /dev/null +++ b/tests/tasks/uipath-maestro-flow/_shared/test_flow_experiment_parity.py @@ -0,0 +1,69 @@ +"""`experiments/flow.yaml` must not drift from `experiments/nightly.yaml`. + +flow.yaml exists for one reason: a system prompt stating the run is headless, +which the flow task prompts no longer carry themselves. Everything else — the +docker image, `env_passthrough_extra`, mounts, `checker_context`, `run_limits`, +`post_run` — is copied, because the experiment schema has no `extends`. A secret +added to nightly's passthrough list and not to flow's breaks flow runs with a +missing environment variable and no obvious cause. + +This asserts every substantive line of nightly.yaml still appears in flow.yaml. +It does not assert the reverse: flow.yaml is allowed to add the headless +paragraph and its own id and description. + +Regex, not PyYAML: CI installs only pytest, and a module-level `import yaml` +would error at collection and take the suite with it (see test_criterion_budgets). +""" + +from __future__ import annotations + +import os + +_HERE = os.path.dirname(os.path.abspath(__file__)) +_EXPERIMENTS = os.path.normpath(os.path.join(_HERE, "..", "..", "..", "experiments")) +_NIGHTLY = os.path.join(_EXPERIMENTS, "nightly.yaml") +_FLOW = os.path.join(_EXPERIMENTS, "flow.yaml") + +# Lines that legitimately differ: the identity of the config itself. +_EXEMPT_PREFIXES = ("experiment_id:", "description:") + + +def _substantive(path: str) -> list[str]: + """Non-blank, non-comment lines, minus the config's own identity block.""" + out: list[str] = [] + in_description = False + for raw in open(path, encoding="utf-8"): + line = raw.rstrip("\n") + stripped = line.strip() + if stripped.startswith("description:"): + # Folded scalar: skip its indented continuation lines too. + in_description = stripped.endswith((">", "|")) + continue + if in_description: + if line and not line[0].isspace(): + in_description = False + else: + continue + if not stripped or stripped.startswith("#"): + continue + if stripped.startswith(_EXEMPT_PREFIXES): + continue + out.append(line) + return out + + +def test_flow_config_carries_every_nightly_setting(): + flow = set(_substantive(_FLOW)) + missing = [ln for ln in _substantive(_NIGHTLY) if ln not in flow] + assert not missing, ( + "experiments/flow.yaml has drifted from nightly.yaml. Copy these lines over " + "(flow.yaml is a snapshot of nightly's runtime plus a headless system prompt):\n " + + "\n ".join(missing) + ) + + +def test_flow_config_states_the_run_is_headless(): + """The one thing flow.yaml exists to add. The task prompts no longer say it.""" + text = open(_FLOW, encoding="utf-8").read() + assert "This run is headless." in text + assert "No user is present" in text diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/contractregistry_crud_filters.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/contractregistry_crud_filters.yaml index e26e458ccc..49a446dfc6 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/contractregistry_crud_filters.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/contractregistry_crud_filters.yaml @@ -35,9 +35,6 @@ initial_prompt: | 6. Map the created, queried, updated, and retrieved values to the flow output where the activity supports output mapping. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml index 420f028c7e..dc544a60bb 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml @@ -41,9 +41,6 @@ initial_prompt: | Entity Record ONCE inside the loop, binding recordId to the loop item's Id. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/integration_create_get.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/integration_create_get.yaml index 2b91cd6977..2b68f20942 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/integration_create_get.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/integration_create_get.yaml @@ -37,9 +37,6 @@ initial_prompt: | at expansionLevel=1, expansionLevel=2, and expansionLevel=3. Delete when done. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_create_all_types.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_create_all_types.yaml index a93cfbad58..bce50ac79b 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_create_all_types.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_create_all_types.yaml @@ -30,9 +30,6 @@ initial_prompt: | * lastUpdated: "2024-03-10T09:00:00" (DATETIME) * externalId: "aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee" (UUID) - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_error.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_error.yaml index db20860861..ec269379bf 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_error.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_error.yaml @@ -22,9 +22,6 @@ initial_prompt: | FlowCodeEvalEntity for title='ErrorTestRecord'. 3. In the same parallel branch, Query FlowCodeEvalEntity with no filter. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_file_activities.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_file_activities.yaml index 281fa4a058..5b232376da 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_file_activities.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_file_activities.yaml @@ -36,9 +36,6 @@ initial_prompt: | 4. DeleteFileFromRecordFieldV2 — bind recordId to the same create activity output Id used in step 3. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_query.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_query.yaml index cdc0523780..bcc6d8e525 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_query.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_query.yaml @@ -39,9 +39,6 @@ initial_prompt: | tree, sort, and limit, but use offset/start 2 for the next page. Query 3 must filter active equals true, sort by score descending, and use a limit of 4. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update.yaml index 30189f76ab..0dfae2ad9f 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update.yaml @@ -24,9 +24,6 @@ initial_prompt: | 3. Retrieves the record by Id to confirm. 4. Deletes the record. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The Flow is not complete until `uip maestro flow validate` passes. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update_existing_flow.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update_existing_flow.yaml index 5239218ff2..5d64379e88 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update_existing_flow.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/smoke_update_existing_flow.yaml @@ -31,9 +31,6 @@ initial_prompt: | 3. Then revert both changes so the activity is back to no filter, no ordering, and its original settings. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. The flow is not complete until the final state matches the initial state on those three fields and `uip maestro flow validate` passes. diff --git a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/trigger_lifecycle.yaml b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/trigger_lifecycle.yaml index 525edc9e8b..15d3e66620 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/trigger_lifecycle.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/datafabric_connector/trigger_lifecycle.yaml @@ -35,9 +35,6 @@ initial_prompt: | For both flows, keep the configured connection and correct entity on every trigger and Data Fabric activity. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flows end-to-end in a - single pass. Validate every final flow file. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/drive_to_slack.yaml b/tests/tasks/uipath-maestro-flow/connector_features/drive_to_slack.yaml index 3bf8394980..b6dc1a1523 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/drive_to_slack.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/drive_to_slack.yaml @@ -20,7 +20,7 @@ run_limits: task_timeout: 1500 max_turns: 120 turn_timeout: 1500 -initial_prompt: "Create a new Flow project called \"DriveToSlackTest\" with a manual trigger.\nIt should download a file from Google Drive and then post that file into a \nSlack channel named \"coding-agent-testing\" using Slack.\nWire the file output of the download operation to the channel input of the slack send file. \nValidate the final flow file.\nFor the Google Drive Download File activity, use the first available file.\nFor Send File, send as `user`.\nFor google drive, use the connection name 'is.sandboxes.test@gmail.com', and for slack\nuse the connection name 'is-sandboxes'.\n\nDo NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass.\n" +initial_prompt: "Create a new Flow project called \"DriveToSlackTest\" with a manual trigger.\nIt should download a file from Google Drive and then post that file into a \nSlack channel named \"coding-agent-testing\" using Slack.\nWire the file output of the download operation to the channel input of the slack send file. \nValidate the final flow file.\nFor the Google Drive Download File activity, use the first available file.\nFor Send File, send as `user`.\nFor google drive, use the connection name 'is.sandboxes.test@gmail.com', and for slack\nuse the connection name 'is-sandboxes'.\n" success_criteria: - type: command_executed description: uip maestro flow validate was called diff --git a/tests/tasks/uipath-maestro-flow/connector_features/generic_dynamic_node/generic_dynamic_node.yaml b/tests/tasks/uipath-maestro-flow/connector_features/generic_dynamic_node/generic_dynamic_node.yaml index a25f18494f..63cee0e53e 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/generic_dynamic_node/generic_dynamic_node.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/generic_dynamic_node/generic_dynamic_node.yaml @@ -38,9 +38,7 @@ initial_prompt: | Use the connection present in Shared/uipath-maestro-flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build, validate, and run end-to-end in a single - pass. Do NOT substitute a mock for the + Do NOT substitute a mock for the connector. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml b/tests/tasks/uipath-maestro-flow/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml index 82db43ecfb..0ce792e8bb 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml @@ -52,9 +52,6 @@ initial_prompt: | Validate the final flow file and debug it. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "uip maestro flow validate was called" diff --git a/tests/tasks/uipath-maestro-flow/connector_features/non-catalog-http-fallback/non_catalog_http_fallback.yaml b/tests/tasks/uipath-maestro-flow/connector_features/non-catalog-http-fallback/non_catalog_http_fallback.yaml index 44e7d74bb8..d3a678ca79 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/non-catalog-http-fallback/non_catalog_http_fallback.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/non-catalog-http-fallback/non_catalog_http_fallback.yaml @@ -28,8 +28,6 @@ initial_prompt: | Validate the final flow file. Use `--output json` on every `uip` command whose output you parse. - Do NOT ask for approval, confirmation, or feedback — build and validate the - flow end-to-end in a single pass. success_criteria: - type: command_executed diff --git a/tests/tasks/uipath-maestro-flow/connector_features/paginated_reference_lookup.yaml b/tests/tasks/uipath-maestro-flow/connector_features/paginated_reference_lookup.yaml index bcdcbdb0d5..6e8f71551d 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/paginated_reference_lookup.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/paginated_reference_lookup.yaml @@ -34,8 +34,6 @@ initial_prompt: | For Slack, use the connection name `is-sandboxes`. If more than one connection matches that name, use any enabled one. Validate the flow when you're done. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: - type: command_executed diff --git a/tests/tasks/uipath-maestro-flow/connector_features/slack-http-fallback/slack_http_fallback.yaml b/tests/tasks/uipath-maestro-flow/connector_features/slack-http-fallback/slack_http_fallback.yaml index dcedeb9bba..f029f84495 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/slack-http-fallback/slack_http_fallback.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/slack-http-fallback/slack_http_fallback.yaml @@ -25,8 +25,6 @@ initial_prompt: | Validate the final flow file. Once it validates, debug it. Use `--output json` on every `uip` command whose output you parse. - Do NOT ask for approval, confirmation, or feedback — build, validate, and - debug the flow end-to-end in a single pass. success_criteria: - type: command_executed diff --git a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_attachments/testmanager_attachments.yaml b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_attachments/testmanager_attachments.yaml index 7c8a562ce1..37ad851663 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_attachments/testmanager_attachments.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_attachments/testmanager_attachments.yaml @@ -29,9 +29,6 @@ initial_prompt: | The task is not complete until validation passed for the flow. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "Agent refreshed the node manifest (`uip maestro flow registry pull`) before searching" diff --git a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml index 28df13cfa5..96baf70526 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml @@ -34,9 +34,6 @@ initial_prompt: | The task is not complete until validation passed for the flow. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "Agent refreshed the node manifest (`uip maestro flow registry pull`)" diff --git a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_execution_results/testmanager_execution_results.yaml b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_execution_results/testmanager_execution_results.yaml index 92aecc8cd6..4e34afb35f 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_execution_results/testmanager_execution_results.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_execution_results/testmanager_execution_results.yaml @@ -34,9 +34,6 @@ initial_prompt: | The task is not complete until validation passed for the flow. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "Agent refreshed the node manifest (`uip maestro flow registry pull`) before searching" diff --git a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_generic_records/testmanager_generic_records.yaml b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_generic_records/testmanager_generic_records.yaml index dc8caf370e..9904502ac5 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_generic_records/testmanager_generic_records.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_generic_records/testmanager_generic_records.yaml @@ -30,9 +30,6 @@ initial_prompt: | The task is not complete until validation passed for the flow. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "Agent refreshed the node manifest (`uip maestro flow registry pull`) before searching" diff --git a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_requirement_lifecycle/testmanager_requirement_lifecycle.yaml b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_requirement_lifecycle/testmanager_requirement_lifecycle.yaml index b02e5261b1..d85af72684 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_requirement_lifecycle/testmanager_requirement_lifecycle.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_requirement_lifecycle/testmanager_requirement_lifecycle.yaml @@ -31,9 +31,6 @@ initial_prompt: | The task is not complete until validation passed for the flow. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "Agent refreshed the node manifest (`uip maestro flow registry pull`) before searching" diff --git a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testcase_lifecycle/testmanager_testcase_lifecycle.yaml b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testcase_lifecycle/testmanager_testcase_lifecycle.yaml index 83716c6a24..d2f04c4e42 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testcase_lifecycle/testmanager_testcase_lifecycle.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testcase_lifecycle/testmanager_testcase_lifecycle.yaml @@ -30,9 +30,6 @@ initial_prompt: | The task is not complete until validation passed for the flow. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "Agent refreshed the node manifest (`uip maestro flow registry pull`) before searching" diff --git a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testset_lifecycle/testmanager_testset_lifecycle.yaml b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testset_lifecycle/testmanager_testset_lifecycle.yaml index cd366f1db8..7f236037ea 100644 --- a/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testset_lifecycle/testmanager_testset_lifecycle.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_features/testmanager_testset_lifecycle/testmanager_testset_lifecycle.yaml @@ -32,9 +32,6 @@ initial_prompt: | The task is not complete until validation passed for the flow. - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. - success_criteria: - type: command_executed description: "Agent refreshed the node manifest (`uip maestro flow registry pull`) before searching" diff --git a/tests/tasks/uipath-maestro-flow/connector_trigger/trigger_with_filter.yaml b/tests/tasks/uipath-maestro-flow/connector_trigger/trigger_with_filter.yaml index 3f23be96b4..765d5cacf2 100644 --- a/tests/tasks/uipath-maestro-flow/connector_trigger/trigger_with_filter.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_trigger/trigger_with_filter.yaml @@ -16,7 +16,6 @@ initial_prompt: | `uip maestro flow node configure --detail ''` to configure that trigger. Use placeholders for any connection/folder/event ids you cannot resolve offline. - success_criteria: # `check_trigger_filter.py` locates trigger_detail.json (root or nested in the # flow project dir) and asserts it is the --detail object: valid JSON, no diff --git a/tests/tasks/uipath-maestro-flow/connector_trigger/webhook_waitfor_parallel.yaml b/tests/tasks/uipath-maestro-flow/connector_trigger/webhook_waitfor_parallel.yaml index 2189db7937..05bb446a77 100644 --- a/tests/tasks/uipath-maestro-flow/connector_trigger/webhook_waitfor_parallel.yaml +++ b/tests/tasks/uipath-maestro-flow/connector_trigger/webhook_waitfor_parallel.yaml @@ -45,8 +45,6 @@ initial_prompt: | After building, validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and - implementation. Build the complete flow end-to-end in a single pass. Use `--output json` on `uip` commands. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/context-grounding/batch_transform/batch_transform.yaml b/tests/tasks/uipath-maestro-flow/context-grounding/batch_transform/batch_transform.yaml index a81538c530..a87c7063f7 100644 --- a/tests/tasks/uipath-maestro-flow/context-grounding/batch_transform/batch_transform.yaml +++ b/tests/tasks/uipath-maestro-flow/context-grounding/batch_transform/batch_transform.yaml @@ -42,8 +42,6 @@ initial_prompt: | `outputs.result.source: "=js:$vars..output"`. Do NOT run flow debug — just validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single pass. Before starting, follow the documented workflow steps exactly — in particular, the batch-transform plugin's `outputColumns` shape (array of `{ name, description }`). diff --git a/tests/tasks/uipath-maestro-flow/context-grounding/summarize/summarize.yaml b/tests/tasks/uipath-maestro-flow/context-grounding/summarize/summarize.yaml index 74c13bbf2a..ed2cecf25f 100644 --- a/tests/tasks/uipath-maestro-flow/context-grounding/summarize/summarize.yaml +++ b/tests/tasks/uipath-maestro-flow/context-grounding/summarize/summarize.yaml @@ -44,8 +44,6 @@ initial_prompt: | resolve to `undefined` at runtime. Do NOT run flow debug — just validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single pass. Before starting, follow the documented workflow steps exactly — in particular, note that the wire-level node type is `uipath.pattern.deep-rag` even though the canvas display name is "Summarize". diff --git a/tests/tasks/uipath-maestro-flow/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml b/tests/tasks/uipath-maestro-flow/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml index c79643e94f..b97b9f3779 100644 --- a/tests/tasks/uipath-maestro-flow/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml +++ b/tests/tasks/uipath-maestro-flow/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml @@ -54,8 +54,7 @@ initial_prompt: | correlationId), and jiraIssueKey (the created issue's key). Validate the flow. Do NOT run or debug the flow — the grader executes it with the - seeded inputs. Do NOT ask for approval and do NOT pause; build end-to-end in a - single pass. Before starting, load the uipath-maestro-flow skill and follow its + seeded inputs. Before starting, load the uipath-maestro-flow skill and follow its workflow. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml b/tests/tasks/uipath-maestro-flow/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml index bb191b3e1a..899c5ddc81 100644 --- a/tests/tasks/uipath-maestro-flow/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml +++ b/tests/tasks/uipath-maestro-flow/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml @@ -82,8 +82,7 @@ initial_prompt: | Salesforce/Jira/Drive are NOT part of this task — the match status is an input. Validate the flow. Do NOT run or debug the flow — the grader executes it with - seeded inputs. Do NOT ask for approval and do NOT pause; build the flow directly. - Before starting, load the uipath-maestro-flow skill and follow its workflow. + seeded inputs. Before starting, load the uipath-maestro-flow skill and follow its workflow. success_criteria: - type: command_executed diff --git a/tests/tasks/uipath-maestro-flow/e2e/escalation_slack_alert/escalation_slack_alert.yaml b/tests/tasks/uipath-maestro-flow/e2e/escalation_slack_alert/escalation_slack_alert.yaml index 0a8076c810..e455c7d0cf 100644 --- a/tests/tasks/uipath-maestro-flow/e2e/escalation_slack_alert/escalation_slack_alert.yaml +++ b/tests/tasks/uipath-maestro-flow/e2e/escalation_slack_alert/escalation_slack_alert.yaml @@ -56,10 +56,8 @@ initial_prompt: | (from the Slack Send Message node's output) Validate the flow. Do NOT run or debug the flow — the grader executes it with - seeded inputs. Do NOT ask for approval, confirmation, or feedback, and do NOT - pause between planning and implementation. Build the complete flow end-to-end - in a single pass. Before starting, load the uipath-maestro-flow skill and - follow its workflow. + seeded inputs. Before starting, load the uipath-maestro-flow skill and follow + its workflow. success_criteria: # ── The agent validated the flow (convention adherence) ───────────────── diff --git a/tests/tasks/uipath-maestro-flow/edit/add_node/add_node.yaml b/tests/tasks/uipath-maestro-flow/edit/add_node/add_node.yaml index 96ec38a796..7e1c927f57 100644 --- a/tests/tasks/uipath-maestro-flow/edit/add_node/add_node.yaml +++ b/tests/tasks/uipath-maestro-flow/edit/add_node/add_node.yaml @@ -18,7 +18,6 @@ initial_prompt: | Add a script node called "convertToCelsius" between getWeather and formatSummary that converts the temp from F to C. Return both values. Also update formatSummary to include the Celsius value in its summary string. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/edit/add_output/add_output.yaml b/tests/tasks/uipath-maestro-flow/edit/add_output/add_output.yaml index 306bf90b46..3c35b97fd0 100644 --- a/tests/tasks/uipath-maestro-flow/edit/add_output/add_output.yaml +++ b/tests/tasks/uipath-maestro-flow/edit/add_output/add_output.yaml @@ -17,7 +17,6 @@ initial_prompt: | Add a "location" field with value "Bellevue, WA" to the summary output of both end nodes. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: - type: run_command diff --git a/tests/tasks/uipath-maestro-flow/edit/group_to_subflow/group_to_subflow.yaml b/tests/tasks/uipath-maestro-flow/edit/group_to_subflow/group_to_subflow.yaml index 73aab55567..df70d565a0 100644 --- a/tests/tasks/uipath-maestro-flow/edit/group_to_subflow/group_to_subflow.yaml +++ b/tests/tasks/uipath-maestro-flow/edit/group_to_subflow/group_to_subflow.yaml @@ -19,7 +19,6 @@ initial_prompt: | Move getWeather and formatSummary into a subflow called "fetchAndFormat". The subflow should output the temperatureF value. The rest of the main flow (decision + end nodes) stays but reads from the subflow output. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/edit/move_node/move_node.yaml b/tests/tasks/uipath-maestro-flow/edit/move_node/move_node.yaml index ddc5bb7045..b92f2e9466 100644 --- a/tests/tasks/uipath-maestro-flow/edit/move_node/move_node.yaml +++ b/tests/tasks/uipath-maestro-flow/edit/move_node/move_node.yaml @@ -17,7 +17,6 @@ initial_prompt: | Rearrange the flow so the decision happens right after the HTTP call. Both the true and false branches should merge back into formatSummary, which then goes to a single end node. formatSummary should pick the message ('nice day' or 'bring a jacket') based on which branch the decision took. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: - type: run_command diff --git a/tests/tasks/uipath-maestro-flow/edit/remove_node/remove_node.yaml b/tests/tasks/uipath-maestro-flow/edit/remove_node/remove_node.yaml index 423a266a1a..8d610ce21f 100644 --- a/tests/tasks/uipath-maestro-flow/edit/remove_node/remove_node.yaml +++ b/tests/tasks/uipath-maestro-flow/edit/remove_node/remove_node.yaml @@ -18,7 +18,6 @@ initial_prompt: | Remove the formatSummary script node. The decision and end nodes should read temperature directly from the HTTP response instead. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/edit/update_node/update_node.yaml b/tests/tasks/uipath-maestro-flow/edit/update_node/update_node.yaml index 7a150b6785..a46f6b5871 100644 --- a/tests/tasks/uipath-maestro-flow/edit/update_node/update_node.yaml +++ b/tests/tasks/uipath-maestro-flow/edit/update_node/update_node.yaml @@ -19,7 +19,6 @@ initial_prompt: | otherwise the message field should be 'go home'. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/evaluate/inline_agent_eval/inline_agent_eval.yaml b/tests/tasks/uipath-maestro-flow/evaluate/inline_agent_eval/inline_agent_eval.yaml index 1d08b8c820..951b788737 100644 --- a/tests/tasks/uipath-maestro-flow/evaluate/inline_agent_eval/inline_agent_eval.yaml +++ b/tests/tasks/uipath-maestro-flow/evaluate/inline_agent_eval/inline_agent_eval.yaml @@ -54,9 +54,6 @@ initial_prompt: | - Local CRUD only. Do NOT call `uip login`. Do NOT call `uip solution upload`. Do NOT call `uip maestro flow eval run *`. Do NOT run `uip maestro flow debug`. - - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build it end-to-end in a single pass. - success_criteria: - type: command_executed description: "Agent scaffolded the inline agent with uip agent init --inline-in-flow" diff --git a/tests/tasks/uipath-maestro-flow/hitl/smoke_01_hitl_node_placed.yaml b/tests/tasks/uipath-maestro-flow/hitl/smoke_01_hitl_node_placed.yaml index f2d648f834..2c04c7b849 100644 --- a/tests/tasks/uipath-maestro-flow/hitl/smoke_01_hitl_node_placed.yaml +++ b/tests/tasks/uipath-maestro-flow/hitl/smoke_01_hitl_node_placed.yaml @@ -40,7 +40,6 @@ success_criteria: weight: 3.0 pass_threshold: 1.0 - - type: run_command description: "uip flow validate passes" command: "python3 $SKILLS_REPO_PATH/tests/tasks/uipath-maestro-flow/_shared/validate_flow.py" diff --git a/tests/tasks/uipath-maestro-flow/ixp/e2e_01_invoice_extraction_greenfield.yaml b/tests/tasks/uipath-maestro-flow/ixp/e2e_01_invoice_extraction_greenfield.yaml index 1ad9e68f4e..f60f2b7a4f 100644 --- a/tests/tasks/uipath-maestro-flow/ixp/e2e_01_invoice_extraction_greenfield.yaml +++ b/tests/tasks/uipath-maestro-flow/ixp/e2e_01_invoice_extraction_greenfield.yaml @@ -30,7 +30,6 @@ initial_prompt: | Validate the final flow file — the task is not complete until `uip maestro flow validate` passes. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: - type: run_command description: uip maestro flow validate passes on the generated flow file diff --git a/tests/tasks/uipath-maestro-flow/ixp/e2e_03_project_creation_handoff/e2e_03_project_creation_handoff.yaml b/tests/tasks/uipath-maestro-flow/ixp/e2e_03_project_creation_handoff/e2e_03_project_creation_handoff.yaml index 0fffd7d922..a1951cc845 100644 --- a/tests/tasks/uipath-maestro-flow/ixp/e2e_03_project_creation_handoff/e2e_03_project_creation_handoff.yaml +++ b/tests/tasks/uipath-maestro-flow/ixp/e2e_03_project_creation_handoff/e2e_03_project_creation_handoff.yaml @@ -97,9 +97,6 @@ initial_prompt: | - The `uip` CLI is already available and the runner is logged in. - Use `--output json` on all uip commands. - Do NOT run `uip maestro flow debug` or `uip maestro flow deploy`. - - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build it end-to-end in a single pass. - success_criteria: - type: skill_triggered description: "Agent invoked the uipath-maestro-flow Skill" diff --git a/tests/tasks/uipath-maestro-flow/multi_node/bellevue_weather/bellevue_weather.yaml b/tests/tasks/uipath-maestro-flow/multi_node/bellevue_weather/bellevue_weather.yaml index 6905a1a767..e97fc21017 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/bellevue_weather/bellevue_weather.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/bellevue_weather/bellevue_weather.yaml @@ -17,7 +17,6 @@ initial_prompt: | Use a Managed HTTP Request node (core.action.http.v2) to call the open-meteo API directly — do not use an Integration Service connector. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/multi_node/calculator/calculator.yaml b/tests/tasks/uipath-maestro-flow/multi_node/calculator/calculator.yaml index 5a6a109a34..077b79a4e8 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/calculator/calculator.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/calculator/calculator.yaml @@ -14,7 +14,6 @@ initial_prompt: | returned as an output variable. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/multi_node/customer_escalation/customer_escalation.yaml b/tests/tasks/uipath-maestro-flow/multi_node/customer_escalation/customer_escalation.yaml index c384d2849e..cb7480a632 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/customer_escalation/customer_escalation.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/customer_escalation/customer_escalation.yaml @@ -32,9 +32,6 @@ initial_prompt: | reply to the sender via Outlook with the ticket details. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. success_criteria: # ── Flow file validity ──────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/multi_node/dice_roller/dice_roller.yaml b/tests/tasks/uipath-maestro-flow/multi_node/dice_roller/dice_roller.yaml index dd75fa2522..aac5bf221d 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/dice_roller/dice_roller.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/dice_roller/dice_roller.yaml @@ -15,7 +15,6 @@ initial_prompt: | die and outputs the result. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/multi_node/feet_inches/feet_inches.yaml b/tests/tasks/uipath-maestro-flow/multi_node/feet_inches/feet_inches.yaml index d64eab27b1..ca028f38ef 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/feet_inches/feet_inches.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/feet_inches/feet_inches.yaml @@ -20,8 +20,6 @@ initial_prompt: | ("f2i", "i2f", "y2f"); each branch applies the corresponding conversion (in a Script node) and converges on a single End node that returns the result. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. - success_criteria: # ── Flow file validity ───────────────────────────────────────────────── - type: run_command diff --git a/tests/tasks/uipath-maestro-flow/multi_node/loop_multiply/loop_multiply.yaml b/tests/tasks/uipath-maestro-flow/multi_node/loop_multiply/loop_multiply.yaml index ab64ffd4da..798d464e07 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/loop_multiply/loop_multiply.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/loop_multiply/loop_multiply.yaml @@ -14,7 +14,6 @@ initial_prompt: | numbers [13, 15, 17] together using a Loop node and returns the product. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/multi_node/multi_city_weather/multi_city_weather.yaml b/tests/tasks/uipath-maestro-flow/multi_node/multi_city_weather/multi_city_weather.yaml index 57e8b3e480..a98be8c1ce 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/multi_city_weather/multi_city_weather.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/multi_city_weather/multi_city_weather.yaml @@ -14,7 +14,6 @@ initial_prompt: | Use a Managed HTTP Request node (core.action.http.v2) to call the open-meteo API directly — do not use an Integration Service connector. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: - type: run_command diff --git a/tests/tasks/uipath-maestro-flow/multi_node/reading_list/reading_list.yaml b/tests/tasks/uipath-maestro-flow/multi_node/reading_list/reading_list.yaml index 305eccd2e3..40a4bc0efb 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/reading_list/reading_list.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/reading_list/reading_list.yaml @@ -41,7 +41,6 @@ initial_prompt: | an inline `=js:[...]` array or `=js:$vars.` expression. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/multi_node/slack_channel_description/slack_channel_description.yaml b/tests/tasks/uipath-maestro-flow/multi_node/slack_channel_description/slack_channel_description.yaml index 70df43e92c..c60193589b 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/slack_channel_description/slack_channel_description.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/slack_channel_description/slack_channel_description.yaml @@ -15,7 +15,6 @@ initial_prompt: | the channel description of #office-bellevue and outputs it. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml b/tests/tasks/uipath-maestro-flow/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml index 21f962aa1d..2f1b1df38d 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml @@ -16,7 +16,6 @@ initial_prompt: | All connections for this task live in Orchestrator folder `Shared/uipath-maestro-flow`. Validate the flow, then run `uip maestro flow debug` and iterate until it completes without incidents. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: - type: run_command diff --git a/tests/tasks/uipath-maestro-flow/multi_node/wiki_pageviews/wiki_pageviews.yaml b/tests/tasks/uipath-maestro-flow/multi_node/wiki_pageviews/wiki_pageviews.yaml index bfcf65dcd1..2865800b91 100644 --- a/tests/tasks/uipath-maestro-flow/multi_node/wiki_pageviews/wiki_pageviews.yaml +++ b/tests/tasks/uipath-maestro-flow/multi_node/wiki_pageviews/wiki_pageviews.yaml @@ -32,8 +32,6 @@ initial_prompt: | does not exist and the API returns an error), the flow must instead return the literal string `Article not found`. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. - success_criteria: # ── Flow file validity ───────────────────────────────────────────────── - type: run_command diff --git a/tests/tasks/uipath-maestro-flow/single_node/api_workflow/api_workflow.yaml b/tests/tasks/uipath-maestro-flow/single_node/api_workflow/api_workflow.yaml index eebfd349f8..232744e608 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/api_workflow/api_workflow.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/api_workflow/api_workflow.yaml @@ -14,7 +14,6 @@ initial_prompt: | API workflow with the name 'tomasz' and returns his age as an output. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/single_node/coded_agent/coded_agent.yaml b/tests/tasks/uipath-maestro-flow/single_node/coded_agent/coded_agent.yaml index 638e7ccf66..f053107444 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/coded_agent/coded_agent.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/coded_agent/coded_agent.yaml @@ -26,7 +26,6 @@ initial_prompt: | number of people who approved as an integer flow output. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: - type: run_command description: uip maestro flow validate passes on the flow file diff --git a/tests/tasks/uipath-maestro-flow/single_node/decision/decision.yaml b/tests/tasks/uipath-maestro-flow/single_node/decision/decision.yaml index fb3a14b056..0a6ca379a0 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/decision/decision.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/decision/decision.yaml @@ -17,7 +17,6 @@ initial_prompt: | the flow should output "warm". Otherwise it should output "cool". Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # -- Flow file validity -- diff --git a/tests/tasks/uipath-maestro-flow/single_node/delay/delay.yaml b/tests/tasks/uipath-maestro-flow/single_node/delay/delay.yaml index c8b35a7534..f33d200df1 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/delay/delay.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/delay/delay.yaml @@ -28,8 +28,6 @@ initial_prompt: | the delay node id, and one whose `sourceNodeId` is the delay node id). Do NOT run flow debug — just validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single pass. Before starting, follow the documented workflow steps exactly — in particular, the delay plugin's `inputs` shape (`timerType` + `timerPreset`). diff --git a/tests/tasks/uipath-maestro-flow/single_node/file_attachment/file_attachment.yaml b/tests/tasks/uipath-maestro-flow/single_node/file_attachment/file_attachment.yaml index eb10dbc135..cc00850d8d 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/file_attachment/file_attachment.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/file_attachment/file_attachment.yaml @@ -30,9 +30,7 @@ initial_prompt: | file to the input. The task is not complete until the debug run reports `finalStatus: "Completed"` and the output reflects the attached file's name. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build, validate, and run end-to-end in a single - pass. Before starting, follow the documented + Before starting, follow the documented workflow steps exactly — in particular the file-attachment binding pre-flight and the requirement to run `solution resources refresh` before debug. diff --git a/tests/tasks/uipath-maestro-flow/single_node/lowcode_agent/lowcode_agent.yaml b/tests/tasks/uipath-maestro-flow/single_node/lowcode_agent/lowcode_agent.yaml index 415736357b..2a263de6f6 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/lowcode_agent/lowcode_agent.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/lowcode_agent/lowcode_agent.yaml @@ -21,7 +21,6 @@ initial_prompt: | in rather than creating a new one. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/single_node/openmeteo_weather/openmeteo_weather.yaml b/tests/tasks/uipath-maestro-flow/single_node/openmeteo_weather/openmeteo_weather.yaml index a911f49ab7..bada110238 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/openmeteo_weather/openmeteo_weather.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/openmeteo_weather/openmeteo_weather.yaml @@ -32,9 +32,7 @@ initial_prompt: | The task is not complete until `uip maestro flow validate` passes. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build and validate the complete flow end-to-end - in a single pass. Do NOT substitute a mock or a manual + Do NOT substitute a mock or a manual HTTP request for the connector. success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/single_node/outlook_trigger_inbox/outlook_trigger_inbox.yaml b/tests/tasks/uipath-maestro-flow/single_node/outlook_trigger_inbox/outlook_trigger_inbox.yaml index 3f27f03f6c..c9c961551f 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/outlook_trigger_inbox/outlook_trigger_inbox.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/outlook_trigger_inbox/outlook_trigger_inbox.yaml @@ -26,8 +26,6 @@ initial_prompt: | from that Script node's output. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single pass. Do not reuse any folder ID you may have seen before. diff --git a/tests/tasks/uipath-maestro-flow/single_node/outlook_waitfor_email/outlook_waitfor_email.yaml b/tests/tasks/uipath-maestro-flow/single_node/outlook_waitfor_email/outlook_waitfor_email.yaml index 52fc561599..dd43b87217 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/outlook_waitfor_email/outlook_waitfor_email.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/outlook_waitfor_email/outlook_waitfor_email.yaml @@ -41,9 +41,6 @@ initial_prompt: | Validate the flow. Do not debug the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single - pass. success_criteria: # ── Structural: flow builds and validates ────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/single_node/rpa/rpa.yaml b/tests/tasks/uipath-maestro-flow/single_node/rpa/rpa.yaml index 54904b6a08..52ddce3fe3 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/rpa/rpa.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/rpa/rpa.yaml @@ -16,7 +16,6 @@ initial_prompt: | return it as an output. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # ── Flow file validity ───────────────────────────────────────────────── diff --git a/tests/tasks/uipath-maestro-flow/single_node/subflow/subflow.yaml b/tests/tasks/uipath-maestro-flow/single_node/subflow/subflow.yaml index 7c359f77f6..50457c204a 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/subflow/subflow.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/subflow/subflow.yaml @@ -18,7 +18,6 @@ initial_prompt: | Return the reversed string as an output. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # -- Flow file validity -- diff --git a/tests/tasks/uipath-maestro-flow/single_node/switch/switch.yaml b/tests/tasks/uipath-maestro-flow/single_node/switch/switch.yaml index b35e47ae0b..25a23cd219 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/switch/switch.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/switch/switch.yaml @@ -20,7 +20,6 @@ initial_prompt: | The flow should branch into separate cases for each quarter value. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # -- Flow file validity -- diff --git a/tests/tasks/uipath-maestro-flow/single_node/terminate/terminate.yaml b/tests/tasks/uipath-maestro-flow/single_node/terminate/terminate.yaml index 33a2a66d03..edee904217 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/terminate/terminate.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/terminate/terminate.yaml @@ -21,7 +21,6 @@ initial_prompt: | Both branches should start at the same time from the trigger node. Validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between planning and implementation. Build the complete flow end-to-end in a single pass. success_criteria: # -- Flow file validity -- diff --git a/tests/tasks/uipath-maestro-flow/single_node/transform_filter/transform_filter.yaml b/tests/tasks/uipath-maestro-flow/single_node/transform_filter/transform_filter.yaml index d1cdf46d46..f6ff66e8c6 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/transform_filter/transform_filter.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/transform_filter/transform_filter.yaml @@ -38,9 +38,7 @@ initial_prompt: | `core.action.transform.filter` — do not hardcode a guessed value. Do NOT run flow debug — just validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single - pass. Before starting, follow the documented workflow steps exactly — in particular the transform plugin's filter + Before starting, follow the documented workflow steps exactly — in particular the transform plugin's filter node shape (`operations[].type == "filter"` with `config.filters`). success_criteria: diff --git a/tests/tasks/uipath-maestro-flow/single_node/transform_group_by/transform_group_by.yaml b/tests/tasks/uipath-maestro-flow/single_node/transform_group_by/transform_group_by.yaml index 1e9feaa5dd..1350ab1932 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/transform_group_by/transform_group_by.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/transform_group_by/transform_group_by.yaml @@ -39,8 +39,6 @@ initial_prompt: | `core.action.transform.group-by` — do NOT hardcode a guessed version. Validate the flow. Do NOT run flow debug — just validate. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single pass. Before starting, follow the documented workflow steps exactly — in particular, the transform plugin's Group By JSON shape and the collection-input contract. diff --git a/tests/tasks/uipath-maestro-flow/single_node/transform_map/transform_map.yaml b/tests/tasks/uipath-maestro-flow/single_node/transform_map/transform_map.yaml index 5ee7f6e3e0..3acdb33bd8 100644 --- a/tests/tasks/uipath-maestro-flow/single_node/transform_map/transform_map.yaml +++ b/tests/tasks/uipath-maestro-flow/single_node/transform_map/transform_map.yaml @@ -37,8 +37,6 @@ initial_prompt: | skill's workflow teaches how to read the registry). Do NOT run flow debug — just validate the flow. - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a single pass. Before starting, follow the documented workflow steps exactly — in particular, the transform plugin's Map JSON shape and the `collection` path contract. diff --git a/tests/tasks/uipath-maestro-flow/smoke/merge_parallel_sync.yaml b/tests/tasks/uipath-maestro-flow/smoke/merge_parallel_sync.yaml index 6e8af776a5..be182fbb1c 100644 --- a/tests/tasks/uipath-maestro-flow/smoke/merge_parallel_sync.yaml +++ b/tests/tasks/uipath-maestro-flow/smoke/merge_parallel_sync.yaml @@ -60,10 +60,6 @@ initial_prompt: | layout/variables, then validate with `uip maestro flow validate` — do NOT run `uip maestro flow debug` (the parallel join only synchronizes in a live engine). - - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. - success_criteria: # ── Flow file validity ───────────────────────────────────────────────── - type: run_command diff --git a/tests/tasks/uipath-maestro-flow/smoke/scheduled_trigger.yaml b/tests/tasks/uipath-maestro-flow/smoke/scheduled_trigger.yaml index 665a01ecb7..ad17a89179 100644 --- a/tests/tasks/uipath-maestro-flow/smoke/scheduled_trigger.yaml +++ b/tests/tasks/uipath-maestro-flow/smoke/scheduled_trigger.yaml @@ -48,10 +48,6 @@ initial_prompt: | - The `uip` CLI is already available; use `--output json` on uip commands. - Validate locally with `uip maestro flow validate` — do NOT run `uip maestro flow debug` (the schedule fires only in a live engine). - - Do NOT ask for approval, confirmation, or feedback. Do NOT pause between - planning and implementation. Build the complete flow end-to-end in a - single pass. - success_criteria: # ── Flow file validity ───────────────────────────────────────────────── - type: run_command diff --git a/tests/templates/test-task-template.yaml b/tests/templates/test-task-template.yaml index b28c427bf6..37dea1dd49 100644 --- a/tests/templates/test-task-template.yaml +++ b/tests/templates/test-task-template.yaml @@ -34,9 +34,13 @@ initial_prompt: | Goal-oriented prompt — describe what to build, not how. Let the skill teach the agent the workflow. + Autonomy language depends on the experiment the task runs under. Under + experiments/flow.yaml the system prompt already states the run is headless, + so do not repeat it — a task that does drifts from the wording every other + task uses. Under experiments/default.yaml or nightly.yaml it does not, so an + e2e task that must not stall on a consent gate still needs to say so itself. + For e2e tests, append: - Do NOT ask for approval, confirmation, or feedback. - Do NOT pause between planning and implementation. Before starting, load the skill and follow its workflow. success_criteria: