From c94b89bb0e3cccc0063eb5906270c446e6a22a68 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:20:53 -0700 Subject: [PATCH 01/35] test(bpmn): live-tier Flow ports, stacked on the structural PR Re-adds the live-tier ports removed from test/bpmn-port-connectors so the work continues here: Jira get/create/lifecycle/search, escalation jira/slack/ orchestrator_paths, bellevue_weather, slack_channel_description, slack_weather_pipeline, billing_invoice_lookup, billing_discrepancy_detector, generic_dynamic_node, jdbc_databricks_query, slack_http_fallback, testmanager_crud_grounded, Data Fabric smoke_error, webhook_waitfor_parallel. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../check_billing_discrepancy_detector.py | 780 +++++++++++++++++ .../_shared/check_billing_invoice_lookup.py | 784 ++++++++++++++++++ .../_shared/check_channel_description.py | 279 +++++++ .../_shared/check_databricks_query.py | 174 ++++ .../_shared/check_df_smoke_error.py | 164 ++++ .../_shared/check_escalation_jira_ticket.py | 453 ++++++++++ .../check_escalation_orchestrator_paths.py | 600 ++++++++++++++ .../_shared/check_escalation_slack_alert.py | 459 ++++++++++ .../_shared/check_generic_dynamic_node.py | 340 ++++++++ .../_shared/check_jira_create_issue.py | 363 ++++++++ .../_shared/check_jira_get_issue.py | 285 +++++++ .../_shared/check_jira_lifecycle.py | 351 ++++++++ .../_shared/check_jira_search_triage.py | 283 +++++++ .../_shared/check_slack_http_fallback.py | 322 +++++++ .../_shared/check_slack_weather_pipeline.py | 346 ++++++++ .../check_testmanager_crud_grounded.py | 170 ++++ .../_shared/check_weather_bpmn.py | 273 ++++++ .../_shared/check_webhook_waitfor_parallel.py | 225 +++++ .../smoke_error/smoke_error.yaml | 100 +++ .../generic_dynamic_node.yaml | 121 +++ .../jdbc_databricks_query.yaml | 97 +++ .../slack_http_fallback.yaml | 98 +++ .../testmanager_crud_grounded/_setup/seed.py | 16 + .../testmanager_crud_grounded.yaml | 107 +++ .../webhook_waitfor_parallel.yaml | 117 +++ .../escalation_jira_ticket/_setup/jira_is.py | 144 ++++ .../e2e/escalation_jira_ticket/_setup/seed.py | 48 ++ .../_setup/teardown_jira.py | 30 + .../escalation_jira_ticket.yaml | 142 ++++ .../escalation_jira_ticket/test_jira_is.py | 71 ++ .../_setup/seed.py | 116 +++ .../escalation_orchestrator_paths.yaml | 160 ++++ .../e2e/escalation_slack_alert/_setup/seed.py | 54 ++ .../escalation_slack_alert.yaml | 125 +++ .../e2e/jira_create_issue/_setup/jira_is.py | 113 +++ .../e2e/jira_create_issue/_setup/seed_jira.py | 24 + .../jira_create_issue/_setup/teardown_jira.py | 22 + .../jira_create_issue/jira_create_issue.yaml | 104 +++ .../e2e/jira_get_issue/_setup/jira_is.py | 66 ++ .../e2e/jira_get_issue/_setup/seed_jira.py | 25 + .../jira_get_issue/_setup/teardown_jira.py | 22 + .../e2e/jira_get_issue/jira_get_issue.yaml | 102 +++ .../e2e/jira_lifecycle/_setup/jira_is.py | 67 ++ .../e2e/jira_lifecycle/_setup/seed_jira.py | 39 + .../jira_lifecycle/_setup/teardown_jira.py | 22 + .../e2e/jira_lifecycle/jira_lifecycle.yaml | 131 +++ .../e2e/jira_search_triage/_setup/jira_is.py | 65 ++ .../jira_search_triage/_setup/seed_jira.py | 26 + .../_setup/teardown_jira.py | 22 + .../jira_search_triage.yaml | 107 +++ .../bellevue_weather/bellevue_weather.yaml | 91 ++ .../billing_discrepancy_detector.yaml | 127 +++ .../billing_invoice_lookup.yaml | 122 +++ .../slack_channel_description.yaml | 87 ++ .../slack_weather_pipeline.yaml | 88 ++ 55 files changed, 9669 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/_setup/seed.py create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/jira_is.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/seed.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/teardown_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/test_jira_is.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/_setup/seed.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/_setup/seed.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/seed_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/seed_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py create mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py new file mode 100644 index 0000000000..a0c3b5e2e0 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py @@ -0,0 +1,780 @@ +#!/usr/bin/env python3 +"""BillingDiscrepancyDetector (BPMN): detector / bindings / advisory checks. + +Ported from Flow `multi_node/billing_discrepancy_detector/`: same scenario +(two independent Data Service reads -- BillingDisputeERP and +BillingDisputeCRM -- fanned out from the trigger/start and rejoined at a +merge/join before an overcharge computation), same three graded scripts +(`check_billing_discrepancy_detector.py`, `_shared/check_bindings_no_stubs.py`, +`_shared/advisory_billing_discrepancy_detector.py`), collapsed here into ONE +module with a `detector` / `bindings` / `advisory` dispatch (per +PORTING-BRIEF: "no new shared utility module" for this batch) so the three +YAML criteria share one script, invoked as: + + check_billing_discrepancy_detector.py detector (live; weight 5.0) + check_billing_discrepancy_detector.py bindings (advisory; weight 1.0) + check_billing_discrepancy_detector.py advisory (advisory; weight 1.0) + +Translated from a JSON node/edge walk to an XML walk over the registry-driven +`Intsvc.ActivityExecution` connector shell (registry-workflow.md §3-4) and the +BPMN live-debug surface (`_shared/bpmn_live.py`, LIVE-ADDENDUM's canonical +pattern: ephemeral solution import, `bpmn debug`, `debug-instance +variables-all`/`incidents`). + +Assertion map (Flow -> BPMN): + F check_billing_discrepancy_detector.py:42 assert_flow_has_any_node_type(ENTITY_QUERY_HINTS) + -> detector(): query_entity_nodes() non-empty + F check_billing_discrepancy_detector.py:43 assert_flow_has_node_type(["core.logic.merge"]) + -> detector(): find_join_gateways() non-empty + F check_billing_discrepancy_detector.py:46 run_debug(inputs=INPUTS, timeout=240) + -> detector(): LIVE-ADDENDUM canonical pattern (ephemeral + solution init + import + sha256 pin + bpmn_live.run_debug + + debug-instance variables-all/incidents) + F check_billing_discrepancy_detector.py:48-51 assert_output_value(payload, 1610/1/"MCS-2026-04872"/"Enterprise") + -> detector(): assert_output_value() over collect_output_leaves() + (root scope Globals + every element's Outputs, per + LIVE-ADDENDUM: a root public output has read back null) + F advisory_billing_discrepancy_detector.py:71-89 two entity-read nodes, one per entity (ERP/CRM) + -> advisory(): query_entity_nodes() + entity_of() + F advisory_billing_discrepancy_detector.py:91-101 exactly one merge, bpmn:ParallelGateway join + -> advisory(): find_join_gateways() (exactly one gateway with + >=2 incoming flows) + F advisory_billing_discrepancy_detector.py:102-107 merge fed by >=2 distinct sources, continues downstream + -> advisory(): distinct incoming sourceRefs + outgoing flow check + F advisory_billing_discrepancy_detector.py:108-110 fork exists (some node has >=2 outgoing edges) + -> advisory(): has_fork() (generic -- not pinned to a gateway + type, matching Flow's own generic node-degree check) + F advisory_billing_discrepancy_detector.py:112-144 the two queries are MUTUALLY UNREACHABLE, both reachable + from the (single) trigger + -> advisory(): graph.reachable()/reaches_blocked() -- an F use + of `_shared/graph.py` `reaches`, per PORTING-BRIEF, because + Flow asserts this unreachability itself + F advisory_billing_discrepancy_detector.py:146-151 both filters computed, each from its own input + (ERP<-invoiceNumber, CRM<-accountNumber) + -> advisory(): reads_field() over the variable-derivation graph + F advisory_billing_discrepancy_detector.py:153-160 no answer literal (1610/2590/4200/Enterprise) + -> advisory(): carries_literal() (scoped to bpmn:process, + excluding bpmndi diagram coordinates) + F advisory_billing_discrepancy_detector.py:162-173 declared contract: in-globals present, out-globals present + with the contract's type + -> advisory(): declared_id() presence + declared type checks + F advisory_billing_discrepancy_detector.py:175-195 each output sourced from its own side (tier<-CRM, + matchedInvoiceNumber<-ERP, both numbers<-ERP) + -> advisory(): sourced_from() over the variable-derivation graph + F advisory_billing_discrepancy_detector.py:197-198 each read resolves to the tenant it is pointed at + (connection/folder PAIR of distinct real uuids) + -> advisory(): assert_connection_resolves() over the declared + block + F check_bindings_no_stubs.py:91-142 bindings*.json Connection resources are non-stub, and a + connector node without one fails + -> bindings(): packed bindings_v2.json Connection resources + (mirrors e2e/customer_escalation_triage/ + check_customer_escalation_package.py) + I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) + -> bpmn_check.find_bpmn_file()/parse_bpmn()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported bytes -- `bpmn debug` + runs against an imported project, unlike `flow debug`, which runs directly against the discovered project + -> LIVE-ADDENDUM canonical pattern (mirrors + e2e/jira_get_issue/_shared/check_jira_get_issue.py) + I `uip maestro bpmn pack` + zip read of bindings_v2.json (BPMN has no bare generated bindings.json outside a + pack/refresh -- registry-workflow.md §4) + -> bindings(): mirrors check_customer_escalation_package.py + T curated|generic entity-CRUD objectName classification (BATCH1-ADDENDUM) + -> query_entity_nodes() + T transitive variable derivation through BPMN.Variables copy tasks (Grading contract's allowed T list) + -> build_variable_graph()/trace() (var-id graph, name-matched, + not node-id/hop-bounded like Flow's $vars..output) + DROPPED require_no_private_connector_values / require_sequence_integrity / require_di_for_visible_elements + (not in Flow; the `bpmn validate` criterion covers structure) + +No native Data Fabric shape exists on the BPMN side of this port +(BATCH1-ADDENDUM: "Build this with the UiPath Data Service Integration Service +connector... for every entity operation" -- BPMN has no `core.datafabric.*` +analogue), so `entity_reads`'s NATIVE_READ branch from Flow's +advisory_flow_utils has no BPMN carrier and is not ported. +""" + +from __future__ import annotations + +import json +import os +import re +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +import zipfile +from collections import Counter, defaultdict +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared import bpmn_live # noqa: E402 +from _shared import graph # noqa: E402 +from _shared.bpmn_check import ( # noqa: E402 + NS, + attr, + elements, + fail, + find_bpmn_file, + parse_bpmn, + resolve_project, +) +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +NAME_HINT = "BillingDiscrepancyDetector" +CONNECTOR_KEY = "uipath-uipath-dataservice" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +# Query Entity Records curated/preview spellings (BATCH1-ADDENDUM). +QUERY_OBJECT_NAMES = {"queryentityrecordscurated", "queryentityrecords_v3"} +ERP = "BillingDisputeERP" +CRM = "BillingDisputeCRM" +FORBIDDEN_LITERALS = ["1610", "2590", "4200", "Enterprise"] +OUT_CONTRACT = { + "totalOvercharge": "number", + "discrepancyCount": "number", + "matchedInvoiceNumber": "string", + "accountTier": "string", +} +IN_CONTRACT = [ + "invoiceNumber", + "accountNumber", + "disputedLineNumber", + "disputedUnitPrice", + "disputedQuantity", +] + +UUID_RE = re.compile( + r"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$" +) +STUB_UUID_RE = re.compile(r"^0{8}-0{4}-0{4}-0{4}-") +VAR_REF = re.compile(r"vars\.([A-Za-z_][\w]*)") + + +def is_real_uuid(value) -> bool: + rendered = str(value or "").strip() + return bool(UUID_RE.fullmatch(rendered)) and not STUB_UUID_RE.match(rendered) + + +# ── connector-activity XML helpers (BATCH1-ADDENDUM shapes) ────────────────── + + +def has_type(el: ET.Element, token: str) -> bool: + return token in ET.tostring(el, encoding="unicode") + + +def activity_root(task: ET.Element) -> ET.Element | None: + return task.find(".//uipath:activity", NS) + + +def all_inputs(task: ET.Element) -> list[ET.Element]: + root = activity_root(task) + if root is None: + return [] + return root.findall(".//uipath:input", NS) + + +def input_val(inp: ET.Element) -> str: + return inp.attrib.get("value") or (inp.text or "") + + +def context_value(task: ET.Element, name: str) -> str: + for inp in all_inputs(task): + if inp.attrib.get("name") == name: + return input_val(inp) + return "" + + +def query_entity_nodes(root: ET.Element) -> list[ET.Element]: + """Every sendTask carrying a Query Entity Records op, curated OR generic.""" + nodes = [] + for task in elements(root, "sendTask"): + if not has_type(task, ACTIVITY_TYPE): + continue + if connector_context(task).get("connectorKey") != CONNECTOR_KEY: + continue + object_name_l = context_value(task, "objectName").strip().lower() + operation_l = context_value(task, "operation").strip().lower() + method_u = context_value(task, "method").strip().upper() + is_curated = object_name_l in QUERY_OBJECT_NAMES + is_generic = object_name_l in {ERP.lower(), CRM.lower()} and ( + operation_l in ("list", "retrieve") or method_u == "GET" + ) + if is_curated or is_generic: + nodes.append(task) + return nodes + + +def entity_of(task: ET.Element) -> str | None: + """The entity a query node addresses (BATCH1-ADDENDUM: any input value or + the context path -- the skill does not pin where entityName lands).""" + path_value = context_value(task, "path").lower() + input_values = {input_val(inp).strip().lower() for inp in all_inputs(task)} + matches = [ + entity + for entity in (ERP, CRM) + if entity.lower() in input_values or entity.lower() in path_value + ] + return matches[0] if len(matches) == 1 else None + + +def find_join_gateways(root: ET.Element) -> list[tuple[ET.Element, list[ET.Element]]]: + """(gateway, incoming flows) for every bpmn:parallelGateway with >=2 incoming.""" + flows = elements(root, "sequenceFlow") + + def in_flows(node_id: str) -> list[ET.Element]: + return [f for f in flows if attr(f, "targetRef") == node_id] + + return [ + (gw, in_flows(attr(gw, "id"))) + for gw in elements(root, "parallelGateway") + if len(in_flows(attr(gw, "id"))) >= 2 + ] + + +def has_fork(root: ET.Element) -> bool: + """Any node with >=2 outgoing sequence flows (Flow's own generic check -- + not pinned to a gateway type).""" + counts = Counter(attr(f, "sourceRef") for f in elements(root, "sequenceFlow")) + return any(count >= 2 for count in counts.values()) + + +def reaches_blocked(root: ET.Element, source: str, target: str, blocked: str) -> bool: + return target in graph.reachable(root, source, blocked={blocked}) + + +# ── variable-derivation graph (var id -> name, var id -> derives-from ids) ─── + + +def build_variable_graph(root: ET.Element): + """(name_by_id, derives_from, var_written_by). + + `name_by_id` comes from every declared `` entry + (input/inputOutput/output). `derives_from`/`var_written_by` come from + every `` anywhere in the document + (a connector activity's own output, or a BPMN.Variables mapping's copy) -- + the T translation of Flow's node-id `$vars..output` dependency graph + onto BPMN's var-id addressing. + """ + name_by_id: dict[str, str] = {} + for var in root.findall(".//uipath:variables/*", NS): + var_id = var.attrib.get("id") + name = var.attrib.get("name") + if var_id and name: + name_by_id[var_id] = name.strip().lower() + + owner_tags = { + f"{{{graph.BPMN_NS}}}{tag}" + for tag in ( + "startEvent", + "endEvent", + "task", + "sendTask", + "receiveTask", + "serviceTask", + "scriptTask", + "userTask", + "businessRuleTask", + "subProcess", + "callActivity", + ) + } + parents = graph.parent_map(root) + + def owner_of(element: ET.Element) -> str | None: + node = element + while node in parents: + node = parents[node] + if node.tag in owner_tags and node.attrib.get("id"): + return node.attrib["id"] + return None + + derives_from: dict[str, set[str]] = defaultdict(set) + var_written_by: dict[str, set[str]] = defaultdict(set) + for out in root.iter(f"{{{NS['uipath']}}}output"): + var_id = out.attrib.get("var") + if not var_id: + continue + owner = owner_of(out) + if owner: + var_written_by[var_id].add(owner) + derives_from[var_id].update(VAR_REF.findall(out.attrib.get("source") or "")) + return name_by_id, derives_from, var_written_by + + +def trace(start_ids, derives_from, predicate) -> bool: + seen: set[str] = set() + stack = list(start_ids) + while stack: + current = stack.pop() + if current in seen: + continue + seen.add(current) + if predicate(current): + return True + stack.extend(derives_from.get(current, ())) + return False + + +def reads_field(task: ET.Element, field_lower: str, name_by_id, derives_from) -> bool: + refs: set[str] = set() + for inp in all_inputs(task): + refs.update(VAR_REF.findall(input_val(inp))) + return trace(refs, derives_from, lambda v: name_by_id.get(v) == field_lower) + + +def declared_id(root: ET.Element, tag: str, name_lower: str) -> str | None: + for var in root.findall(f".//uipath:variables/uipath:{tag}", NS): + if (var.attrib.get("name") or "").strip().lower() == name_lower: + return var.attrib.get("id") + return None + + +def declared_type(root: ET.Element, tag: str, var_id: str) -> str | None: + for var in root.findall(f".//uipath:variables/uipath:{tag}", NS): + if var.attrib.get("id") == var_id: + return var.attrib.get("type") + return None + + +def sourced_from(root, output_name, task_id, derives_from, var_written_by): + out_id = declared_id(root, "output", output_name.lower()) + if not out_id: + return False, None + return trace({out_id}, derives_from, lambda v: task_id in var_written_by.get(v, set())), out_id + + +def carries_literal(process_el: ET.Element, forbidden: str) -> bool: + """Search only the process subtree (excludes bpmndi diagram coordinates, + which can coincidentally collide with a small forbidden number).""" + token = re.compile(rf"(? dict[str, ET.Element]: + return { + b.attrib.get("id"): b + for b in root.findall(".//uipath:bindings/uipath:binding", NS) + if b.attrib.get("id") + } + + +def resolve_binding_ref(bindings_by_id: dict[str, ET.Element], ref: str) -> ET.Element | None: + match = re.fullmatch(r"=bindings\.([\w-]+)", (ref or "").strip()) + return bindings_by_id.get(match.group(1)) if match else None + + +def assert_connection_resolves(task: ET.Element, label: str, bindings_by_id) -> str: + """F advisory_billing_discrepancy_detector.py:197-198: connection/folder + PAIR of distinct real uuids -- the connector-shape half of + `assert_read_resolves` (the native half has no BPMN carrier, see module + docstring).""" + connection_ref = context_value(task, "connection") + conn_binding = resolve_binding_ref(bindings_by_id, connection_ref) + if conn_binding is None: + fail( + f"the {label} query's context 'connection' is {connection_ref!r}, which does not " + "resolve to a declared " + ) + conn_value = conn_binding.attrib.get("default") or conn_binding.attrib.get("resourceKey") + if not is_real_uuid(conn_value): + fail( + f"the {label} query's connection binding {conn_binding.attrib.get('id')!r} " + f"default/resourceKey is {conn_value!r}, not a real connection id" + ) + + folder_ref = context_value(task, "folderKey") + folder_binding = resolve_binding_ref(bindings_by_id, folder_ref) + if folder_binding is None: + fail( + f"the {label} query's context 'folderKey' is {folder_ref!r}, which does not resolve " + "to a declared (a folder-scoped connector activity needs a paired " + "folder binding -- registry-workflow.md §4)" + ) + folder_value = folder_binding.attrib.get("default") or folder_binding.attrib.get("resourceKey") + if not is_real_uuid(folder_value): + fail( + f"the {label} query's folder binding {folder_binding.attrib.get('id')!r} " + f"default/resourceKey is {folder_value!r}, not a real folder id" + ) + if conn_value == folder_value: + fail( + f"the {label} query's connection and folder bindings both resolve to the SAME uuid " + f"({conn_value}) -- the folder binding needs the connection's FOLDER key, not its id" + ) + return f"connection={conn_value[:8]}... folder={folder_value[:8]}..." + + +# ── advisory: STRUCTURAL shape (weight 1.0, pass_threshold 0.0) ───────────── + + +def advisory() -> None: + path, root = parse_bpmn(NAME_HINT) + process = root.find("bpmn:process", NS) + if process is None: + fail(f"{path}: no bpmn:process element found") + + # 1. two entity-read nodes, one per entity. + nodes = query_entity_nodes(root) + if len(nodes) != 2: + fail( + f"expected exactly TWO Data Service query-entity-records connector nodes " + f"(ERP + CRM), found {len(nodes)}" + ) + by_entity: dict[str, ET.Element] = {} + for task in nodes: + entity = entity_of(task) + if entity is None: + fail( + f"query node {attr(task, 'id')!r} does not clearly address exactly one of " + f"{ERP!r}/{CRM!r} in any input value or context path" + ) + if entity in by_entity: + fail(f"both query nodes address {entity!r}; the scenario needs one {ERP} and one {CRM}") + by_entity[entity] = task + missing = [e for e in (ERP, CRM) if e not in by_entity] + if missing: + fail(f"no query node addresses {missing}; entities queried: {sorted(by_entity)}") + erp, crm = by_entity[ERP], by_entity[CRM] + erp_id, crm_id = attr(erp, "id"), attr(crm, "id") + + # 2. exactly one join, a real join, continuing downstream; a fork exists. + joins = find_join_gateways(root) + if len(joins) != 1: + fail( + f"expected exactly one bpmn:parallelGateway acting as a join (>=2 incoming flows), " + f"found {len(joins)}" + ) + join_gw, incoming = joins[0] + join_id = attr(join_gw, "id") + sources = sorted({attr(f, "sourceRef") for f in incoming}) + if len(sources) < 2: + fail(f"join gateway {join_id!r} is fed by {len(sources)} distinct source(s) ({sources})") + if not [f for f in elements(root, "sequenceFlow") if attr(f, "sourceRef") == join_id]: + fail(f"join gateway {join_id!r} has no outgoing sequence flow; the joined path must continue") + if not has_fork(root): + fail("no fork found: no node has >=2 outgoing sequence flows, so nothing fans out before the join") + + # 3. the two queries are on MUTUALLY UNREACHABLE branches, both reachable + # from the (single) start event. + if reaches_blocked(root, erp_id, crm_id, join_id) or reaches_blocked(root, crm_id, erp_id, join_id): + first, second = ( + (erp_id, crm_id) if reaches_blocked(root, erp_id, crm_id, join_id) else (crm_id, erp_id) + ) + fail( + f"{second!r} is downstream of {first!r} -- the two lookups are CHAINED with a join " + "bolted on, not a fan-out from the start event" + ) + starts = elements(root, "startEvent") + if len(starts) != 1: + fail(f"a process has exactly one root start event, found {len(starts)}") + start_id = attr(starts[0], "id") + reachable_from_start = graph.reachable(root, start_id) + for node_id in (erp_id, crm_id): + if node_id not in reachable_from_start: + fail(f"query node {node_id!r} is not reachable from the start event {start_id!r}") + + # 4. both filters computed, each from its own input. + name_by_id, derives_from, var_written_by = build_variable_graph(root) + for label, task, field in ((ERP, erp, "invoicenumber"), (CRM, crm, "accountnumber")): + if not reads_field(task, field, name_by_id, derives_from): + fail( + f"the {label} query node does not reference the {field} input (directly, or " + "transitively through a BPMN.Variables copy task) in any of its inputs" + ) + + # 5. none of the answers is written in. + for bad in FORBIDDEN_LITERALS: + if carries_literal(process, bad): + fail( + f"the process carries the literal {bad!r}. Every one of the answers " + f"({', '.join(FORBIDDEN_LITERALS)}) has to come from the tenant" + ) + + # 6. the declared contract, and where each output comes from. + for name in IN_CONTRACT: + if declared_id(root, "input", name.lower()) is None: + fail(f"the process declares no public input named {name!r}") + for name, want in OUT_CONTRACT.items(): + out_id = declared_id(root, "output", name.lower()) + if out_id is None: + fail(f"the process declares no public output named {name!r}") + got = declared_type(root, "output", out_id) + if got != want: + fail(f"output {name!r} is declared type {got!r}; the contract asks for {want!r}") + + ok, out_id = sourced_from(root, "accountTier", crm_id, derives_from, var_written_by) + if not ok: + fail(f"output 'accountTier' (var {out_id!r}) does not derive from {crm_id!r} -- the tier comes from {CRM}") + ok, out_id = sourced_from(root, "matchedInvoiceNumber", erp_id, derives_from, var_written_by) + if not ok: + fail( + f"output 'matchedInvoiceNumber' (var {out_id!r}) does not derive from {erp_id!r} -- " + f"the matched invoice comes from {ERP}" + ) + for name in ("totalOvercharge", "discrepancyCount"): + ok, out_id = sourced_from(root, name, erp_id, derives_from, var_written_by) + if not ok: + fail( + f"output {name!r} (var {out_id!r}) does not derive from {erp_id!r} -- the contracted " + f"amount it is computed from lives in the {ERP} rows" + ) + + # 7. each read resolves to the tenant it is pointed at. + bindings_by_id = declared_bindings(root) + resolutions = [ + assert_connection_resolves(task, label, bindings_by_id) + for label, task in ((ERP, erp), (CRM, crm)) + ] + + print( + f"OK: {path} -- {ERP}={erp_id!r} and {CRM}={crm_id!r} on mutually unreachable branches from " + f"{start_id!r}, converging on parallelGateway join {join_id!r} ({len(sources)} sources, " + "continues downstream); both filters computed from their own inputs; no answer literals; " + f"outputs {sorted(OUT_CONTRACT)} each sourced from its own side; {'; '.join(resolutions)}" + ) + + +# ── bindings: no-stub advisory (weight 1.0, pass_threshold 0.0) ───────────── + + +def bindings() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + root = ET.parse(bpmn_path).getroot() + needs_connection = any(connector_context(node).get("connectorKey") for node in root.iter()) + + project_dir = resolve_project(os.path.basename(bpmn_path)) + with tempfile.TemporaryDirectory(prefix="billing-pack-") as out_dir: + packed = subprocess.run( + ["uip", "maestro", "bpmn", "pack", str(project_dir), out_dir, "--output", "json"], + capture_output=True, + text=True, + timeout=90, + ) + if packed.returncode != 0: + fail(f"uip maestro bpmn pack failed (exit {packed.returncode}): {packed.stdout}\n{packed.stderr}") + try: + payload = bpmn_live.parse_json_output(packed.stdout, "pack") + except CheckFailure as error: + fail(str(error)) + if not isinstance(payload, dict) or str(payload.get("Result", "")).casefold() != "success": + fail(f"pack JSON did not report Success: {payload}") + + packages = list(Path(out_dir).glob("*.nupkg")) + if len(packages) != 1: + fail(f"expected exactly one .nupkg, found: {[p.name for p in packages]}") + with zipfile.ZipFile(packages[0]) as archive: + by_basename = {Path(name).name: name for name in archive.namelist()} + if "bindings_v2.json" not in by_basename: + if needs_connection: + fail( + "packed archive has no bindings_v2.json, but the process carries a " + "connector node that needs a connection binding" + ) + print("no bindings_v2.json, and no connector node that needs one") + return + bindings_doc = json.loads(archive.read(by_basename["bindings_v2.json"])) + + resources = bindings_doc.get("resources") + if not isinstance(resources, list): + fail(f"bindings_v2.json has no resources array: {bindings_doc}") + connections = [r for r in resources if isinstance(r, dict) and r.get("resource") == "Connection"] + if not connections: + if needs_connection: + fail("bindings_v2.json declares no Connection resources, but the process carries a connector node") + print("no Connection resources, and no connector node that needs one") + return + stubbed = [r.get("key") for r in connections if not is_real_uuid(r.get("key"))] + if stubbed: + fail(f"bindings_v2.json Connection keys must be real connection ids, not unresolved stubs: {stubbed}") + print(f"{len(connections)} connection binding(s) in bindings_v2.json, all populated with non-stub values") + + +# ── detector: live run (weight 5.0) ───────────────────────────────────────── + +INPUTS = { + "invoiceNumber": "MCS-2026-04872", + "accountNumber": "ACCT-98201-NE", + "disputedLineNumber": 5, + "disputedUnitPrice": 300, + "disputedQuantity": 14, +} +EXPECTED_OUTPUTS = (1610, 1, "MCS-2026-04872", "Enterprise") + +LIVE_RUN_DIR = Path("billing-discrepancy-detector-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker's `detector` criterion can spend, priced +# the way _shared/test_criterion_budgets.py prices a run_debug(...) call: the +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). The +# surrounding CLI steps (solution init/import, variables-all, incidents) are +# not priced by that guard, so their sum is added by hand here, mirroring +# e2e/jira_get_issue and multi_node/slack_weather_pipeline: +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (600) does not cover this; raised to 1050 in +# billing_discrepancy_detector.yaml (documented deviation, sanctioned by +# LIVE-ADDENDUM: the budget is a property of the CLI surface, not of what is +# graded). + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def collect_output_leaves(variables_data) -> list: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- mirrors Flow's own assert_output_value(), + which flattens the whole debug payload's declared outputs. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return leaves + + +def assert_output_value(leaves: list, expected) -> None: + """F check_billing_discrepancy_detector.py:48-51 (assert_output_value): + numeric exact match; string case-insensitive substring.""" + for value in leaves: + if value == expected: + return + if isinstance(expected, str) and isinstance(value, str) and expected.lower() in value.lower(): + return + raise CheckFailure(f"no output equals expected {expected!r}; outputs: {str(leaves)[:1000]}") + + +def detector() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + root = ET.parse(bpmn_path).getroot() + + if len(query_entity_nodes(root)) < 2: + fail( + "expected at least two Data Service query-entity-records connector nodes " + f"(found {len(query_entity_nodes(root))})" + ) + if not find_join_gateways(root): + fail("no bpmn:parallelGateway acts as a join (>=2 incoming flows)") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "BillingDiscrepancyDetectorLiveEval" + initialized = run_cli(["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in {solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + print(f"debug inputs: {INPUTS}") + debug_data, instance_id = bpmn_live.run_debug(imported_project, INPUTS, LIVE_RUN_DIR / "debug.log") + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure(f"final status was {final_status!r}" + ("; " + "; ".join(detail) if detail else "")) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print(f"OK: bpmn debug completed (FinalStatus={final_status}, no incidents)") + + leaves = collect_output_leaves(variables_data) + for expected in EXPECTED_OUTPUTS: + assert_output_value(leaves, expected) + print("OK: overcharge=1610, count=1, invoice MCS-2026-04872, tier Enterprise") + print("PASS: all BillingDiscrepancyDetector checks passed") + + +_MODES = {"detector": detector, "bindings": bindings, "advisory": advisory} + + +def main() -> None: + mode = sys.argv[1] if len(sys.argv) > 1 else "detector" + handler = _MODES.get(mode) + if handler is None: + sys.exit(f"FAIL: unknown mode {mode!r}; expected one of {sorted(_MODES)}") + handler() + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py new file mode 100644 index 0000000000..13e6b8d792 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py @@ -0,0 +1,784 @@ +#!/usr/bin/env python3 +"""BillingInvoiceLookup (BPMN): four modes, one per Flow checker script. + +Ported from `uipath-maestro-flow/multi_node/billing_invoice_lookup/`: +`check_billing_invoice_lookup.py` (live 3x debug), `check_server_side_filter.py` +(advisory), `_shared/check_bindings_no_stubs.py` (advisory), and +`_shared/advisory_billing_invoice_lookup.py` (advisory structural gate). Kept +as ONE file with a mode-dispatch positional argument (`lookup` / +`server_side_filter` / `bindings` / `advisory`) per BATCH1/PORTING-BRIEF +instructions -- no new `_shared` module. The dispatch shape (module-level +functions + a DISPATCH dict keyed on `sys.argv[1]`) mirrors +`check_outlook_trigger_inbox.py`, and doubles as the entry point +`tests/_shared/test_criterion_budgets.py` prices per subcommand. + +Assertion map (Flow -> BPMN): + F check_billing_invoice_lookup.py:45 assert_flow_has_any_node_type(ENTITY_QUERY_HINTS) + -> query_entity_nodes() non-empty (lookup) + F check_billing_invoice_lookup.py:47-50 read_flow_input_vars()/var = in_vars[0] + -> declared_input_names(root)[0] (lookup) + F check_billing_invoice_lookup.py:52-58 CASES loop, run_debug(inputs=...), assert_output_value x2 + -> CASES loop, bpmn_live.run_debug(project, inputs, log, timeout=180) + + assert_output_value() over output_leaves() (lookup) + I locate/parse .bpmn; ephemeral solution init + `solution projects + import` + sha256 pin; `bpmn debug` returns an instance id, not + inline variables; `debug-instance variables-all`/`incidents` reads + replace the single inline debug payload + -> LIVE-ADDENDUM canonical live pattern + (mirrors e2e/jira_get_issue/_shared/check_jira_get_issue.py) (lookup) + T curated|generic entity-CRUD classification; entity anywhere in + inputs/objectName/path; GETBYID/GET(List) equivalence + (BATCH1-ADDENDUM) -> query_entity_nodes() / entity_ok() (all modes) + + F check_server_side_filter.py:65-80, advisory_flow_utils.has_filter/filters_server_side + -> has_nonempty_filter() (server_side_filter) + DROPPED: the native core.datafabric.read branch -- BPMN has + no native Data Fabric node (BATCH1-ADDENDUM); only the + connector shape applies here. + + F _shared/check_bindings_no_stubs.py:45-47,67-80 is_real_key/UUID/STUB_UUID, + invalid_ids() over bindings[].resourceKey/default + -> is_real_connection_key(), stubbed-key scan (bindings) + DROPPED: the "no bindings file needed" branch for a native + entity read -- this port always uses the connector, so a + Connection resource is always required (BATCH1-ADDENDUM). + I bindings_v2.json is emitted only at PACK time for BPMN + (authoring leaves the connection as a symbolic + `=bindings.` reference, unlike Flow's compiler, which + resolved it straight into the node) -> `uip maestro bpmn + pack` + zip read, modeled on + e2e/customer_escalation_triage/check_customer_escalation_package.py + (PORTING-BRIEF's pointer) + + F advisory_billing_invoice_lookup.py:64-67 exactly ONE entity-read node + -> len(reads) != 1 check (advisory) + F advisory_billing_invoice_lookup.py:71-74 no raw HTTP fallback + -> no Intsvc.HttpExecution node check (advisory) + F advisory_billing_invoice_lookup.py:76-79 entity slot carries ENTITY + -> entity_ok() (advisory) + F advisory_billing_invoice_lookup.py:82 filter COMPUTED from input, not constant + -> filter_reference_ids() + derives_from(ref, {input_id}) (advisory) + T does not pin the filter to one named column (Flow's + `column="invoiceNumber"`): no local Data Service connection + exists to `describe` the entity's real filter field name + (BATCH1-ADDENDUM), so any input at any depth that derives + from the declared input is accepted. + F advisory_billing_invoice_lookup.py:85-96 canonical answer + raw test inputs + never appear as literals -> raw substring scan on the .bpmn + text (I: re-homed from Flow's `_authored_values` walk over + JSON to a whole-file text scan, matching this suite's own + convention in check_jira_get_issue.py's `JIRA_KEY not in raw` + / `issue_key not in raw` checks) (advisory) + F advisory_billing_invoice_lookup.py:98-108 outputs declared with contract's + names AND types -> declared_outputs() (advisory) + F advisory_billing_invoice_lookup.py:110-127 outputs READ FROM the query step + -> derives_from(output_id, query_out_vars) over the + var-source graph built from every `` in the document, T: transitive derivation through + BPMN.Variables copy tasks (BATCH1-ADDENDUM), extended here + to also follow a scriptTask's own `` CDATA + `vars.` reads when its output's `source` starts with + `=result` (a script's return value cannot be traced through + `source=` alone -- see build_var_graph()) (advisory) + F advisory_billing_invoice_lookup.py:130-141 `assert_read_resolves` -- SPLIT + across two artifact stages, since BPMN resolves a + connection binding in two steps Flow's compiler collapsed + into one: + (a) connection/folder DISTINCT, non-blank `default` + values -- fully visible in the raw authored .bpmn (the + registry-workflow.md §4 two-binding shape), so checked + here in `advisory` mode as connection_and_folder_distinct(). + This directly ports the Flow assertion's failure mode + ("the two collapse into one binding... answering 401"). + (b) the connection id is a REAL (non-stub) tenant uuid -- + Flow's compiler baked the resolved connection straight into + the node's own `inputs.detail`; a hand-authored .bpmn keeps + it SYMBOLIC (`=bindings.`) even when correct, so there + is no real id to inspect until pack time. Carried instead + by the `bindings` mode (packed bindings_v2.json), which is + where BPMN actually materializes a resolved connection + value. Note bindings_v2.json's `resources[]` never carries + a folder key at all -- "Only the ConnectionId binding + becomes a bindings_v2.json resource" (registry-workflow.md + §4) -- so the folder side of the distinctness check has no + packed carrier and stays authoring-time-only (a), not + duplicated in (b). + What IS checkable at the authoring stage -- that the + query's `connection` input is a `=bindings.` reference + resolving to a declared `resource="Connection"` binding -- + is asserted here + as `connection_binding_wired()`. + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / uniqueness rules beyond the + one Flow itself asserted -- not in Flow; `bpmn validate` + criterion covers structure. + +No tenant re-read is performed outside the `lookup` mode's own live sequence, +matching Flow's own grader (the query result IS the tenant re-read). +""" + +from __future__ import annotations + +import json +import os +import re +import subprocess +import sys +import tempfile +import xml.etree.ElementTree as ET +import zipfile +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import ( # noqa: E402 + NS, + elements, + fail, + find_bpmn_file, + has_typed_uipath_extension, + parse_bpmn, + resolve_project, +) +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +NAME_HINT = "BillingInvoiceLookup" +CONNECTOR_KEY = "uipath-uipath-dataservice" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +ENTITY = "BillingDisputeERP" +QUERY_CURATED_OBJECTS = {"queryentityrecordscurated", "queryentityrecords_v3"} + +EXPECTED_INVOICE = "MCS-2026-04872" +EXPECTED_LINE_COUNT = 8 +CANONICAL = EXPECTED_INVOICE +# raw input form the caller might send -> human label for failure messages +# (verbatim from Flow's check_billing_invoice_lookup.py CASES). +CASES = [ + ("2026-04872", "missing MCS- prefix"), + ("mcs-2026-04872", "wrong casing"), + (" MCS-2026-04872", "leading whitespace"), +] +# The malformed forms the offline CASES above drive -- carried as a literal +# would let a lookup-table pass every case (advisory_billing_invoice_lookup.py). +RAW_INPUTS = ["2026-04872", "mcs-2026-04872"] + +FILTER_INPUT_NAMES = {"queryexpression", "where", "filter", "filtergroup", "filtervariables"} + +LIVE_RUN_DIR = Path("billing-invoice-lookup-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock, priced the way _shared/test_criterion_budgets.py +# prices a run_debug(...) call inside a static loop: bpmn_live.debug_budget(180) +# x 3 (the module-level CASES list) = 540s, which the guard multiplies +# automatically. The guard prices only run_debug calls; the surrounding CLI +# round trips run once (solution init, solution import) or three times +# (variables-all, incidents per case) and are added by hand here, the same way +# e2e/jira_get_issue/_shared/check_jira_get_issue.py documents its own +# arithmetic: +# 90 (solution init) + 180 (solution import) +# + 3 x (180 debug + 120 variables-all + 120 incidents) +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) +# = 90 + 180 + 1260 + 60 = 1590 +# billing_invoice_lookup.yaml raises this criterion's timeout from Flow's 1200 +# to 1590 to cover it -- the LIVE-ADDENDUM-sanctioned deviation for the extra +# CLI round trips a `bpmn debug` sequence needs per case that +# `flow_check.run_debug`'s single inline call did not. The guard's own +# arithmetic only requires >= 540 + 60 = 600s, comfortably inside 1590. + + +# ── shared XML helpers (copied from check_df_smoke_query_filter.py / +# check_df_contractregistry_crud_filters.py -- no new _shared module) ───── + + +def activity_root(task: ET.Element) -> ET.Element | None: + return task.find(".//uipath:activity", NS) + + +def all_inputs(task: ET.Element) -> list[ET.Element]: + root_el = activity_root(task) + if root_el is None: + return [] + return root_el.findall(".//uipath:input", NS) + + +def input_val(inp: ET.Element) -> str: + return inp.attrib.get("value") or (inp.text or "") + + +def context_value(task: ET.Element, name: str) -> str: + for inp in all_inputs(task): + if inp.attrib.get("name") == name: + return input_val(inp) + return "" + + +def all_node_values(task: ET.Element) -> list[str]: + values: list[str] = [] + for inp in all_inputs(task): + v = inp.attrib.get("value") + if v: + values.append(v) + if inp.text and inp.text.strip(): + values.append(inp.text.strip()) + return values + + +def output_vars(task: ET.Element) -> list[str]: + return [out.attrib["var"] for out in task.findall(".//uipath:output", NS) if out.attrib.get("var")] + + +def entity_ok(task: ET.Element) -> bool: + """BATCH1-ADDENDUM: entity anywhere in path/query/body inputs or context path.""" + if context_value(task, "objectName").strip().lower() == ENTITY.lower(): + return True + values = all_node_values(task) + [context_value(task, "path")] + return any(v and ENTITY in v for v in values) + + +def query_entity_nodes(root: ET.Element) -> list[ET.Element]: + """Curated OR generic entity-CRUD classification (BATCH1-ADDENDUM).""" + nodes = [] + for task in elements(root, "sendTask"): + if not has_typed_uipath_extension(task, "activity", ACTIVITY_TYPE): + continue + if context_value(task, "connectorKey") != CONNECTOR_KEY: + continue + object_name = context_value(task, "objectName").strip().lower() + operation = context_value(task, "operation").strip().lower() + method = context_value(task, "method").strip().upper() + is_curated = object_name in QUERY_CURATED_OBJECTS + is_dynamic_list = object_name == ENTITY.lower() and (operation == "list" or method == "GET") + if is_curated or is_dynamic_list: + nodes.append(task) + return nodes + + +def parse_json_maybe(value: str): + if not isinstance(value, str): + return None + text = value.strip() + if not text: + return None + try: + return json.loads(text) + except (json.JSONDecodeError, TypeError): + return None + + +def find_saved_filter_trees(node, out: list) -> None: + if isinstance(node, dict): + sft = node.get("savedFilterTrees") + if isinstance(sft, dict) and "queryExpression" in sft: + out.append(sft["queryExpression"]) + for value in node.values(): + find_saved_filter_trees(value, out) + elif isinstance(node, list): + for item in node: + find_saved_filter_trees(item, out) + + +def filter_leaves_from_tree(tree, leaves: list) -> None: + if not isinstance(tree, dict): + return + leaves.extend(tree.get("filters") or []) + for child in tree.get("groups") or []: + filter_leaves_from_tree(child, leaves) + + +def structured_filter_leaves(parsed_json) -> list: + trees: list = [] + find_saved_filter_trees(parsed_json, trees) + leaves: list = [] + for tree in trees: + filter_leaves_from_tree(tree, leaves) + return leaves + + +# ── mode: lookup (F: check_billing_invoice_lookup.py) ──────────────────────── + + +def _variable_children(root: ET.Element, tag: str) -> list[ET.Element]: + variables = root.find(".//uipath:variables", NS) + if variables is None: + return [] + wanted = f"{{{NS['uipath']}}}{tag}" + return [child for child in variables if child.tag == wanted] + + +def declared_input_names(root: ET.Element) -> list[str]: + return [c.attrib.get("name") for c in _variable_children(root, "input") if c.attrib.get("name")] + + +def declared_outputs(root: ET.Element) -> dict[str, tuple[str, str]]: + result: dict[str, tuple[str, str]] = {} + for c in _variable_children(root, "output"): + name = c.attrib.get("name") + if name: + result[name] = (c.attrib.get("id", ""), c.attrib.get("type", "")) + return result + + +def input_id_for_name(root: ET.Element, name: str) -> str | None: + for c in _variable_children(root, "input"): + if (c.attrib.get("name") or "").strip().lower() == name.lower(): + return c.attrib.get("id") + return None + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def output_leaves(variables_data: object) -> list: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has read back null even when correctly mapped + (LIVE-ADDENDUM), so the search is not scoped to one declared output + variable -- mirrors check_jira_get_issue.py's collect_output_haystack, but + keeps each leaf's native type so a numeric expectation is not spuriously + matched by a digit embedded in an unrelated string (flow_check.assert_output_value's + own reason for exact numeric equality, not substring, on numerics). + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return leaves + + +def assert_output_value(leaves: list, expected) -> bool: + """F: flow_check.assert_output_value -- exact-equal numerics, case-insensitive substring strings.""" + for v in leaves: + if v == expected: + return True + if isinstance(expected, str) and isinstance(v, str) and expected.lower() in v.lower(): + return True + return False + + +def lookup() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + fail(f"{bpmn_path} is not well-formed XML: {exc}") + + if not query_entity_nodes(root): + fail("bpmn does not reference a Data Service Query Entity Records connector node (Intsvc.ActivityExecution)") + print(f"OK: bpmn references a Query Entity Records node against {ENTITY}") + + input_names = declared_input_names(root) + if not input_names: + fail("process declares no public uipath:input variable for the invoice number") + var_name = input_names[0] + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "BillingInvoiceLookupLiveEval" + try: + initialized = run_cli(["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + for raw_value, label in CASES: + inputs = {var_name: raw_value} + print(f"[{label}] debug inputs: {inputs}") + debug_data, instance_id = bpmn_live.run_debug( + imported_project, + inputs, + LIVE_RUN_DIR / f"debug-{label.replace(' ', '-')}.log", + timeout=180, + ) + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"[{label}] final status was {final_status!r}" + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"[{label}] incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"[{label}] unexpected incidents: {incidents_list}") + + leaves = output_leaves(variables_data) + if not assert_output_value(leaves, EXPECTED_INVOICE): + raise CheckFailure(f"[{label}] no output equals expected {EXPECTED_INVOICE!r}") + if not assert_output_value(leaves, EXPECTED_LINE_COUNT): + raise CheckFailure(f"[{label}] no output equals expected {EXPECTED_LINE_COUNT!r}") + print(f"OK: [{label}] instance {instance_id} -> {EXPECTED_INVOICE}, {EXPECTED_LINE_COUNT} line items") + except CheckFailure as error: + fail(str(error)) + + print(f"OK: all {len(CASES)} malformed forms normalized and queried correctly") + + +# ── mode: server_side_filter (F: check_server_side_filter.py) ──────────────── + + +def has_nonempty_filter(task: ET.Element) -> bool: + for inp in all_inputs(task): + name = (inp.attrib.get("name") or "").strip().lower() + value = input_val(inp) + if name in FILTER_INPUT_NAMES and str(value).strip() not in ("", "{}", "[]"): + return True + parsed = parse_json_maybe(value) + if parsed is not None and structured_filter_leaves(parsed): + return True + return False + + +def server_side_filter() -> None: + path, root = parse_bpmn(NAME_HINT) + nodes = query_entity_nodes(root) + if not nodes: + fail("no entity-read node (Data Service Query Entity Records connector activity)") + unfiltered = [task.attrib.get("id") for task in nodes if not has_nonempty_filter(task)] + if unfiltered: + fail( + f"no server-side filter on {', '.join(unfiltered)} -- entity fetched whole and " + "filtered client-side; breaks silently past the page limit" + ) + print(f"OK: {path} -- query filters server-side") + + +# ── mode: bindings (F: _shared/check_bindings_no_stubs.py) ─────────────────── + +UUID_PATTERN = re.compile( + r"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$" +) +STUB_UUID_PATTERN = re.compile(r"^0{8}-0{4}-0{4}-0{4}-") + + +def is_real_connection_key(value) -> bool: + rendered = str(value or "").strip() + return bool(UUID_PATTERN.fullmatch(rendered)) and not STUB_UUID_PATTERN.match(rendered) + + +def bindings() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + project_dir = resolve_project(os.path.basename(bpmn_path)) + with tempfile.TemporaryDirectory(prefix="bpmn-eval-pack-") as output_dir: + result = subprocess.run( + ["uip", "maestro", "bpmn", "pack", str(project_dir), output_dir, "--output", "json"], + capture_output=True, + text=True, + timeout=150, + ) + if result.returncode != 0: + fail(f"pack exited {result.returncode}\nstdout: {result.stdout}\nstderr: {result.stderr}") + try: + payload = bpmn_live.parse_json_output(result.stdout, "pack") + except CheckFailure as exc: + fail(str(exc)) + if not isinstance(payload, dict) or str(payload.get("Result", "")).casefold() != "success": + fail(f"pack JSON did not report Success: {payload}") + + packages = list(Path(output_dir).glob("*.nupkg")) + if len(packages) != 1: + fail(f"expected exactly one .nupkg, found: {[p.name for p in packages]}") + package = packages[0] + if package.stat().st_size <= 0 or not zipfile.is_zipfile(package): + fail(f"packed file is not a valid non-empty archive: {package.name}") + + with zipfile.ZipFile(package) as archive: + by_basename = {Path(name).name: name for name in archive.namelist()} + if "bindings_v2.json" not in by_basename: + fail("packed archive has no bindings_v2.json") + doc = json.loads(archive.read(by_basename["bindings_v2.json"])) + + resources = [ + r for r in (doc.get("resources") or []) if isinstance(r, dict) and str(r.get("resource") or "").lower() == "connection" + ] + # Flow's own check only requires non-empty + non-stub, never an exact + # count (a connector port always needs >=1, unlike Flow's native-read + # branch, which this port drops -- see module docstring). + if not resources: + fail("packed bindings_v2.json declares no Connection resources") + stubbed = [r.get("key") for r in resources if not is_real_connection_key(r.get("key"))] + if stubbed: + fail(f"packed bindings_v2.json Connection keys must be real connection ids, not unresolved stubs: {stubbed}") + print(f"OK: {len(resources)} connection binding(s) across bindings_v2.json, all non-stub") + + +# ── mode: advisory (F: _shared/advisory_billing_invoice_lookup.py) ─────────── + + +def build_var_graph(root: ET.Element) -> tuple[dict[str, str], dict[str, set]]: + """var_sources: var -> raw `source=` text, from every `` in the document (mapping tasks and connector activity outputs + alike -- T: transitive derivation through BPMN.Variables copy tasks, + BATCH1-ADDENDUM). extra_refs supplements it for a scriptTask output whose + `source` is opaque (`=result.response...`, so it carries no `vars.` token + of its own): the real data dependency lives inside that scriptTask's own + `` CDATA (structural-bpmn.md's Jint contract: `vars.` dot + access), so those refs are attached to that output's var id. + """ + var_sources: dict[str, str] = {} + for out in root.findall(".//uipath:output", NS): + var = out.attrib.get("var") + source = out.attrib.get("source") + if var and source and var not in var_sources: + var_sources[var] = source + + extra_refs: dict[str, set] = {} + for task in elements(root, "scriptTask"): + script_el = task.find("bpmn:script", NS) + script_text = (script_el.text or "") if script_el is not None else "" + script_refs = set(re.findall(r"vars\.([A-Za-z0-9_]+)", script_text)) + if not script_refs: + continue + for out in task.findall(".//uipath:output", NS): + var = out.attrib.get("var") + source = (out.attrib.get("source") or "").strip().lower() + if var and source.startswith("=result"): + extra_refs.setdefault(var, set()).update(script_refs) + return var_sources, extra_refs + + +def derives_from(var_id: str, targets: set, var_sources: dict, extra_refs: dict, hops: int = 3) -> bool: + if var_id in targets: + return True + if hops <= 0: + return False + refs = set(re.findall(r"vars\.([A-Za-z0-9_]+)", var_sources.get(var_id, ""))) + refs |= extra_refs.get(var_id, set()) + return any(derives_from(ref, targets, var_sources, extra_refs, hops - 1) for ref in refs) + + +def filter_reference_ids(task: ET.Element) -> set: + ids: set = set() + for inp in all_inputs(task): + value = input_val(inp) + ids.update(re.findall(r"vars\.([A-Za-z0-9_]+)", value)) + parsed = parse_json_maybe(value) + if parsed is not None: + ids.update(re.findall(r"vars\.([A-Za-z0-9_]+)", json.dumps(parsed))) + return ids + + +def connection_binding_wired(root: ET.Element, task: ET.Element) -> bool: + """The query's `connection` input is `=bindings.` and that id is a + declared `resource="Connection"` binding (registry-workflow.md §4). This is + the authoring-time half of Flow's `assert_read_resolves`; the real-ID half + is checked by the `bindings` mode after packing (see module docstring).""" + connection_ref = context_value(task, "connection") + match = re.match(r"^=bindings\.([A-Za-z0-9_]+)$", connection_ref.strip()) + if not match: + return False + binding_id = match.group(1) + return any( + binding.attrib.get("id") == binding_id + and binding.attrib.get("resource") == "Connection" + and binding.attrib.get("propertyAttribute") == "ConnectionId" + for binding in root.findall(".//uipath:binding", NS) + ) + + +def connection_and_folder_distinct(root: ET.Element, task: ET.Element) -> bool | None: + """F: advisory_billing_invoice_lookup.py's `assert_read_resolves` distinctness + half -- "the two collapse into one binding at FIL emission and the live + dispatch sends the folder key as --connection-id, answering 401". Fully + visible pre-pack: a folder-scoped connector activity declares TWO bindings + sharing one `resourceKey` (registry-workflow.md §4), so this compares their + `default` values directly in the authored XML. Returns None when the + activity is not folder-scoped (no sibling folderKey binding) -- nothing to + compare, not a failure. + """ + connection_ref = context_value(task, "connection") + match = re.match(r"^=bindings\.([A-Za-z0-9_]+)$", connection_ref.strip()) + if not match: + return None + bindings = root.findall(".//uipath:binding", NS) + conn = next((b for b in bindings if b.attrib.get("id") == match.group(1)), None) + if conn is None: + return None + resource_key = conn.attrib.get("resourceKey") + folder = next( + ( + b + for b in bindings + if b.attrib.get("resourceKey") == resource_key and b.attrib.get("propertyAttribute") == "folderKey" + ), + None, + ) + if folder is None: + return None + conn_default = (conn.attrib.get("default") or "").strip() + folder_default = (folder.attrib.get("default") or "").strip() + if not conn_default or not folder_default: + return False + return conn_default != folder_default + + +def advisory() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + raw = Path(bpmn_path).read_text(encoding="utf-8") + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + fail(f"{bpmn_path} is not well-formed XML: {exc}") + + # 1. exactly one entity-read node (F: advisory_billing_invoice_lookup.py:64-67) + reads = query_entity_nodes(root) + if len(reads) != 1: + fail(f"expected exactly ONE Data Service Query Entity Records node, found {len(reads)}") + query = reads[0] + + # no raw HTTP fallback (F: :71-74) + http_nodes = [ + task.attrib.get("id") + for task in (*elements(root, "sendTask"), *elements(root, "serviceTask")) + if has_typed_uipath_extension(task, "activity", "Intsvc.HttpExecution") + ] + if http_nodes: + fail(f"the process calls Data Service over raw HTTP ({http_nodes}); use the connector action") + + # 2. entity slot carries the seeded entity (F: :76-79) + if not entity_ok(query): + fail(f"the read does not address {ENTITY!r} in any path/query/body input or context path") + + # 3. filter COMPUTED from the input, not a constant (F: :82) + input_names = declared_input_names(root) + input_id = input_id_for_name(root, "invoiceNumber") + if input_id is None: + fail(f"process declares public inputs {input_names}; the contract asks for `invoiceNumber`") + var_sources, extra_refs = build_var_graph(root) + filter_refs = filter_reference_ids(query) + if not any(derives_from(ref, {input_id}, var_sources, extra_refs) for ref in filter_refs): + fail( + f"the query's filter does not derive from the declared invoiceNumber input " + f"(vars.{input_id}); refs found on the node: {sorted(filter_refs)}" + ) + + # 4. the ANSWER is nowhere in the file, and neither is a lookup table (F: :85-96) + if CANONICAL in raw: + fail(f"the bpmn contains the literal {CANONICAL!r} -- the invoice number must be COMPUTED, never written in") + for bad in RAW_INPUTS: + if bad in raw: + fail(f"the bpmn contains the test input {bad!r} as a literal -- normalising by matching known inputs generalises to nothing") + + # 5. outputs declared with the contract's names AND types (F: :98-108) + outputs = declared_outputs(root) + for name, want in (("matchedInvoiceNumber", "string"), ("lineItemCount", "number")): + if name not in outputs: + fail(f"process declares public outputs {sorted(outputs)}; the contract asks for {name}") + if outputs[name][1] != want: + fail(f"output {name} is declared {outputs[name][1]!r}; the contract asks for {want}") + + # 6. both outputs are READ FROM the query step (F: :110-127) + query_out_vars = set(output_vars(query)) + if not query_out_vars: + fail("the Query Entity Records node has no for downstream nodes to reference") + for name in ("matchedInvoiceNumber", "lineItemCount"): + out_id = outputs[name][0] + if not derives_from(out_id, query_out_vars, var_sources, extra_refs): + fail( + f"public output {name!r} does not derive from the Query Entity Records node's own " + f"output (vars.{{{', '.join(sorted(query_out_vars))}}}) -- the value must come FROM the query step" + ) + + # 7. connection binding wired, and (when folder-scoped) resolves to a + # DISTINCT folder key (authoring-time half of :130-141; see module docstring) + if not connection_binding_wired(root, query): + fail( + "the query's `connection` input is not a `=bindings.` reference to a declared " + "resource=\"Connection\" propertyAttribute=\"ConnectionId\" binding" + ) + distinct = connection_and_folder_distinct(root, query) + if distinct is False: + fail( + "the connection binding's `default` and its folderKey binding's `default` are the same " + "value (or blank) -- the folder binding needs the connection's FOLDER key, not its " + "connection id (measured live: the two collapse into one binding and the dispatch sends " + "the folder key as --connection-id, answering 401)" + ) + + print( + f"OK: {bpmn_path} -- 1 Query Entity Records node on {ENTITY!r}; filter computed from vars; " + "outputs read from query; connection binding wired" + + ("" if distinct is None else " with a distinct folder key") + ) + + +DISPATCH = { + "lookup": lookup, + "server_side_filter": server_side_filter, + "bindings": bindings, + "advisory": advisory, +} + + +def main() -> None: + if len(sys.argv) < 2 or sys.argv[1] not in DISPATCH: + sys.exit(f"usage: {sys.argv[0]} {{{'|'.join(DISPATCH)}}}") + DISPATCH[sys.argv[1]]() + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py new file mode 100644 index 0000000000..bfde963a73 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py @@ -0,0 +1,279 @@ +#!/usr/bin/env python3 +"""SlackChannelDescription (BPMN): structural + live checks. + +Ported from Flow `multi_node/slack_channel_description/_shared/check_channel_description.py` +(via `tests/tasks/uipath-maestro-flow/_shared/check_channel_description.py`): same +scenario (a manual-start process retrieves the channel description of +#office-bellevue via the Slack Integration Service connector and outputs it), +translated from a JSON node walk + inline `flow debug` payload to an XML walk +over the registry-driven `Intsvc.ActivityExecution` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) plus the +BPMN live-debug surface (`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical +pattern: ephemeral solution import, `bpmn debug`, `debug-instance +variables-all`/`incidents`). + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. Nothing is created against the tenant by this scenario (it only reads +a channel's description), so there is no side-effect record to tear down -- +matching Flow's own grader, which has no teardown either. + +Assertion map (Flow → BPMN): + F check_channel_description.py:27 assert_flow_uses_connector_target('uipath-salesforce-slack') + → find_connector_nodes(): any element carrying + Intsvc.ActivityExecution whose connectorKey context field + equals uipath-salesforce-slack. No operation filter -- Flow's + own assertion has none either, so this port keeps the same + breadth rather than narrowing it. + F check_channel_description.py:28 run_debug(timeout=240) implicitly requires finalStatus == + "Completed" (flow_check.run_debug raises on a non-Completed + status internally; `bpmn debug` returns only an instance id, so + the check is explicit here) + → FinalStatus in COMPLETED_STATUSES and debug-instance + incidents is empty + F check_channel_description.py:29 assert_outputs_contain(payload, ADDRESS_FRAGMENTS, require_all=True) + → every fragment found among the root scope's variable leaves + AND every element's Outputs (incl. nested connector `response`) + in `debug-instance variables-all` (LIVE-ADDENDUM: a root PUBLIC + OUTPUT has read back null even when mapped correctly, so the + search is not scoped to one declared output variable) + I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) + → bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported + bytes against the submitted file -- `bpmn debug` runs against an imported project, + unlike `flow debug`, which runs directly against the discovered project directory + → LIVE-ADDENDUM canonical live pattern (mirrors + e2e/jira_get_issue/_shared/check_jira_get_issue.py) + DROPPED the HTTP-proxy fallback branch of assert_flow_uses_connector_target (a + `core.action.http.v2` node with bodyParameters.targetConnector) -- a legacy Flow + accommodation for connector-backed flows authored before native connector node + types existed. The BPMN skill's registry enrichment always emits + Intsvc.ActivityExecution for a connector activity (registry-workflow.md §3), so no + analogous construct exists to translate. + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- not in Flow; the + `bpmn validate` criterion covers structure. +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +CONNECTOR_KEY = "uipath-salesforce-slack" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +NAME_HINT = "SlackChannelDescription" + +ADDRESS_FRAGMENTS = [ + "700 Bellevue Way NE", + "Suite 2000", + "Bellevue", + "WA 98004", +] + +LIVE_RUN_DIR = Path("slack-channel-description-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in slack_channel_description.yaml documents the +# arithmetic (mirrors e2e/jira_get_issue/jira_get_issue.yaml): +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (600) does not fit the extra CLI steps a BPMN +# live grade needs (solution init/import, separate variables-all/incidents +# reads), so it is raised to 1050 -- the one sanctioned deviation from +# "criteria identical" (LIVE-ADDENDUM: a property of the CLI surface, not of +# what is graded). + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution targeting connector_key. + + No operation filter: Flow's own assertion (`assert_flow_uses_connector_target`) + only requires SOME connector node for the key, not a specific op, so this + mirrors that breadth. Scans every descendant, not a fixed tag list (registry + templates may emit a connector activity as sendTask, serviceTask, or a plain + task) -- mirrors bpmn_live.index_runtime_connectors' own scanning discipline. + """ + found = [] + for node in root.iter(): + context = connector_context(node) + if context.get("connectorKey") != connector_key: + continue + if ACTIVITY_TYPE not in ET.tostring(node, encoding="unicode"): + continue + found.append(node) + return found + + +def collect_output_haystack(variables_data: object) -> str: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- it mirrors Flow's own assert_outputs_contain(), + which flattens the whole outputs payload. Element Outputs include a + connector's nested `response` object, which _leaves() flattens along with + everything else. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return "\n".join(str(v) for v in leaves).lower() + + +def main() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + raw = Path(bpmn_path).read_text(encoding="utf-8") + if CONNECTOR_KEY not in raw: + _fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") + print(f"OK: bpmn references {CONNECTOR_KEY}") + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + connector_nodes = find_connector_nodes(root, CONNECTOR_KEY) + if not connector_nodes: + _fail( + f"bpmn does not reference a {CONNECTOR_KEY} connector node " + f"({ACTIVITY_TYPE})" + ) + print(f"OK: bpmn references a {CONNECTOR_KEY} connector node") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "SlackChannelDescriptionLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + haystack = collect_output_haystack(variables_data) + missing = [f for f in ADDRESS_FRAGMENTS if f.lower() not in haystack] + if missing: + _fail( + f"outputs missing address fragments {missing}; " + f"expected all of {ADDRESS_FRAGMENTS}\noutputs: {haystack[:1000]}" + ) + print("OK: bpmn outputs contain the Bellevue office address") + print("PASS: all SlackChannelDescription checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py new file mode 100644 index 0000000000..dcd7415b05 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py @@ -0,0 +1,174 @@ +#!/usr/bin/env python3 +"""DatabricksQuery (BPMN): structural checks for the JDBC Execute-Query sendTask. + +Ported from Flow +`connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml`'s +`check_databricks_query.py`: same scenario (a Databricks-via-JDBC aggregate +SQL query must route through the Database Hub / JDBC gateway connector +`uipath-uipath-jdbc`, not the native Databricks connector or a manual HTTP +call), translated from a JSON node/edge walk to an XML walk over the +registry-driven `Intsvc.ActivityExecution` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4). + +Catalog checked read-only via `uip is activities list uipath-uipath-jdbc +--output json`: the connector exposes exactly one curated activity capable +of running raw SQL -- `ExecuteQuerySynchronously` (ObjectName `query`, +MethodName `POST`, IsCurated `Yes`). Its record-CRUD siblings +(DeleteRecord/GetRecord/InsertRecord/ListAllRecords/ReplaceRecord) cannot +express the prompt's GROUP BY / HAVING / AVG / ORDER BY aggregate, so there +is no generic-CRUD alternate form to also accept here (contrast the Data +Service ports' curated-or-generic duality). + +Assertion map (Flow -> BPMN): + F check_databricks_query.py:66 assert_flow_uses_connector_target(JDBC_KEY) + -- native `uipath.connector..*` node-type match + (flow_check.py:743-744) -> connector_tasks() finds >=1 bpmn:sendTask carrying Intsvc.ActivityExecution + with context connectorKey == uipath-uipath-jdbc + F check_databricks_query.py:72-78 _references_op (node type OR + inputs.detail JSON contains "execute-query-synchronously") -> references_op() matches the catalog's curated objectName+method + ("query"/"POST") OR an "executequerysynchronously" token anywhere in + the node's inputs (name/value/text, any depth) + F check_databricks_query.py:83-86 native-Databricks guard + (NATIVE_DATABRICKS_KEY not in any node type) -> no sendTask context connectorKey == uipath-databricks-databricks + I locate/parse .bpmn (file exists, well-formed XML) -> parse_bpmn("DatabricksQuery") + T inputs collected at any depth under uipath:activity (Flow read a + single JSON `inputs.detail` dict; BPMN may nest context/path/query/ + body inputs) -> node_inputs() / node_text_blob() walk `.//uipath:input` + T curated objectName/method spelling as an alternate to a literal + operation-name token match (registry curated activity; no + generic-CRUD alternate form exists for this operation) -> references_op() + + DROPPED connection/folderKey binding presence check -- Flow's + assert_flow_uses_connector_target only enforces non-empty + connection id + folder key on the generic core.action.http + fallback branch (flow_check.py:757-764); the native + `uipath.connector.*` node-type branch this task actually + exercises returns on connectorKey match alone + (flow_check.py:743-744), with no binding check. (not enforced by Flow for this node shape) + DROPPED require_no_private_connector_values / require_sequence_integrity + / require_di_for_visible_elements (not in Flow; `bpmn validate` criterion covers structure) + DROPPED a distinct "HTTP-fallback with connector auth" acceptance path + (Flow's assert_flow_uses_connector_target has one for + core.action.http nodes) (no separate BPMN wrapper for connector-authenticated HTTP exists -- + see check_non_catalog_http_fallback.py's GUESS note: all catalog + connector activities, including HTTP-connector-mode ones, route + through the same Intsvc.ActivityExecution shell) + +Checks performed: + 1. BPMN file exists and is well-formed XML. + 2. A bpmn:sendTask carries Intsvc.ActivityExecution with connectorKey + uipath-uipath-jdbc. + 3. That node (or another uipath-uipath-jdbc node) references the Execute + Query Synchronously operation. + 4. No sendTask carries connectorKey uipath-databricks-databricks (the + native Databricks connector) -- Databricks SQL must route through the + JDBC gateway, per the special-SDK-case the prompt exercises. + +The prompt asks for an aggregate over the `employees` table (group by +department, keep departments with two or more employees, order by average +salary) -- a shape expressible only via raw SQL, not the generic record +activities. As in the Flow original, the exact SQL the agent authors is not +asserted here; this checker validates only the connector wiring described +above. +""" + +from __future__ import annotations + +import os +import re +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import NS, elements, fail, parse_bpmn # noqa: E402 + +JDBC_KEY = "uipath-uipath-jdbc" +NATIVE_DATABRICKS_KEY = "uipath-databricks-databricks" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +OP_TOKEN = "executequerysynchronously" + + +def _normalize(text: str) -> str: + return re.sub(r"[^a-z0-9]", "", text.lower()) + + +def has_type(el: ET.Element, token: str) -> bool: + return token in ET.tostring(el, encoding="unicode") + + +def node_inputs(task: ET.Element) -> list[ET.Element]: + # Any depth under the sendTask: agents sometimes nest body/query/path + # inputs inside uipath:context rather than as its siblings. + return task.findall(".//uipath:input", NS) + + +def context_value(task: ET.Element, name: str) -> str: + for inp in node_inputs(task): + if inp.attrib.get("name") == name: + return inp.attrib.get("value") or (inp.text or "") + return "" + + +def node_text_blob(task: ET.Element) -> str: + # Serialize only the uipath:activity payload (context + inputs/outputs), + # not the enclosing bpmn:sendTask's own id/name attributes -- a node's + # free-text display name (e.g. a human label) is not wiring evidence and + # must not satisfy the operation-token fallback below. + activity = task.find(".//uipath:activity", NS) + return ET.tostring(activity if activity is not None else task, encoding="unicode") + + +def connector_tasks(root: ET.Element, connector_key: str) -> list[ET.Element]: + return [ + task + for task in elements(root, "sendTask") + if has_type(task, ACTIVITY_TYPE) + and context_value(task, "connectorKey") == connector_key + ] + + +def references_op(task: ET.Element) -> bool: + object_name = context_value(task, "objectName").strip().lower() + method = context_value(task, "method").strip().upper() + if object_name == "query" and method == "POST": + return True + return OP_TOKEN in _normalize(node_text_blob(task)) + + +def main() -> None: + path, root = parse_bpmn("DatabricksQuery") + + jdbc_tasks = connector_tasks(root, JDBC_KEY) + if not jdbc_tasks: + fail(f"no bpmn:sendTask carrying {ACTIVITY_TYPE} for connector key {JDBC_KEY!r}") + print(f"OK: {JDBC_KEY} sendTask present") + + if not any(references_op(task) for task in jdbc_tasks): + fail( + f"no {JDBC_KEY} node references the Execute Query Synchronously " + "operation (expected objectName=query + method=POST, or an " + "ExecuteQuerySynchronously token in the node's inputs)" + ) + print( + f"OK: {JDBC_KEY} node references Execute Query Synchronously " + "(objectName=query, method=POST)" + ) + + native_tasks = connector_tasks(root, NATIVE_DATABRICKS_KEY) + if native_tasks: + fail( + "process references the native Databricks connector " + f"({NATIVE_DATABRICKS_KEY}) -- Databricks SQL must route through " + "the JDBC gateway, not the native key" + ) + print("OK: no native Databricks connector node present") + + print( + f"OK: {path} wires {JDBC_KEY}.ExecuteQuerySynchronously with no " + "native-Databricks fallback" + ) + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py new file mode 100644 index 0000000000..95078183bd --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -0,0 +1,164 @@ +#!/usr/bin/env python3 +"""Data Fabric smoke_error (BPMN): Create against a non-existent entity; +two Queries against FlowCodeEvalEntity are unaffected by the failure. + +Ported from Flow `connector_features/datafabric_connector/smoke_error.yaml`'s +``check_smoke_error.py``: same scenario and the same entity-binding-only +check (Flow's own docstring: "entity-binding check only; topology [the +parallel branch] is not parsed -- the prompt-driven shape plus entity split +already blocks the common wrong-reason paths"), translated from a JSON +node/`inputs.detail` walk to an XML walk over the registry-driven +``Intsvc.ActivityExecution`` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4). + +BPMN has no fixed home for `entityName`: the registry does not pin it to a +path param, query param, or JSON body key, and a node may also carry it only +as the generic-form `objectName` when the skill emits the generic +entity-CRUD shape instead of the curated per-operation one -- see +BATCH1-ADDENDUM.md "Where connector node values live in BPMN" and its "CI +run 35488848026" section on the two valid activity shapes. Both forms are +accepted here, exactly as `check_df_integration_create_get.py` and +`check_df_smoke_query_filter.py` already do for this connector. + +Assertion map (Flow -> BPMN): + F check_smoke_error.py:29-31 entity_of(node) == NonExistentEntity on a + `.create-entity-record` node -> is_create_node() + mentions_entity() + F check_smoke_error.py:32-34 entity_of(node) == FlowCodeEvalEntity on + `.query-entity-records` nodes -> is_query_node() + mentions_entity() + F check_smoke_error.py:37-39 `not error_creates` -> fail -> `error_creates < 1` check + F check_smoke_error.py:40-42 `len(good_queries) < 2` -> fail -> `good_queries < 2` check + I locate/parse .bpmn -> parse_bpmn() + T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() + T entity name anywhere in inputs -> mentions_entity() + DROPPED topology/parallel-branch parsing (Flow's own grader does not parse it either -- see its docstring) + DROPPED require_no_private_connector_values (not in Flow) + DROPPED require_sequence_integrity (not in Flow; `bpmn validate` criterion covers structure) + DROPPED require_di_for_visible_elements (not in Flow; `bpmn validate` criterion covers structure) + +Checks performed: + 1. BPMN file exists and is well-formed XML. + 2. >=1 bpmn:sendTask carries Intsvc.ActivityExecution with connectorKey + uipath-uipath-dataservice, classified as a Create (curated objectName, + or generic objectName + Create/POST verb), and mentions + NonExistentEntity somewhere in its inputs. + 3. >=2 such nodes classified as a Query (curated Query Entity Records + objectName, or generic objectName + List/GET verb), and mention + FlowCodeEvalEntity somewhere in their inputs. +""" + +from __future__ import annotations + +import os +import re +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import ( # noqa: E402 + NS, + elements, + fail, + has_typed_uipath_extension, + parse_bpmn, +) + +CONNECTOR_KEY = "uipath-uipath-dataservice" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +CREATE_ENTITY = "NonExistentEntity" +QUERY_ENTITY = "FlowCodeEvalEntity" +CREATE_CURATED_NAMES = {"createentityrecordcurated", "createentityrecord_v3"} +QUERY_CURATED_NAMES = {"queryentityrecordscurated", "queryentityrecords_v3"} + +_CREATE_OP_RE = re.compile(r"^create$", re.IGNORECASE) +_LIST_OP_RE = re.compile(r"^list$", re.IGNORECASE) + + +def node_inputs(task: ET.Element) -> list[ET.Element]: + return task.findall(".//uipath:input", NS) + + +def input_val(inp: ET.Element) -> str: + return inp.attrib.get("value") or (inp.text or "") + + +def context_value(task: ET.Element, name: str) -> str: + for inp in node_inputs(task): + if inp.attrib.get("name") == name: + return input_val(inp) + return "" + + +def mentions_entity(task: ET.Element, entity: str) -> bool: + for inp in node_inputs(task): + value = inp.attrib.get("value") or "" + text = inp.text or "" + if entity in value or entity in text: + return True + return False + + +def is_generic_entity_object(object_name: str, entity: str) -> bool: + return object_name.strip().lower() == entity.strip().lower() + + +def connector_nodes(root: ET.Element) -> list[ET.Element]: + return [ + task + for task in elements(root, "sendTask") + if has_typed_uipath_extension(task, "activity", ACTIVITY_TYPE) + and context_value(task, "connectorKey") == CONNECTOR_KEY + ] + + +def is_create_node(task: ET.Element, object_name: str, entity: str) -> bool: + if object_name.strip().lower() in CREATE_CURATED_NAMES: + return True + if not is_generic_entity_object(object_name, entity): + return False + operation = context_value(task, "operation").strip() + method = context_value(task, "method").strip().upper() + return bool(_CREATE_OP_RE.match(operation)) or method == "POST" + + +def is_query_node(task: ET.Element, object_name: str, entity: str) -> bool: + if object_name.strip().lower() in QUERY_CURATED_NAMES: + return True + if not is_generic_entity_object(object_name, entity): + return False + operation = context_value(task, "operation").strip() + method = context_value(task, "method").strip().upper() + return bool(_LIST_OP_RE.match(operation)) or method == "GET" + + +def main() -> None: + path, root = parse_bpmn() + + error_creates = 0 + good_queries = 0 + + for task in connector_nodes(root): + object_name = context_value(task, "objectName") + if is_create_node(task, object_name, CREATE_ENTITY) and mentions_entity(task, CREATE_ENTITY): + error_creates += 1 + if is_query_node(task, object_name, QUERY_ENTITY) and mentions_entity(task, QUERY_ENTITY): + good_queries += 1 + + if error_creates < 1: + fail(f"no Create Entity Record node targeting {CREATE_ENTITY!r}") + print(f"OK: {error_creates} Create Entity Record node(s) on {CREATE_ENTITY}") + + if good_queries < 2: + fail( + f"expected >=2 Query Entity Records node(s) on {QUERY_ENTITY!r}, " + f"found {good_queries}" + ) + print(f"OK: {good_queries} Query Entity Records node(s) on {QUERY_ENTITY}") + + print( + f"OK: {path} -- create on {CREATE_ENTITY}, {good_queries} query on {QUERY_ENTITY}" + ) + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py new file mode 100644 index 0000000000..2a0be82903 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py @@ -0,0 +1,453 @@ +#!/usr/bin/env python3 +"""Verify the escalation BPMN actually creates a Jira ticket. + +Ported from Flow's ``uipath-maestro-flow/e2e/escalation_jira_ticket/check_escalation_jira_ticket.py``, +adapted to the BPMN CLI surface the way ``e2e/customer_escalation_triage/``'s +``check_customer_escalation_behavior.py`` adapted the sibling Flow live check: +Flow's ``flow debug`` returns variables inline and addresses outputs by name; +``uip maestro bpmn debug`` returns an instance id whose evidence comes from +``debug-instance variables-all``/``incidents``, with runtime variables +addressed by id. No Slack step here — Flow's ``escalation_jira_ticket`` task +has none either (unlike ``customer_escalation_triage``, which is a different, +Jira+Slack scenario); only Flow's Jira assertions are ported. + +Outcome-based, tenant-confirmed: + 1. The submitted BPMN references the uipath-atlassian-jira connector and + declares the three public outputs plus one Jira Create-Issue activity. + 2. LIVE: the exact submitted project (sha256-pinned through solution + import) runs the seeded Sev1 case via `uip maestro bpmn debug` inside an + ephemeral solution under the sandbox CWD (repo-standard + `_setup/cleanup_solutions.py` post_run sweep finds the .uipx there). + 3. TENANT: re-reading the created Jira key returns an issue whose summary + carries the seeded correlationId — proof the process created THIS run's + ticket, not a fabricated key. + +Assertion map (Flow -> BPMN): + I locate/parse .bpmn (no pinned path -- prompt names only the process) -> bpmn_check.parse_bpmn("EscalationJiraTicket") + resolve_project() + F check_escalation_jira_ticket.py:53-56 .flow references JIRA_KEY -> .bpmn source text contains JIRA_KEY substring + I uipath:variables output ids / Jira Create-Issue element ids / -> resolve_contract() (mirrors check_customer_escalation_behavior.py's Contract, Jira-only) + scriptTask ids needed to address runtime evidence by id + T flow_check.run_debug(inputs=..., retries=1) -- single attempt, -> ephemeral solution import (sha256-pinned) + bpmn_live.run_debug(); + finalStatus == "Completed" checked inline FinalStatus/incidents asserted explicitly afterward (LIVE-ADDENDUM + canonical live pattern; bpmn debug already makes one attempt, no backoff) + F check_escalation_jira_ticket.py:64-68 no whole-run retries -> bpmn_live.run_debug has no retry/backoff parameter to begin with + (a retried Create-Issue would duplicate the ticket) + F check_escalation_jira_ticket.py:70-90 except-branch: on a debug -> on subprocess.TimeoutExpired from run_debug, scrape partial + timeout, best-effort scrape partial output for -\\d+ stdout/stderr for -\\d+ candidates, keep only ones whose + candidates, keep only ones owned (summary carries correlationId) summary carries correlationId, journal them, then fail + F check_escalation_jira_ticket.py:104-106 Jira Create-Issue node -> Jira Create-Issue element has exactly one Completed + specifically must have completed (not merely any Jira node) ElementExecutions record + F check_escalation_jira_ticket.py:108-111 candidate keys from -> connector_response_values() on the Create-Issue element's OWN + collect_outputs()/raw debug text, ISSUE_KEY_RE-shaped Outputs (element_output_records), value "key" + F check_escalation_jira_ticket.py:118-123 persist proven-created keys -> journal the harvested key to `.created_keys` BEFORE the tenant + BEFORE the fallible tenant reread reread, mirroring the exemplar's harvest-before-assert order + F check_escalation_jira_ticket.py:129-140 tenant reread: get_issue(), -> jira_is.get_issue(conn, key) (copied verbatim from Flow), summary + summary contains correlationId, never an unrelated pre-existing issue contains correlationId + F check_escalation_jira_ticket.py:143-149 created key must be in the -> inherent by construction: jira_key is sourced ONLY from the + executed Create-Issue node's OWN output Create-Issue element's own Outputs (see harvest above) + F check_escalation_jira_ticket.py:154-155 assert_named_equals( -> declared output `jiraIssueKey` resolved via its uipath:variables + "jiraIssueKey", match, case_sensitive=True) output id (root scope Globals), compared case-sensitively + F check_escalation_jira_ticket.py:158-160 assert_named_equals per -> same, via declared output ids; severity case-insensitive, + seed["expected"] (severity case-insensitive, caseKey case-sensitive) caseKey case-sensitive (CASE_SENSITIVE set, ported verbatim) + F check_escalation_jira_ticket.py:162-186 severity AND engineeringNeeded -> same binding, over bpmn:scriptTask elements' own Outputs + bound to the SAME executed Script node, not split across two nodes (element_output_records); a node's response value-pool must + contain both the expected severity and engineeringNeeded value + T check_escalation_jira_ticket.py's severity/engineeringNeeded lookup -> matched against the VALUES of the scriptTask's own response + is by normalized field NAME (find_node_output_value) dict rather than by key name (mirrors check_customer_escalation_ + behavior.py's `carries()`, the CI-proven pattern for this exact + runtime shape, since BPMN scriptTask output key-naming is agent- + chosen and not part of the registry contract) + + DROPPED check_customer_escalation_behavior.py's OUTPUT_TYPES exact_type() (not in Flow -- assert_named_equals never type-checks output values) + check on output values + DROPPED check_customer_escalation_behavior.py's Jira project.key/ (not in Flow -- Flow only checks the summary contains correlationId) + issuetype.id remote-field re-check + DROPPED assert_live_target() tenant-lock guard (not in Flow's jira_is.py, which this task's _setup/jira_is.py is a + verbatim copy of; not adding it keeps that copy faithful) + +Budget arithmetic (LIVE-ADDENDUM): bpmn_live.debug_budget() default (480) + +SOLUTION_INIT_TIMEOUT (90) + SOLUTION_IMPORT_TIMEOUT (180) + +VARIABLES_ALL_TIMEOUT (120) + INCIDENTS_TIMEOUT (120) + one jira_is. +connection_id() call (120, hardcoded inside jira_is._run) + up to +MAX_CANDIDATE_ISSUE_READS jira_is.get_issue() calls (2 * 120 = 240) = 1350, +plus bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1410 -- under Flow's original +1600s criterion timeout, so it is kept verbatim (see escalation_jira_ticket.yaml). + +The confirmed key is written to `.created_keys` so post_run's `teardown_jira.py` +(copied verbatim from Flow) deletes it even if a later assertion fails. +""" + +from __future__ import annotations + +import json +import os +import re +import subprocess +import sys +import xml.etree.ElementTree as ET +from dataclasses import dataclass +from pathlib import Path + +HERE = os.path.dirname(os.path.abspath(__file__)) # .../uipath-maestro-bpmn/_shared +SUITE_ROOT = os.path.dirname(HERE) # .../uipath-maestro-bpmn +TASK_SETUP = os.path.join(SUITE_ROOT, "e2e", "escalation_jira_ticket", "_setup") +sys.path.insert(0, TASK_SETUP) # jira_is.py, copied verbatim from the Flow task +sys.path.insert(0, SUITE_ROOT) # _shared package (bpmn_check, bpmn_live) + +import jira_is # noqa: E402 +from _shared.bpmn_check import parse_bpmn, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + BPMN_NS, + CheckFailure, + connector_response_values, + element_output_records, + get_ci, + incident_records, + index_runtime_connectors, + payload_data, + q, + resolve_runtime_key, + root_scope, + run_cli, + run_debug, + sha256, + UIPATH_NS, +) + +JIRA_CONNECTOR = jira_is.CONNECTOR # "uipath-atlassian-jira" +JIRA_CREATE_OP = "curated_create_issue" # matches the catalog op customer_escalation_triage's exemplar graded against +ISSUE_KEY_RE = re.compile(r"^[A-Z][A-Z0-9]*-\d+$") +CASE_SENSITIVE = {"caseKey", "jiraIssueKey"} # opaque ids -- exact-case match +OUTPUT_NAMES = ("severity", "caseKey", "jiraIssueKey") +COMPLETED_STATUSES = {"Completed", "Successful"} + +LIVE_RUN_DIR = Path("escalation-jira-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +MAX_CANDIDATE_ISSUE_READS = 2 # headroom; one Create-Issue node normally executes once + + +def _fail(msg: str) -> None: + raise CheckFailure(msg) + + +def _normalized(value, *, case_fold: bool = True): + """Mirror flow_check.normalized: trim strings, coerce true/false, fold case + for enum-like values. Pass case_fold=False for opaque identifiers.""" + if isinstance(value, str): + text = value.strip() + lowered = text.casefold() + if lowered == "true": + return True + if lowered == "false": + return False + return lowered if case_fold else text + return value + + +@dataclass(frozen=True) +class Contract: + output_ids: dict + jira_create_ids: tuple + classifier_ids: tuple + + +def resolve_contract(root: ET.Element) -> Contract: + process = root.find(q(BPMN_NS, "process")) + if process is None: + _fail("BPMN must contain one root process") + + variables = process.find( + f"./{q(BPMN_NS, 'extensionElements')}/{q(UIPATH_NS, 'variables')}" + ) + if variables is None: + _fail("root process is missing uipath:variables") + output_ids: dict = {} + for variable in variables: + name = variable.attrib.get("name") + identifier = variable.attrib.get("id") + if ( + variable.tag.rsplit("}", 1)[-1] != "output" + or not name + or not identifier + or name not in OUTPUT_NAMES + ): + continue + if name in output_ids: + _fail(f"public output {name!r} is declared more than once") + output_ids[name] = identifier + missing = sorted(set(OUTPUT_NAMES) - set(output_ids)) + if missing: + _fail(f"public outputs not declared: {missing}") + + # Every scriptTask that could be the classifier -- Flow binds severity to + # the executed Script whose own output carries it; without an equivalent, a + # process that exposes a literal "Sev1" (computing nothing) would satisfy + # every other criterion here. + classifier_ids = tuple( + element.attrib["id"] + for element in process.iter(q(BPMN_NS, "scriptTask")) + if element.attrib.get("id") + ) + if not classifier_ids: + _fail( + "no bpmn:scriptTask to classify severity; the value would be a " + "literal rather than computed" + ) + + connectors = index_runtime_connectors(process) + jira_create_ids = tuple( + element_id + for (key, route), element_ids in connectors.items() + if key == JIRA_CONNECTOR and JIRA_CREATE_OP in route + for element_id in element_ids + ) + if not jira_create_ids: + _fail( + f"no {JIRA_CONNECTOR} activity with a registry path containing " + f"{JIRA_CREATE_OP!r}" + ) + + return Contract( + output_ids=output_ids, + jira_create_ids=jira_create_ids, + classifier_ids=classifier_ids, + ) + + +def _harvest_jira_keys(contract: Contract, variables_data) -> list: + """Candidate Jira keys from the Create-Issue element's OWN Outputs. + + Called (and journalled) BEFORE any assertion -- the debug already created + a real issue, and this read is the only place its key appears. + """ + outputs = element_output_records(variables_data, contract.jira_create_ids) + keys = [v for v in connector_response_values(outputs, "key") if isinstance(v, str)] + return list(dict.fromkeys(keys)) + + +def _journal(keys) -> None: + if keys: + Path(".created_keys").write_text("\n".join(keys) + "\n") + + +def _recover_partial_keys(project_key: str, correlation: str, raw_text: str) -> list: + """Best-effort: on a client-side debug timeout, the Create-Issue call may + already have succeeded server-side. Scrape any - candidates + from partial CLI output and keep only ones tenant-confirmed as THIS run's + (summary carries correlationId) -- never an unrelated pre-existing issue. + Mirrors Flow's except-branch (check_escalation_jira_ticket.py:70-90).""" + cands = list(dict.fromkeys(re.findall(rf"\b{re.escape(project_key)}-\d+\b", raw_text))) + if not cands: + return [] + try: + conn = jira_is.connection_id() + except SystemExit: + return [] + owned = [] + for key in cands: + try: + fields = jira_is.get_issue(conn, key) + except Exception: # noqa: BLE001 -- best-effort recovery, never mask the real failure + continue + if fields is not None and correlation in str(fields.get("summary", "")): + owned.append(key) + return owned + + +def main() -> None: + seed = json.loads(Path("seed.json").read_text(encoding="utf-8")) + correlation = seed["correlationId"] + project_key = seed["project_key"] + + bpmn_path, root = parse_bpmn("EscalationJiraTicket") + bpmn_text = Path(bpmn_path).read_text(encoding="utf-8") + if JIRA_CONNECTOR not in bpmn_text: + _fail(f"no .bpmn references the {JIRA_CONNECTOR} connector (found {bpmn_path})") + print(f"OK: BPMN references {JIRA_CONNECTOR}") + + contract = resolve_contract(root) + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "EscalationJiraLiveEval" + initialized = run_cli(["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + _fail( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", "solution", "projects", "import", str(project_dir.resolve()), + "--solutionFile", str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + _fail("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + # No whole-run retries: this process CREATES a Jira issue, so a retried + # whole run on a transient error could create a duplicate ticket that this + # checker (deriving keys from the final attempt) wouldn't see or clean up. + # bpmn_live.run_debug makes a single attempt with no backoff by design. + log_file = LIVE_RUN_DIR / "debug.log" + try: + debug_data, instance_id = run_debug(imported_project, seed["inputs"], log_file) + except subprocess.TimeoutExpired as exc: + partial = "".join( + s.decode() if isinstance(s, bytes) else (s or "") for s in (exc.stdout, exc.stderr) + ) + owned = _recover_partial_keys(project_key, correlation, partial) + _journal(owned) + _fail( + f"bpmn debug timed out after {exc.timeout}s" + + (f"; recorded this-run key(s) {owned} for teardown" if owned else "") + ) + print(f"OK: debug completed (instance {instance_id})") + + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + # Journal the created key BEFORE any assertion -- it was created regardless + # of the verdict below, and post_run's teardown_jira.py replays the journal + # even if this process is killed mid-assertion. + jira_keys = _harvest_jira_keys(contract, variables_data) + _journal(jira_keys) + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + + final_status = get_ci(debug_data, "FinalStatus") + if final_status not in COMPLETED_STATUSES: + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + records = incident_records(incidents_data) + detail = [] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if records: + detail.append(f"incidents: {json.dumps(records)[:1500]}") + _fail(f"bpmn debug did not complete (finalStatus={final_status})" + ("; " + "; ".join(detail) if detail else "")) + print("OK: bpmn debug completed") + + records = incident_records(incidents_data) + if records is None: + _fail(f"incidents response has an unknown shape: {incidents_data!r}") + if records: + _fail(f"unexpected incidents: {records}") + + # Execution evidence: the Jira CREATE-ISSUE element specifically must have + # executed (not merely any Jira element -- a read op could surface an + # authoring-time key). + executions = get_ci(debug_data, "ElementExecutions", []) + executed_ids = [get_ci(item, "ElementId") for item in executions if isinstance(item, dict)] + count = sum(executed_ids.count(eid) for eid in contract.jira_create_ids) + if count < 1: + _fail( + "no Jira Create-Issue element completed in the debug trace -- the " + "debugged BPMN did not execute the Create Issue activity" + ) + + if not jira_keys: + _fail( + f"no Jira issue key (e.g. {project_key}-123) in the Create-Issue " + "element's own output -- the process did not create a ticket" + ) + print(f"OK: candidate keys from debug: {jira_keys}") + + conn = jira_is.connection_id() + + # Tenant-confirm: the key belongs to THIS run -- an issue whose summary + # carries this run's correlationId. Never deletes an unrelated + # pre-existing issue. + owned = [ + k for k in jira_keys + for fields in [jira_is.get_issue(conn, k)] + if fields is not None and correlation in str(fields.get("summary", "")) + ] + _journal(owned or jira_keys) # re-journal narrowed to confirmed-owned when possible + if not owned: + _fail( + f"none of {jira_keys} is a Jira issue whose summary contains " + f"{correlation!r} -- the process did not create the expected " + "escalation ticket" + ) + match = owned[0] + print(f"OK: Jira ticket {match} exists and its summary carries {correlation!r}") + + # Public outputs, addressed by id in the root scope's globals. + globals_map = get_ci(root_scope(variables_data), "Globals", {}) + if not isinstance(globals_map, dict): + _fail(f"root scope Globals is not a map: {globals_map!r}") + + def assert_named_equals(name: str, expected, *, case_sensitive: bool = False) -> None: + identifier = contract.output_ids[name] + actual = resolve_runtime_key(globals_map, identifier, name) + if actual is None or (isinstance(actual, str) and not actual.strip()): + _fail(f"output {name!r} missing or empty") + if _normalized(actual, case_fold=not case_sensitive) != _normalized(expected, case_fold=not case_sensitive): + _fail(f"output {name!r}: expected {expected!r}, got {actual!r}") + + # The exposed jiraIssueKey must be the executed Create-Issue element's OWN + # response key -- harvesting some other key cannot satisfy this (jira_keys + # is sourced only from that element's Outputs, so this is inherent). + assert_named_equals("jiraIssueKey", match, case_sensitive=True) + + for name, expected in (seed.get("expected") or {}).items(): + assert_named_equals(name, expected, case_sensitive=(name in CASE_SENSITIVE)) + + # Severity must be COMPUTED, not exposed as a literal, AND bound to the + # SAME executed scriptTask that also carries engineeringNeeded -- Flow + # requires both fields from ONE node, not split across two cosmetic Scripts. + expected_severity = (seed.get("expected") or {}).get("severity") + expected_script = seed.get("expected_script") or {} + + def node_carries_all(node_id: str) -> bool: + for record in element_output_records(variables_data, node_id): + response = get_ci(record, "response") + pool = list(response.values()) if isinstance(response, dict) else [response] + if not any(_normalized(v) == _normalized(expected_severity) for v in pool): + continue + if all( + any(_normalized(v) == _normalized(expected_value) for v in pool) + for expected_value in expected_script.values() + ): + return True + return False + + if not any(node_carries_all(nid) for nid in contract.classifier_ids): + _fail( + "no single executed scriptTask carries the expected severity AND " + f"classification fields together (severity={expected_severity!r}, " + f"{expected_script})" + ) + + print("PASS: escalation BPMN created a real Jira ticket with the expected classification") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py new file mode 100644 index 0000000000..e34a6cb210 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py @@ -0,0 +1,600 @@ +#!/usr/bin/env python3 +"""Run each seeded escalation-orchestrator path case and verify its outcome. + +Ported from Flow `e2e/escalation_orchestrator_paths/check_escalation_orchestrator_paths.py`: +same seven seeded cases (Sev1/Sev2/Sev3 escalation + four triage reasons), same +graded behaviours, translated from `flow debug`'s inline-variable payload to +`bpmn debug`'s instance-id + `debug-instance variables-all` surface (see +`_shared/bpmn_live.py` and the LIVE-tier addendum this port followed). + +Outcome-based: for every case the grader runs `uip maestro bpmn debug --inputs` +against ONE ephemeral-solution import of the exact submitted project, then reads +`debug-instance variables-all` for the expected values. For cases flagged +`expect_slack`, it additionally requires the completed Slack sendTask's OWN +runtime response to carry a real message timestamp — Slack only returns one +when a message is really delivered, not merely reached. + +Assertion map (Flow -> BPMN): + F check_escalation_orchestrator_paths.py:106 assert_flow_uses_connector_target(SLACK_KEY) -> Contract.slack_ids non-empty (resolve_contract) + F check_escalation_orchestrator_paths.py:144 assert_connector_error_handlers(SLACK_KEY, ...) -> assert_error_handlers(): boundary errorEventDefinition on each Slack sendTask reaches a connector-free, acyclic, terminating path + F check_escalation_orchestrator_paths.py:145 assert_connector_send_identity(SLACK_KEY, "user", ...) -> assert_send_identity(): every Slack sendTask carries target=query name=send_as value=user + F check_escalation_orchestrator_paths.py:55 assert_named_equals(payload, name, expected, ...) -> public_value_present(): normalized value present among root Globals leaves + non-classifier element Outputs leaves (see NOTE below) + F check_escalation_orchestrator_paths.py:65 assert_slack_message_posted(payload, "slackMessageId", ...) -> assert_slack_posted(): fired Slack sendTask's own Outputs.response carries a ts-shaped id, the seeded channel, and correlationId + escalationPath in its text + F check_escalation_orchestrator_paths.py:78-90 completed_node_ids_of_type(payload,"script") + is_classifier -> classifier candidate set per case: a scriptTask whose OWN Outputs.response dict carries all four CLASSIFICATION_FIELDS matching expected + F check_escalation_orchestrator_paths.py:95 assert_node_type_executed(payload,"core.logic.decision") -> per-case fired_gateways non-empty + F check_escalation_orchestrator_paths.py:97-98 completed_node_ids_of_type(payload, "core.logic.decision"/"core.control.end") -> per-case fired_gateways / fired_ends via debug_data.ElementExecutions + F check_escalation_orchestrator_paths.py:132-139 common_classifier = intersection(per_case_classifiers) -> same intersection, same failure message shape + F check_escalation_orchestrator_paths.py:149-158 escalation/triage Slack-node disjointness -> same set overlap check + F check_escalation_orchestrator_paths.py:167-169 routing_decisions + assert_decision_branches_reach -> assert_decision_branches_reach(): exclusiveGateway's two outgoing sequenceFlows separate the fired Slack node sets (graph.reachable) + F check_escalation_orchestrator_paths.py:179 assert_distinct_branch_ends(escalation_nodes, triage_nodes) -> assert_distinct_branch_ends(): each fired-Slack-node set reaches its own endEvent, the other's not reachable (graph.reachable) + F check_escalation_orchestrator_paths.py:180-189 runtime escalation_ends/triage_ends disjointness -> same runtime disjointness check over per-case fired_ends + I locate/parse .bpmn, resolve project directory -> resolve_project() / resolve_contract() + I ephemeral solution init + import + sha256 pin, run bpmn debug per case, read variables-all -> LIVE-tier canonical pattern (bpmn_live.py; copied from e2e/customer_escalation_triage/check_customer_escalation_behavior.py) + T finalStatus/elementExecutions completion check (flow_check.run_debug does this inline for `flow debug`; `bpmn debug` does not) -> per-case FinalStatus check in verify_case() + T vars. / element Outputs in place of Flow's globals[".output"] -> element_output_records() / root_scope() (bpmn_live.py) + T NOTE (LIVE-ADDENDUM): a BPMN process with two end events may declare the SAME public + output name twice (once per end event, elementId-scoped per structural-bpmn.md), and + the branch that did not run has been observed to read back null even when the OTHER + branch's identically-named declaration is correctly populated. So the named-output + check searches value leaves broadly (root Globals + every OTHER element's Outputs) + instead of one pinned root-output id, per the addendum's explicit tolerance for this. + The classifier's OWN Outputs are excluded from that search (see CLASSIFICATION_FIELDS + binding above) so a value that was only ever computed, never mapped to a public output, + cannot satisfy this check by coincidence. + DROPPED Flow's exact_type check on public outputs (customer_escalation_triage-style) (not in this Flow task's grader) +""" + +from __future__ import annotations + +import json +import os +import re +import sys +import xml.etree.ElementTree as ET +from dataclasses import dataclass +from pathlib import Path + +HERE = os.path.dirname(os.path.abspath(__file__)) +sys.path.insert(0, os.path.dirname(HERE)) # tests/tasks/uipath-maestro-bpmn/ + +from _shared import graph # noqa: E402 +from _shared.bpmn_check import NS, attr, elements, resolve_project # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + BPMN_NS, + CheckFailure, + UIPATH_NS, + connector_context, + element_output_records, + get_ci, + index_runtime_connectors, + payload_data, + q, + root_scope, + run_cli, + run_debug, + sha256, +) + +SLACK_KEY = "uipath-salesforce-slack" +SLACK_CHANNEL = "C0B2FDZD1M3" # coding-agent-testing (same tenant target as the flow suite) +SLACK_SEND_AS = "user" +# Registry object name for "Send Message to Channel" is send_message_to_channel_v2 +# (registry-workflow.md L214); match loosely on the concept, as the flow suite's +# native_op_hint="send-message-to-channel" does. +SLACK_SEND_PATH_HINT = "send_message_to_channel" + +PROJECT_NAME = "CustomerEscalationOrchestrator" +BPMN_NAME = f"{PROJECT_NAME}.bpmn" + +CASE_SENSITIVE = {"caseKey"} # opaque id -- exact-case match +CLASSIFICATION_FIELDS = ("escalationPath", "severity", "engineeringNeeded", "responseMode") +NAMED_OUTPUT_FIELDS = CLASSIFICATION_FIELDS + ("caseKey",) + +LIVE_RUN_DIR = Path("escalation-orchestrator-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +DEBUG_TIMEOUT_SECONDS = 300 # literal on the run_debug call below -- the budget guard reads this statically +VARIABLES_ALL_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} +_SLACK_TS_RE = re.compile(r"^\d{9,11}\.\d{4,6}$") + + +@dataclass(frozen=True) +class Contract: + classifier_ids: tuple[str, ...] + gateway_ids: tuple[str, ...] + slack_ids: tuple[str, ...] + end_ids: tuple[str, ...] + + +def resolve_contract(path: Path) -> Contract: + root = ET.parse(path).getroot() + process = root.find(q(BPMN_NS, "process")) + if process is None: + raise CheckFailure("BPMN must contain one root process") + + classifier_ids = tuple( + el.attrib["id"] for el in process.iter(q(BPMN_NS, "scriptTask")) if el.attrib.get("id") + ) + if not classifier_ids: + raise CheckFailure( + "no bpmn:scriptTask to classify the escalation path; the prompt " + "requires ALL routing logic in one classifier scriptTask" + ) + + gateway_ids = tuple( + el.attrib["id"] for el in process.iter(q(BPMN_NS, "exclusiveGateway")) if el.attrib.get("id") + ) + if not gateway_ids: + raise CheckFailure("no bpmn:exclusiveGateway to route escalation vs triage") + + end_ids = tuple( + el.attrib["id"] for el in process.iter(q(BPMN_NS, "endEvent")) if el.attrib.get("id") + ) + if not end_ids: + raise CheckFailure("no bpmn:endEvent in the process") + + connectors = index_runtime_connectors(process) + slack_ids = tuple( + element_id + for (key, route), element_ids in connectors.items() + if key == SLACK_KEY and SLACK_SEND_PATH_HINT in route + for element_id in element_ids + ) + if not slack_ids: + raise CheckFailure( + f"no {SLACK_KEY} activity with a registry path containing " + f"{SLACK_SEND_PATH_HINT!r} -- the process must use the real Slack connector" + ) + + return Contract( + classifier_ids=classifier_ids, + gateway_ids=gateway_ids, + slack_ids=slack_ids, + end_ids=end_ids, + ) + + +def assert_send_identity(process: ET.Element, slack_ids: tuple[str, ...]) -> None: + bad = [] + for element in process.iter(): + if element.attrib.get("id") not in slack_ids: + continue + sends_as = [ + item.attrib.get("value") + for item in element.iter(q(UIPATH_NS, "input")) + if item.attrib.get("name") == "send_as" + ] + if sends_as != [SLACK_SEND_AS]: + bad.append((element.attrib.get("id"), sends_as)) + if bad: + raise CheckFailure( + f"Slack sendTask(s) do not carry exactly one send_as input with " + f"value {SLACK_SEND_AS!r}: {bad}; the prompt requires sending as the " + "requested identity" + ) + + +def assert_error_handlers(root: ET.Element, process: ET.Element, slack_ids: tuple[str, ...]) -> None: + """Every Slack sendTask's error boundary event must degrade gracefully. + + Mirrors flow_check.assert_connector_error_handlers: the handler chain must + be connector-free, acyclic, and reach a terminating node (endEvent, or a + node with no outgoing flow). + """ + + by_id = {el.attrib.get("id"): el for el in process.iter() if el.attrib.get("id")} + boundary_events = elements(root, "boundaryEvent") + outgoing: dict[str, list[str]] = {} + for source, target in graph.edges(root): + outgoing.setdefault(source, []).append(target) + + def is_connector(node_id: str) -> bool: + element = by_id.get(node_id) + return element is not None and bool(connector_context(element).get("connectorKey")) + + def reaches_terminating(start: str) -> bool: + color: dict[str, int] = {} + GRAY, BLACK = 1, 2 + + def dfs(node_id: str) -> bool: + if is_connector(node_id): + return False + color[node_id] = GRAY + outs = outgoing.get(node_id, []) + element = by_id.get(node_id) + tag = element.tag.rsplit("}", 1)[-1] if element is not None else "" + if tag == "endEvent" or not outs: + color[node_id] = BLACK + return True + for target in outs: + state = color.get(target, 0) + if state == GRAY: + return False + if state == BLACK: + continue + if not dfs(target): + return False + color[node_id] = BLACK + return True + + return dfs(start) + + bad = [] + for slack_id in slack_ids: + handlers = [ + b + for b in boundary_events + if attr(b, "attachedToRef") == slack_id + and b.find("bpmn:errorEventDefinition", NS) is not None + ] + if not handlers: + bad.append((slack_id, "no error boundary event attached")) + continue + targets = [t for b in handlers for t in outgoing.get(attr(b, "id"), [])] + if not targets or not any(reaches_terminating(t) for t in targets): + bad.append((slack_id, "error boundary event does not reach a terminating, connector-free path")) + if bad: + raise CheckFailure( + f"{SLACK_KEY} sendTask(s) do not degrade gracefully on failure: {bad}" + ) + + +def assert_decision_branches_reach( + root: ET.Element, + gateway_ids: set, + escalation_targets: set, + triage_targets: set, +) -> None: + outgoing: dict[str, list[str]] = {} + for source, target in graph.edges(root): + outgoing.setdefault(source, []).append(target) + + for gateway_id in gateway_ids: + branch_targets = outgoing.get(gateway_id, []) + reach_by_target = {t: ({t} | graph.reachable(root, t)) for t in branch_targets} + for ta, ra in reach_by_target.items(): + if not escalation_targets <= ra: + continue + for tb, rb in reach_by_target.items(): + if tb == ta: + continue + if triage_targets <= rb and not (escalation_targets & rb) and not (triage_targets & ra): + return + raise CheckFailure( + f"no exclusiveGateway routes {sorted(escalation_targets)} and " + f"{sorted(triage_targets)} through separate branches -- the distinct " + "Slack sendTasks are not the gateway's two outgoing paths" + ) + + +def assert_distinct_branch_ends( + root: ET.Element, end_ids: tuple[str, ...], branch_a_nodes: set, branch_b_nodes: set +) -> None: + def reachable_ends(nodes: set) -> set: + found: set = set() + for node in nodes: + found |= ({node} | graph.reachable(root, node)) & set(end_ids) + return found + + ends_a = reachable_ends(branch_a_nodes) + ends_b = reachable_ends(branch_b_nodes) + if not (ends_a - ends_b) or not (ends_b - ends_a): + raise CheckFailure( + "escalation and triage branches do not each reach their OWN " + f"endEvent (escalation-reachable={sorted(ends_a)}, " + f"triage-reachable={sorted(ends_b)}); the prompt requires two " + "branch-specific end events, not a single merged one" + ) + + +def normalized(value, *, case_fold: bool = True): + if isinstance(value, str): + text = value.strip() + lowered = text.casefold() + if lowered == "true": + return True + if lowered == "false": + return False + return lowered if case_fold else text + return value + + +def _loose_contains(haystack: str, needle: str) -> bool: + norm = lambda s: re.sub(r"[^a-z0-9]", "", str(s).lower()) + return norm(needle) in norm(haystack) + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def public_value_present( + variables_data, expected, *, exclude_ids: tuple[str, ...], case_sensitive: bool +) -> bool: + """Whether `expected` shows up among root Globals or a non-excluded + element's Outputs -- see the module docstring NOTE on why this is a broad + leaf search rather than one pinned root-output id.""" + + target = normalized(expected, case_fold=not case_sensitive) + candidates = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + if get_ci(element, "ElementId") in exclude_ids: + continue + candidates.extend(_leaves(get_ci(element, "Outputs", {}))) + return any(normalized(v, case_fold=not case_sensitive) == target for v in candidates) + + +def is_classifier_output(record, expected: dict) -> bool: + response = get_ci(record, "response") + if not isinstance(response, dict): + return False + return all( + normalized(get_ci(response, field)) == normalized(expected[field]) + for field in CLASSIFICATION_FIELDS + ) + + +def assert_slack_posted(variables_data, fired_slack_ids: set, case: dict) -> None: + if not fired_slack_ids: + raise CheckFailure(f"{case['name']}: no Slack sendTask executed; expected a real Slack post") + outputs = element_output_records(variables_data, tuple(fired_slack_ids)) + matched = None + for output in outputs: + response = get_ci(output, "response") + ts = get_ci(response, "ts") + if isinstance(ts, str) and _SLACK_TS_RE.match(ts.strip()): + matched = response + break + if matched is None: + raise CheckFailure( + f"{case['name']}: no executed Slack sendTask's response carries a " + "message ts; the process did not actually post to Slack" + ) + channel = get_ci(matched, "channel") + if channel != SLACK_CHANNEL: + raise CheckFailure( + f"{case['name']}: Slack message posted to channel {channel!r}, " + f"expected {SLACK_CHANNEL!r}" + ) + message = get_ci(matched, "message") + text = get_ci(message, "text") if isinstance(message, dict) else None + correlation_id = case["inputs"]["correlationId"] + if not isinstance(text, str) or correlation_id not in text: + raise CheckFailure( + f"{case['name']}: Slack message text does not carry correlationId " + f"{correlation_id!r}: {text!r}" + ) + if not _loose_contains(text, case["expected"]["escalationPath"]): + raise CheckFailure( + f"{case['name']}: Slack message text does not carry escalationPath " + f"{case['expected']['escalationPath']!r}: {text!r}" + ) + + +def verify_case(contract: Contract, imported_project: Path, case: dict) -> dict: + log_file = LIVE_RUN_DIR / f"debug-{case['name']}.log" + # DEBUG_TIMEOUT_SECONDS is inlined below (not passed by name) so the + # static budget guard (test_criterion_budgets.py) can read the literal. + debug_data, _instance_id = run_debug( + imported_project, case["inputs"], log_file, timeout=300 + ) + + final_status = get_ci(debug_data, "FinalStatus") + executions = get_ci(debug_data, "ElementExecutions", []) or [] + if final_status not in COMPLETED_STATUSES: + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in executions + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + raise CheckFailure( + f"{case['name']}: final status was {final_status!r}" + + (f"; non-completed elements: {faulted}" if faulted else "") + ) + + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", _instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, f"{case['name']}: variables-all") + + completed_ids = { + get_ci(item, "ElementId") + for item in executions + if isinstance(item, dict) and str(get_ci(item, "Status") or "").casefold() == "completed" + } + fired_slack = completed_ids & set(contract.slack_ids) + fired_gateways = completed_ids & set(contract.gateway_ids) + fired_ends = completed_ids & set(contract.end_ids) + + if not fired_gateways: + raise CheckFailure(f"{case['name']}: no exclusiveGateway executed in the debug trace") + + classifier_candidates = { + node_id + for node_id in contract.classifier_ids + if any( + is_classifier_output(record, case["expected"]) + for record in element_output_records(variables_data, (node_id,)) + ) + } + if not classifier_candidates: + raise CheckFailure( + f"{case['name']}: no single executed scriptTask computed all of " + f"{CLASSIFICATION_FIELDS} together -- the prompt requires ALL " + "routing logic in ONE scriptTask that returns every field" + ) + + for field in NAMED_OUTPUT_FIELDS: + expected_value = case["expected"][field] + if not public_value_present( + variables_data, + expected_value, + exclude_ids=contract.classifier_ids, + case_sensitive=field in CASE_SENSITIVE, + ): + raise CheckFailure(f"{case['name']}: no public output carries {field}={expected_value!r}") + + if case.get("expect_slack"): + assert_slack_posted(variables_data, fired_slack, case) + + print( + f"OK: {case['name']} produced the expected outcome" + + (" + Slack message posted" if case.get("expect_slack") else "") + ) + return { + "fired_slack": fired_slack, + "fired_gateways": fired_gateways, + "fired_ends": fired_ends, + "classifier_candidates": classifier_candidates, + } + + +def main() -> None: + seed_path = Path("seed.json") + if not seed_path.is_file(): + raise CheckFailure("seed.json is missing; pre_run did not complete") + seed = json.loads(seed_path.read_text(encoding="utf-8")) + cases = seed.get("cases") + if not isinstance(cases, list) or not cases: + raise CheckFailure("seed.json must contain at least one case") + + project_dir = resolve_project(BPMN_NAME) + bpmn_path = project_dir / BPMN_NAME + contract = resolve_contract(bpmn_path) + root = ET.parse(bpmn_path).getroot() + process = root.find(q(BPMN_NS, "process")) + + # The escalation alert must go through the real Slack connector, wired for + # graceful degradation and the requested send identity -- static checks, + # before spending any live budget. + assert_send_identity(process, contract.slack_ids) + assert_error_handlers(root, process, contract.slack_ids) + + original_hash = sha256(bpmn_path) + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "EscalationOrchestratorLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / BPMN_NAME) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + escalation_fired: set = set() + triage_fired: set = set() + escalation_ends: set = set() + triage_ends: set = set() + per_case_gateways: list = [] + per_case_classifiers: list = [] + + for case in cases: + result = verify_case(contract, imported_project, case) + is_escalation = case["expected"]["escalationPath"] == "escalation" + (escalation_fired if is_escalation else triage_fired).update(result["fired_slack"]) + (escalation_ends if is_escalation else triage_ends).update(result["fired_ends"]) + per_case_gateways.append(result["fired_gateways"]) + per_case_classifiers.append(result["classifier_candidates"]) + + # ONE classifier scriptTask must classify EVERY case (prompt: ALL routing + # logic in one scriptTask). + common_classifier = set.intersection(*per_case_classifiers) if per_case_classifiers else set() + if not common_classifier: + raise CheckFailure( + "no single scriptTask classified every case -- the prompt requires " + "ALL routing logic in ONE scriptTask, but the cases were classified " + "by different (path-specific) scriptTasks (per-case candidates: " + f"{[sorted(c) for c in per_case_classifiers]})" + ) + print(f"OK: one classifier scriptTask handled all cases: {sorted(common_classifier)}") + + # The gateway must genuinely branch: escalation and triage post via + # DIFFERENT Slack sendTasks. + if not escalation_fired or not triage_fired: + raise CheckFailure( + "expected both escalation and triage cases to fire a Slack " + f"sendTask (escalation={escalation_fired}, triage={triage_fired})" + ) + overlap = escalation_fired & triage_fired + if overlap: + raise CheckFailure( + f"escalation and triage cases fired the SAME Slack sendTask(s) " + f"{overlap} -- the gateway does not route to two distinct branches" + ) + + # And prove those two sendTasks are the gateway's OWN outgoing branches, + # routed by a gateway that executed in EVERY case. + routing_gateways = set.intersection(*per_case_gateways) if per_case_gateways else set() + if not routing_gateways: + raise CheckFailure( + "no EXECUTED exclusiveGateway in every run (executed sets: " + f"{[sorted(g) for g in per_case_gateways]}); routing did not go " + "through a gateway that actually ran every time" + ) + assert_decision_branches_reach(root, routing_gateways, escalation_fired, triage_fired) + + # The prompt requires TWO end events -- one per branch. Static reachability + # closes the "unused private end event + shared merged end" gaming; the + # runtime disjointness below closes the "both real paths merge into one + # shared end event" gaming. + assert_distinct_branch_ends(root, contract.end_ids, escalation_fired, triage_fired) + if not escalation_ends or not triage_ends: + raise CheckFailure( + "expected both escalation and triage cases to complete an " + f"endEvent (escalation_ends={sorted(escalation_ends)}, " + f"triage_ends={sorted(triage_ends)})" + ) + shared_ends = escalation_ends & triage_ends + if shared_ends: + raise CheckFailure( + f"escalation and triage cases completed the SAME endEvent(s) " + f"{sorted(shared_ends)} -- both branches merge into one shared " + "end event; the prompt requires a distinct end event per branch" + ) + print( + "OK: branches complete distinct end events " + f"(escalation={sorted(escalation_ends)}, triage={sorted(triage_ends)})" + ) + print( + "OK: an exclusiveGateway routes escalation vs triage through separate " + f"branches (escalation={sorted(escalation_fired)}, triage={sorted(triage_fired)})" + ) + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py new file mode 100644 index 0000000000..aefec7855c --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py @@ -0,0 +1,459 @@ +#!/usr/bin/env python3 +"""EscalationSlackAlert (BPMN): live outcome check, seeded Sev1 case. + +Ported from Flow `e2e/escalation_slack_alert/check_escalation_slack_alert.py`: +same scenario (a manual-start escalation-triage process classifies severity +and posts a Slack alert), same seeded Sev1 case, same OUTCOME assertions — +translated from `uip maestro flow debug`'s inline variables to `uip maestro +bpmn debug`'s instance-id + `debug-instance variables-all`/`incidents` +surface (canonical live pattern, see LIVE-ADDENDUM.md; copies the plumbing of +the CI-proven `e2e/customer_escalation_triage/check_customer_escalation_behavior.py`, +trimmed to this task's single Slack connector — no Jira, no tenant re-read, no +teardown, because Flow's own checker never re-reads the tenant or deletes the +posted message and this task's Flow `post_run` never sweeps Slack either). + +Assertion map (Flow → BPMN): + F check_escalation_slack_alert.py:44 assert_flow_uses_connector_target(SLACK_KEY) -> resolve_contract(): ids_for(*SLACK_SEND) via index_runtime_connectors + F check_escalation_slack_alert.py:45 assert_connector_send_identity(key, "user", ...) -> resolve_contract(): send_as input == "user" on every slack_send_id + F check_escalation_slack_alert.py:58 run_debug(inputs=case["inputs"], retries=1) -> ephemeral `solution init` + `solution projects import` (sha256-pinned) + bpmn_live.run_debug(project_dir, inputs, log) + F flow_check.py:551-552 (run_debug's own exact-match status gate) -> assert_outcome(): FinalStatus == "Completed" + F check_escalation_slack_alert.py:63-64 assert_named_equals(payload, name, expected) -> assert_outcome(): assert_named_equals(actual, name, expected) per output (severity, engineeringNeeded, caseKey) + F flow_check.py:1445-1464 completed_node_ids_of_type(payload, "script") -> resolve_contract(): classifier_ids = every bpmn:scriptTask id + F check_escalation_slack_alert.py:66-88 sev_scripts / next_steps binding (same node, both fields) -> bind_classifier(): severity AND nextSteps must come from the SAME scriptTask's own output + F flow_check.py:1328-1334 slackMessageId ts-shape gate -> assert_outcome(): SLACK_TS_RE match on the exposed output + F flow_check.py:1339-1355 >=1 Completed connector node in the debug trace -> assert_outcome(): >=1 "completed" element among contract.slack_send_ids + F flow_check.py:1357-1384 mapped id must equal an executed send's OWN response ts -> assert_outcome(): match slackMessageId against a slack_send_ids element's response["ts"] + F flow_check.py:1386-1394 posted channel must equal expected_channel -> assert_outcome(): matched response["channel"] == SLACK_CHANNEL + F flow_check.py:1395-1405 must_contain: correlationId, severity, nextSteps -> assert_outcome(): matched response["message"]["text"] contains all three + I locate/parse .bpmn; resolve the project dir; import into an ephemeral solution (sha256-pinned); + read runtime evidence via `debug-instance variables-all`/`incidents` — `bpmn debug` returns an + instance id rather than inline variables, so this whole sequence stands in for Flow's single + `flow debug` call (LIVE-ADDENDUM.md "canonical live pattern") + T public outputs addressed by declared `` id (`vars.`) and a node's own + `Outputs.response` in place of Flow's globals[".output"] map + DROPPED exact-type checks on public outputs (customer_escalation_triage's own addition; Flow's + assert_named_equals never asserts a Python type, only non-empty + normalized equality) + DROPPED "Successful" as an alternate FinalStatus / tenant re-read / connection lookup / journal + teardown (Flow's checker only ever accepts exact "Completed" and never re-reads the tenant + or deletes the posted message; this task's Flow post_run has no Slack teardown either) +""" + +from __future__ import annotations + +import json +import os +import re +import sys +import xml.etree.ElementTree as ET +from dataclasses import dataclass +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared import bpmn_check # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + BPMN_NS, + CheckFailure, + element_output_records, + get_ci, + incident_records, + index_runtime_connectors, + payload_data, + q, + resolve_runtime_key, + root_scope, + run_cli, + run_debug, + sha256, + UIPATH_NS, +) + +SLACK_CONNECTOR = "uipath-salesforce-slack" +# Route substring, not the curated activity's exact spelling: registry +# templates may emit a versioned path (`/send_message_to_channel_v2`), and +# `index_runtime_connectors` correlates on a substring match the same way the +# CI-proven customer_escalation_triage checker does. +SLACK_SEND = (SLACK_CONNECTOR, "send_message_to_channel") +SLACK_SEND_AS = "user" +SLACK_CHANNEL = "C0B2FDZD1M3" # coding-agent-testing +OUTPUT_NAMES = ("severity", "engineeringNeeded", "caseKey", "slackMessageId") +CASE_SENSITIVE_OUTPUTS = {"caseKey"} # opaque id -- exact-case match, like Flow's CASE_SENSITIVE +SLACK_TS_RE = re.compile(r"^\d{9,11}\.\d{4,6}$") + +LIVE_RUN_DIR = Path("escalation-slack-alert-live") +# Per-call budgets for the live CLI steps this checker makes. test_criterion_budgets.py +# statically prices every run_debug(...) call; the YAML criterion's `timeout:` must +# cover their sum plus bpmn_live.CRITERION_MARGIN_SECONDS. +DEBUG_TIMEOUT_SECONDS = 480 # bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT, spelled out so the +# static budget guard can price this call without following the import. +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +STEP_TIMEOUTS = ( + DEBUG_TIMEOUT_SECONDS, + SOLUTION_INIT_TIMEOUT, + SOLUTION_IMPORT_TIMEOUT, + VARIABLES_ALL_TIMEOUT, + INCIDENTS_TIMEOUT, +) + + +@dataclass(frozen=True) +class Contract: + """Element and variable ids needed to read runtime outcomes. + + Discovery, not grading: BPMN runtime variables are addressed by id, so the + ids of the public outputs, the classifier script(s), and the Slack send + node(s) must be resolved from the source before the live evidence can be + read. Mirrors customer_escalation_triage's Contract, trimmed to one + connector. + """ + + output_ids: dict[str, str] + slack_send_ids: tuple[str, ...] + classifier_ids: tuple[str, ...] + + +def resolve_contract(root: ET.Element) -> Contract: + process = root.find(q(BPMN_NS, "process")) + if process is None: + raise CheckFailure("BPMN must contain one root process") + + variables = process.find( + f"./{q(BPMN_NS, 'extensionElements')}/{q(UIPATH_NS, 'variables')}" + ) + if variables is None: + raise CheckFailure("root process is missing uipath:variables") + output_ids: dict[str, str] = {} + for variable in variables: + name = variable.attrib.get("name") + identifier = variable.attrib.get("id") + if ( + variable.tag.rsplit("}", 1)[-1] != "output" + or not name + or not identifier + or name not in OUTPUT_NAMES + ): + continue + if name in output_ids: + raise CheckFailure( + f"public output {name!r} is declared more than once, so its " + "runtime value cannot be addressed" + ) + output_ids[name] = identifier + missing = sorted(set(OUTPUT_NAMES) - set(output_ids)) + if missing: + raise CheckFailure(f"public outputs not declared: {missing}") + + # Every scriptTask that could be the classifier -- bind_classifier() below + # requires severity AND nextSteps to come from the SAME executed one. + classifier_ids = tuple( + element.attrib["id"] + for element in process.iter(q(BPMN_NS, "scriptTask")) + if element.attrib.get("id") + ) + if not classifier_ids: + raise CheckFailure( + "no bpmn:scriptTask to classify severity; the value would be a " + "literal rather than computed" + ) + + connectors = index_runtime_connectors(process) + + def ids_for(connector_key: str, path_needle: str) -> tuple[str, ...]: + found = tuple( + element_id + for (key, route), element_ids in connectors.items() + if key == connector_key and path_needle in route + for element_id in element_ids + ) + if not found: + raise CheckFailure( + f"no {connector_key} activity with a registry path " + f"containing {path_needle!r}" + ) + return found + + # Send identity, as Flow grades it: a node that posts as the default bot + # instead of the prompt-required user has an indistinguishable runtime + # response, so it can only be caught on the authored artifact. + slack_ids = set(ids_for(*SLACK_SEND)) + for element in process.iter(): + if element.attrib.get("id") not in slack_ids: + continue + sends_as = [ + item.attrib.get("value") + for item in element.iter(q(UIPATH_NS, "input")) + if item.attrib.get("name") == "send_as" + ] + if sends_as != [SLACK_SEND_AS]: + raise CheckFailure( + f"Slack node {element.attrib.get('id')!r} must carry exactly " + f"one send_as input with value {SLACK_SEND_AS!r}, found " + f"{sends_as!r}" + ) + + return Contract( + output_ids=output_ids, + slack_send_ids=tuple(sorted(slack_ids)), + classifier_ids=classifier_ids, + ) + + +def normalized(value, *, case_fold: bool = True): + """Port of flow_check.normalized: trim strings, coerce "true"/"false" to + booleans, and (by default) fold case for enum-like values.""" + if isinstance(value, str): + text = value.strip() + lowered = text.casefold() + if lowered == "true": + return True + if lowered == "false": + return False + return lowered if case_fold else text + return value + + +def assert_named_equals( + actual_map: dict, name: str, expected, *, case_sensitive: bool = False +) -> None: + """Port of flow_check.assert_named_equals: the named output must be + present, non-empty, and equal `expected` (case-insensitively unless + `case_sensitive`).""" + if name not in actual_map: + raise CheckFailure(f"output {name!r} is missing from the runtime scope") + actual = actual_map[name] + if actual is None or (isinstance(actual, str) and not actual.strip()): + raise CheckFailure(f"output {name!r} is empty") + if normalized(actual, case_fold=not case_sensitive) != normalized( + expected, case_fold=not case_sensitive + ): + raise CheckFailure(f"output {name!r}: expected {expected!r}, got {actual!r}") + + +def bind_classifier(variables_data, classifier_ids: tuple[str, ...], expected_severity: str) -> str: + """Bind severity AND nextSteps to the SAME executed scriptTask. + + Port of check_escalation_slack_alert.py's script_nodes/sev_scripts/ + next_steps logic: the prompt requires the alert to include severity, + correlationId, and next steps, and nextSteps is an intermediate script + output (not a named End/public output). Binding both to one node's own + response means an unrelated or cosmetic scriptTask can't supply either + value. + """ + records = element_output_records(variables_data, classifier_ids) + sev_responses = [] + for record in records: + response = get_ci(record, "response") + severity_value = get_ci(response, "severity") if isinstance(response, dict) else response + if severity_value is not None and normalized(severity_value) == normalized(expected_severity): + sev_responses.append(response) + if not sev_responses: + raise CheckFailure( + f"no executed bpmn:scriptTask among {list(classifier_ids)} produced " + f"the expected severity {expected_severity!r} in its own output -- " + "cannot bind the nextSteps check to the classification script" + ) + for response in sev_responses: + if isinstance(response, dict): + candidate = get_ci(response, "nextSteps") + if isinstance(candidate, str) and candidate.strip(): + return candidate.strip() + raise CheckFailure( + "the bpmn:scriptTask that produced the expected severity did not also " + "produce a nextSteps value in its own output -- the prompt requires " + "classifying a short next-steps string" + ) + + +def assert_outcome( + contract: Contract, + case: dict, + debug_data, + variables_data, + incidents_data, +) -> str: + """Assert the runtime evidence for the seeded case; return the Slack ts.""" + + final_status = get_ci(debug_data, "FinalStatus") + if final_status != "Completed": + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + detail = [f"non-completed elements: {faulted}"] if faulted else [] + records = incident_records(incidents_data) + if records: + detail.append(f"incidents: {json.dumps(records)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + + # `bpmn debug` reports success via an instance id, not inline variables; + # a completed run with incidents is still a failure (LIVE-ADDENDUM.md). + incidents = incident_records(incidents_data) + if incidents is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents: + raise CheckFailure(f"unexpected incidents: {incidents}") + + executions = get_ci(debug_data, "ElementExecutions", []) + slack_completed = [ + item + for item in executions + if isinstance(item, dict) + and get_ci(item, "ElementId") in contract.slack_send_ids + and str(get_ci(item, "Status") or "").casefold() == "completed" + ] + if not slack_completed: + raise CheckFailure( + f"no Slack send element among {list(contract.slack_send_ids)} " + "completed in the debug trace; the alert was not actually posted" + ) + + globals_map = get_ci(root_scope(variables_data), "Globals", {}) + if not isinstance(globals_map, dict): + raise CheckFailure(f"root scope Globals is not a map: {globals_map!r}") + actual = { + name: resolve_runtime_key(globals_map, identifier, name) + for name, identifier in contract.output_ids.items() + } + + expected = case["expected"] + assert_named_equals(actual, "severity", expected["severity"]) + assert_named_equals(actual, "engineeringNeeded", expected["engineeringNeeded"]) + assert_named_equals(actual, "caseKey", expected["caseKey"], case_sensitive=True) + + # Severity must be COMPUTED, not exposed as a literal, and nextSteps (not + # a public output) must come from the same executed classification node. + next_steps = bind_classifier(variables_data, contract.classifier_ids, expected["severity"]) + + # Shape gate: reject a hard-coded placeholder like "ok"/"sent"/"1". + slack_message_id = actual.get("slackMessageId") + text_id = str(slack_message_id).strip() if slack_message_id is not None else "" + if not SLACK_TS_RE.match(text_id): + raise CheckFailure( + f"output 'slackMessageId'={text_id!r} is not a Slack message ts " + r"(expected \d{9,11}\.\d{4,6}); the process did not actually post to Slack" + ) + + # Trace gate: the mapped ts must equal an EXECUTED Slack node's own + # response ts -- a constant ts mapped past an idle/unrelated node fails. + slack_outputs = element_output_records(variables_data, contract.slack_send_ids) + matched_response = None + for output in slack_outputs: + response = get_ci(output, "response") + if not isinstance(response, dict): + continue + ts = get_ci(response, "ts") + if isinstance(ts, str) and ts.strip() == text_id: + matched_response = response + break + if matched_response is None: + raise CheckFailure( + f"slackMessageId {text_id!r} does not match any executed Slack " + "node's response ts; the mapped ts was not produced by the " + "executed send" + ) + + channel = get_ci(matched_response, "channel") + if channel != SLACK_CHANNEL: + raise CheckFailure( + f"Slack message posted to channel {channel!r}, expected {SLACK_CHANNEL!r}" + ) + + message = get_ci(matched_response, "message") + text = get_ci(message, "text") if isinstance(message, dict) else None + required_tokens = (case["inputs"]["correlationId"], expected["severity"], next_steps) + if not isinstance(text, str) or any(token not in text for token in required_tokens): + raise CheckFailure( + "posted Slack message is missing required text -- the message " + f"must carry every required field (severity, correlationId, next " + f"steps), not just some: {text!r}" + ) + return text_id + + +def main() -> None: + bpmn_path, root = bpmn_check.parse_bpmn("EscalationSlackAlert") + project_dir = bpmn_check.resolve_project(os.path.basename(bpmn_path)) + # Artifact-level gates first (connector target + send identity + declared + # outputs), mirroring Flow's assert_flow_uses_connector_target / + # assert_connector_send_identity running before the live debug call. + contract = resolve_contract(root) + + seed_path = Path("seed.json") + if not seed_path.is_file(): + raise CheckFailure("seed.json is missing; pre_run did not complete") + cases = json.loads(seed_path.read_text(encoding="utf-8")).get("cases") + if not isinstance(cases, list) or len(cases) != 1: + raise CheckFailure("seed.json must contain exactly one case") + case = cases[0] + + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "EscalationSlackAlertLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], + timeout=SOLUTION_INIT_TIMEOUT, + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / Path(bpmn_path).name) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = run_debug( + imported_project, case["inputs"], LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + + ts = assert_outcome(contract, case, debug_data, variables_data, incidents_data) + print( + f"OK: {case['name']} completed -- Sev1 + engineering classified, " + f"correlationId preserved, and the Slack alert was posted (ts={ts})" + ) + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py new file mode 100644 index 0000000000..d379ef2c3a --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py @@ -0,0 +1,340 @@ +#!/usr/bin/env python3 +"""AcrUserList generic dynamic node (BPMN): structural + live checks. + +Ported from Flow `connector_features/generic_dynamic_node/_shared/check_generic_dynamic_node.py`: +same scenario (a manual-start process calls ServiceNow's generic, object-agnostic +"List Records" activity — display name "List All Records" — on the dynamically +resolved `acr_user` object, then surfaces the result as a process output), +translated from a JSON node walk + inline `flow debug` payload to an XML walk +over the registry-driven `Intsvc.ActivityExecution` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3) plus the BPMN +live-debug surface (`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical +pattern: ephemeral solution import, `bpmn debug`, `debug-instance +variables-all`/`incidents`, cloned from `check_jira_get_issue.py`). + +A generic activity encodes only the operation (List) in its registry type; the +object (`acr_user`) is supplied dynamically at authoring time via +`uip maestro bpmn registry get Intsvc.ActivityExecution --connection-id +--object-name acr_user` — the BPMN analog of Flow's "set the object name on +the node". Confirmed live against this tenant's ServiceNow catalog +(`uip is activities list uipath-servicenow-servicenow --output json`): +`ListAllRecords` / "List Records" is `IsCurated: No`, `ObjectName: N/A`, +`Operation: List` — i.e. the generic (non-curated) form is the ONLY form for +this activity, unlike the curated-or-generic ambiguity BATCH1-ADDENDUM +describes for Data Service/Test Manager. The enrichment call above (run +read-only against the live tenant while authoring this port) confirmed the +resulting context fields: `objectName=acr_user`, `operation=List`, +`method=GET`, `path=/acr_user`. + +The `acr_user` table is empty in the codereval ServiceNow tenant, so the list +call legitimately returns `[]`. The runtime assertion therefore checks that +the connector completed with no incidents and surfaced an array-typed output +(empty allowed) — not that rows came back. + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. + +Assertion map (Flow → BPMN): + F check_generic_dynamic_node.py:128-146 assert_flow_uses_connector_target(CONNECTOR_KEY) + → CONNECTOR_KEY substring present in raw .bpmn text (fast pre-check) + + a node's connector_context()["connectorKey"] == CONNECTOR_KEY + F check_generic_dynamic_node.py:108-125 _is_generic_list(): Generic activityType + `list` operation + → is_generic_list_node(): a node carrying Intsvc.ActivityExecution + whose connectorKey is uipath-servicenow-servicenow and whose + objectName == "acr_user" with operation == "list" or method == "GET" + (registry-confirmed above: the ONLY form ListAllRecords takes) + F check_generic_dynamic_node.py:148-164 objectName resolved to OBJECT_NAME ("acr_user") + → objectName context field == "acr_user" (folded into is_generic_list_node) + F check_generic_dynamic_node.py:203-205 run_debug(...) implicitly requires finalStatus == "Completed" + (flow_check.run_debug raises on a non-Completed status internally; + `bpmn debug` returns only an instance id, so the check is explicit here) + → FinalStatus in COMPLETED_STATUSES and debug-instance incidents is empty + F check_generic_dynamic_node.py:167-200 _assert_array_output(): an array-typed global output (empty allowed), + reporting a flattened sys_id-bearing record if present + → array-typed value among the root scope's Globals AND every element's + Outputs (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read back null even + when correctly mapped, so the search is not scoped to one declared + output variable — mirrors check_jira_get_issue.collect_output_haystack) + I locate/parse .bpmn, resolve project directory + → bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported bytes + against the submitted file — `bpmn debug` runs against an imported project, unlike + `flow debug`, which runs directly against the discovered project directory + → LIVE-ADDENDUM canonical live pattern (mirrors check_jira_get_issue.py) + T Generic (non-curated) object CRUD classification (BATCH1-ADDENDUM): objectName == entity, + method GET (or operation "list") in place of a curated node-type slug + → is_generic_list_node() + DROPPED OPERATION_SLUGS node-type-slug matching (Flow's `…list-records` / `…list-all-records` + substring fallback) — BPMN's registry type is always the fixed literal + `Intsvc.ActivityExecution`; there is no per-operation node-type slug to match against, so + the slug-matching branch of Flow's check drops entirely rather than translating. The + dynamically-resolved `objectName` context field (Flow's own preferred, slug-agnostic signal) + carries the whole check here. + +No tenant re-read beyond the process's own debug run is performed, matching Flow's own grader (a read-only +list call has nothing else to verify against). +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +CONNECTOR_KEY = "uipath-servicenow-servicenow" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +# API object name (not the "Acr User" display name) — the registry's +# `--object-name` enrichment takes the connector's case-sensitive `Name`, +# which for this table is `acr_user`. +OBJECT_NAME = "acr_user" +GENERIC_LIST_OPERATIONS = {"list"} +GENERIC_LIST_METHODS = {"GET"} +NAME_HINT = "AcrUserList" + +LIVE_RUN_DIR = Path("acr-user-list-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here, exactly +# as check_jira_get_issue.py documents: +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (720) does not cover this arithmetic — a +# single flow_check.run_debug(timeout=300) call has no separate solution +# init/import/variables-all/incidents steps — so the BPMN criterion timeout +# in generic_dynamic_node.yaml is raised to 1050 and documented there as the +# one sanctioned deviation from "criteria identical" (LIVE-ADDENDUM: the +# budget is a property of the CLI surface, not of what is graded). + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def is_generic_list_node(context: dict) -> bool: + """Generic (non-curated) list activity: objectName == acr_user, list op. + + Registry-confirmed (see module docstring): ListAllRecords has no curated + alternative for this connector, so objectName + operation/method is the + whole signal — there is no node-type slug in BPMN to fall back on.""" + object_name = (context.get("objectName") or "").strip().lower() + if object_name != OBJECT_NAME: + return False + operation = (context.get("operation") or "").strip().lower() + method = (context.get("method") or "").strip().upper() + return operation in GENERIC_LIST_OPERATIONS or method in GENERIC_LIST_METHODS + + +def find_generic_list_nodes(root: ET.Element) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution generic list op. + + Scans every descendant, not a fixed tag list (registry templates may emit + a connector activity as sendTask, serviceTask, or a plain task) — mirrors + bpmn_live.index_runtime_connectors' own scanning discipline. + """ + found = [] + for node in root.iter(): + context = connector_context(node) + if context.get("connectorKey") != CONNECTOR_KEY: + continue + if ACTIVITY_TYPE not in ET.tostring(node, encoding="unicode"): + continue + if is_generic_list_node(context): + found.append(node) + return found + + +def _array_leaves(value): + """Yield every list found in ``value``, recursing through dict values.""" + if isinstance(value, list): + yield value + elif isinstance(value, dict): + for v in value.values(): + yield from _array_leaves(v) + + +def collect_array_candidates(variables_data: object) -> list[tuple[str, list]]: + """Array-typed values among the root scope's Globals AND every element's + Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable — mirrors check_jira_get_issue.py's + collect_output_haystack, adapted to look for an array shape rather than a + substring. + """ + candidates: list[tuple[str, list]] = [] + globals_ = get_ci(root_scope(variables_data), "Globals", {}) or {} + if isinstance(globals_, dict): + for name, value in globals_.items(): + for array in _array_leaves(value): + candidates.append((f"global {name!r}", array)) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + outputs = get_ci(element, "Outputs", {}) + for array in _array_leaves(outputs): + candidates.append( + (f"element {get_ci(element, 'ElementId')!r} Outputs", array) + ) + return candidates + + +def main() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + raw = Path(bpmn_path).read_text(encoding="utf-8") + if CONNECTOR_KEY not in raw: + _fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") + if OBJECT_NAME not in raw: + _fail(f"{bpmn_path} does not reference the object {OBJECT_NAME!r}") + print(f"OK: bpmn references {CONNECTOR_KEY} and object {OBJECT_NAME!r}") + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + list_nodes = find_generic_list_nodes(root) + if not list_nodes: + seen = sorted( + { + json.dumps(connector_context(node), sort_keys=True) + for node in root.iter() + if connector_context(node).get("connectorKey") == CONNECTOR_KEY + } + ) + _fail( + f"No generic ServiceNow list activity found on the {CONNECTOR_KEY} " + f"connector with objectName={OBJECT_NAME!r} (expected operation " + f"'list' or method 'GET'). Connector node contexts seen: {seen}" + ) + print(f"OK: found {len(list_nodes)} generic list node(s) on objectName={OBJECT_NAME!r}") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "AcrUserListLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + candidates = collect_array_candidates(variables_data) + if not candidates: + _fail( + "No output variable holds an array — the connector result was not " + "surfaced as a process output. Checked root Globals and every " + "element's Outputs." + ) + label, value = candidates[0] + if value and all(isinstance(r, dict) for r in value) and any("sys_id" in r for r in value): + print( + f"OK: connector returned {len(value)} Acr User record(s) in {label} " + f"(first sys_id={value[0].get('sys_id')!r})" + ) + else: + print( + f"OK: connector completed and surfaced an array output in {label} " + f"({len(value)} record(s)). Empty is expected — the acr_user table " + "is empty in this tenant." + ) + print("PASS: all AcrUserList generic dynamic node checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py new file mode 100644 index 0000000000..891b58a588 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py @@ -0,0 +1,363 @@ +#!/usr/bin/env python3 +"""JiraCreateIssue (BPMN): structural + live + tenant checks. + +Ported from Flow `e2e/jira_create_issue/_shared/check_jira_create_issue.py`: +same scenario (a manual-start process creates one Jira issue via the +Atlassian Jira "Create Issue" connector activity using seeded field values, +then exposes the new issue's key), translated from a JSON node walk + inline +`flow debug` payload to an XML walk over the registry-driven +`Intsvc.ActivityExecution` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) plus the +BPMN live-debug surface (`_shared/bpmn_live.py`, per LIVE-ADDENDUM's +canonical pattern: ephemeral solution import, `bpmn debug`, `debug-instance +variables-all`/`incidents`). + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. The confirmed key(s) are journaled to ``.created_keys`` so post_run +teardown (`teardown_jira.py`) deletes them even if a later step fails. + +Assertion map (Flow -> BPMN): + F check_jira_create_issue.py:49 JIRA_KEY not in raw or '"nodes"' not in raw + ('"nodes"' marker dropped -- XML has no JSON "nodes" key, see I below) + -> JIRA_KEY not in raw text of the .bpmn + T curated OR generic entity-CRUD classification (BATCH1-ADDENDUM); + catalog checked via `uip is activities list uipath-atlassian-jira + --output json` (CreateIssue -> objectName curated_create_issue, + method POST; generic form mirrors sibling script + check_jira_get_issue.py's GENERIC_OBJECT="issue" convention) + -> find_create_issue_nodes(): a sendTask carrying + Intsvc.ActivityExecution whose connectorKey is + uipath-atlassian-jira and whose objectName/method + classify as Create Issue + ADDED (T) not gated by Flow's own create script (which relies solely on the + tenant re-read below); translates the PROMPT requirement shared by + both suites ("use project_key/issuetype_id/summary/reporter_id + exactly as given") using the same literal-value technique as sibling + script check_jira_get_issue.py's F-tagged `issue_key not in raw` + check (its line 48) -- fails fast on an invented value before + spending the live-debug budget, without narrowing anything Flow's + own create script accepts (the tenant re-read below still gates) + -> each of seed["project_key"], seed["issuetype_id"], + seed["summary"], seed["reporter_id"] found + literally, anywhere in the raw .bpmn text + F check_jira_create_issue.py:53 run_debug(timeout=480) implicitly requires + finalStatus == "Completed" (flow_check.run_debug + raises on a non-Completed status internally; `bpmn + debug` returns only an instance id, so the check is + explicit here) + -> FinalStatus in COMPLETED_STATUSES and + debug-instance incidents is empty + F check_jira_create_issue.py:58-61 candidate issue keys: clean output leaves + (`collect_outputs(payload)`) + a project-scoped + regex scan of the raw debug payload + (`get_last_debug_raw()`) + -> collect_candidate_keys(): value leaves of the + root scope's Globals AND every element's Outputs + in `debug-instance variables-all`, plus the same + project-scoped regex scan run over the raw + variables-all response text (the nearest BPMN + analog of Flow's "everything the debug call + returned" raw text) + F check_jira_create_issue.py:62-63 no candidate keys -> fail + -> same + F check_jira_create_issue.py:66-77 tenant re-read via jira_is.get_issue(conn, key); + first candidate whose `summary` equals the seed + summary wins; confirmed key journaled for teardown + -> same logic, unchanged jira_is.py (task's own + `_setup/jira_is.py` copy, imported via the + sandbox-mounted path since this checker lives in + `_shared/`, not the task dir) + I locate/parse .bpmn (file exists, well-formed XML, project directory + resolved); ephemeral solution init + `solution projects import` + + sha256 pin of the imported bytes against the submitted file -- + `bpmn debug` runs against an imported project, unlike `flow debug`, + which runs directly against the discovered project directory + -> LIVE-ADDENDUM canonical live pattern (mirrors + e2e/customer_escalation_triage/ + check_customer_escalation_behavior.py and + check_jira_get_issue.py) + T journal every candidate key BEFORE the tenant-confirmation loop, + not only the one that matches (LIVE-ADDENDUM: "side-effect ids go + to a flat journal the moment they are visible") -- Flow's own + script only journals the confirmed match, but a decoy issue + created from a wrong body is still a real tenant record that must + not leak just because its summary didn't match + -> `.created_keys` written right after candidate + collection, one key per line + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- + not in Flow; the `bpmn validate` criterion covers structure +""" + +from __future__ import annotations + +import json +import os +import re +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# …/uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +JIRA_KEY = "uipath-atlassian-jira" +CREATE_OP_RE = re.compile(r"create[\s_-]?issue|curated_create_issue", re.IGNORECASE) +GENERIC_OBJECT = "issue" +GENERIC_CREATE_METHODS = {"POST"} +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +NAME_HINT = "JiraCreateIssue" +ISSUE_KEY_RE = re.compile(r"^[A-Z][A-Z0-9]*-\d+$") +SEED_LITERAL_FIELDS = ("project_key", "issuetype_id", "summary", "reporter_id") + +LIVE_RUN_DIR = Path("jira-create-issue-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in jira_create_issue.yaml documents the arithmetic: +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (1080) already covers this, so it is kept +# verbatim rather than raised. + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _import_jira_is(): + """Lazy import so an empty/pre-authoring sandbox fails cleanly on the + seed.json/bpmn checks above, rather than an ImportError traceback here. + + This checker lives in `_shared/`, not the task dir, so `jira_is` is not a + sibling module (unlike escalation_is.py, imported from the task's own + `_setup/`). Grading runs with CWD at the sandbox root, where the task's + `_setup/jira_is.py` is mounted -- the same file pre_run/post_run use. + """ + sys.path.insert(0, os.path.abspath("_setup")) + try: + import jira_is # noqa: PLC0415 + except ImportError as exc: + raise CheckFailure(f"jira_is module not found under _setup/ ({exc})") from exc + return jira_is + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def is_create_issue_node(node_name: str, object_name: str, method: str) -> bool: + """Curated (`curated_create_issue`) OR generic (`issue` + POST) form.""" + if CREATE_OP_RE.search(object_name or "") or CREATE_OP_RE.search(node_name or ""): + return True + return ( + (object_name or "").strip().lower() == GENERIC_OBJECT + and (method or "").strip().upper() in GENERIC_CREATE_METHODS + ) + + +def find_create_issue_nodes(root: ET.Element) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution Jira Create-Issue op. + + Scans every descendant, not a fixed tag list (registry templates may emit + a connector activity as sendTask, serviceTask, or a plain task) -- + mirrors check_jira_get_issue.py's find_get_issue_nodes(). + """ + found = [] + for node in root.iter(): + context = connector_context(node) + if context.get("connectorKey") != JIRA_KEY: + continue + if ACTIVITY_TYPE not in ET.tostring(node, encoding="unicode"): + continue + node_name = node.attrib.get("name", "") + if is_create_issue_node(node_name, context.get("objectName", ""), context.get("method", "")): + found.append(node) + return found + + +def collect_candidate_keys( + variables_data: object, raw_variables_text: str, project: str +) -> list[str]: + """Clean output leaves (any depth) + a project-scoped regex scan of the raw + variables-all response text (covers a key buried in a nested response + blob) -- mirrors Flow's `collect_outputs(payload)` + `get_last_debug_raw()` + dual candidate collection. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + cands = [s for leaf in leaves for s in [str(leaf).strip()] if ISSUE_KEY_RE.match(s)] + cands += re.findall(rf"\b{re.escape(project)}-\d+\b", raw_variables_text) + return list(dict.fromkeys(cands)) # de-dup, keep order + + +def main() -> None: + seed_path = Path("seed.json") + if not seed_path.is_file(): + _fail("seed.json is missing from the sandbox (pre_run seed did not run)") + seed = json.loads(seed_path.read_text(encoding="utf-8")) + project = seed["project_key"] + + bpmn_path = find_bpmn_file(NAME_HINT) + raw = Path(bpmn_path).read_text(encoding="utf-8") + if JIRA_KEY not in raw: + _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") + print(f"OK: bpmn references {JIRA_KEY}") + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + create_nodes = find_create_issue_nodes(root) + if not create_nodes: + _fail("bpmn does not reference a Jira Create-Issue connector node (Intsvc.ActivityExecution)") + print("OK: bpmn references a Create-Issue op") + + missing = [field for field in SEED_LITERAL_FIELDS if str(seed[field]) not in raw] + if missing: + _fail( + "bpmn does not reference the seeded " + f"{', '.join(f'{field}={seed[field]!r}' for field in missing)} " + "(agent must use seed.json values verbatim, not invented ones)" + ) + print("OK: bpmn references the seeded project_key/issuetype_id/summary/reporter_id") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "JiraCreateIssueLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + cands = collect_candidate_keys(variables_data, variables.stdout or "", project) + if not cands: + _fail(f"no issue key (e.g. {project}-123) in bpmn debug outputs") + print(f"OK: candidate keys from debug: {cands}") + + # Journal every candidate BEFORE the tenant read: a real issue may exist + # even if its summary doesn't end up matching below, and this journal is + # the only sweep that survives coder_eval SIGKILLing this process on the + # criterion timeout (LIVE-ADDENDUM: journal side-effect ids the moment + # they are visible). + Path(".created_keys").write_text("\n".join(cands) + "\n") + + jira_is = _import_jira_is() + conn = jira_is.connection_id() + for key in cands: + fields = jira_is.get_issue(conn, key) + if fields and fields.get("summary") == seed["summary"]: + print(f"OK: Jira issue {key} exists with the seed summary") + print("PASS: all JiraCreateIssue checks passed") + return + _fail( + f"none of {cands} carries the seed summary {seed['summary']!r} — the " + "bpmn process did not create the expected issue in Jira" + ) + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py new file mode 100644 index 0000000000..7d3a2ea45d --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py @@ -0,0 +1,285 @@ +#!/usr/bin/env python3 +"""JiraGetIssue (BPMN): structural + live checks. + +Ported from Flow `e2e/jira_get_issue/_shared/check_jira_get_issue.py`: same +scenario (a manual-start process reads one pre-seeded Jira issue by key via +the Atlassian Jira "Get Issue" connector activity and exposes its summary), +translated from a JSON node walk + inline `flow debug` payload to an XML walk +over the registry-driven `Intsvc.ActivityExecution` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) plus the +BPMN live-debug surface (`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical +pattern: ephemeral solution import, `bpmn debug`, `debug-instance +variables-all`/`incidents`). + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. The issue is created/cleaned up by seed_jira.py / teardown_jira.py; +this check only reads the tenant through the process it grades, never +directly. + +Assertion map (Flow → BPMN): + F check_jira_get_issue.py:43 JIRA_KEY not in raw ('"nodes"' marker dropped -- XML has no JSON "nodes" key, see I below) + → JIRA_KEY not in raw text of the .bpmn + F check_jira_get_issue.py:46 GET_OP_RE.search(raw) (Get-Issue op referenced) + → find_get_issue_nodes(): a sendTask carrying Intsvc.ActivityExecution + whose connectorKey is uipath-atlassian-jira and whose objectName/method + classify as Get Issue (curated `curated_get_issue` OR generic object + `issue` + method GETBYID/GET -- catalog checked via + `uip is activities list uipath-atlassian-jira --output json`) + F check_jira_get_issue.py:48 issue_key not in raw → issue_key not in raw text of the .bpmn + F check_jira_get_issue.py:52 run_debug(...) implicitly requires finalStatus == "Completed" + (flow_check.run_debug raises on a non-Completed status internally; + `bpmn debug` returns only an instance id, so the check is explicit here) + → FinalStatus in COMPLETED_STATUSES and debug-instance incidents is empty + F check_jira_get_issue.py:55 assert_outputs_contain(payload, seed["summary"]) + → seeded summary found among the root scope's variable leaves AND every + element's Outputs in `debug-instance variables-all` (LIVE-ADDENDUM: a + root PUBLIC OUTPUT has read back null even when mapped correctly, so the + search is not scoped to a declared output variable) + I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) + → bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported + bytes against the submitted file -- `bpmn debug` runs against an imported project, + unlike `flow debug`, which runs directly against the discovered project directory + → LIVE-ADDENDUM canonical live pattern (mirrors + e2e/customer_escalation_triage/check_customer_escalation_behavior.py) + T GETBYID/GET equivalence; curated OR generic entity-CRUD classification (BATCH1-ADDENDUM) + → find_get_issue_nodes() + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- not in Flow; the + `bpmn validate` criterion covers structure + +No tenant re-read is performed (unlike the escalation Create-Issue port): this +task only GETs a pre-existing issue, so there is no newly-created record to +verify against the tenant, matching Flow's own grader. +""" + +from __future__ import annotations + +import json +import os +import re +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_drive_to_slack.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +JIRA_KEY = "uipath-atlassian-jira" +GET_OP_RE = re.compile(r"get[\s_-]?issue|curated_get_issue", re.IGNORECASE) +GENERIC_OBJECT = "issue" +GENERIC_GET_METHODS = {"GETBYID", "GET"} +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +NAME_HINT = "JiraGetIssue" + +LIVE_RUN_DIR = Path("jira-get-issue-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in jira_get_issue.yaml documents the arithmetic: +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (1080) already covers this, so it is kept +# verbatim rather than raised. + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def is_get_issue_node(node_name: str, object_name: str, method: str) -> bool: + """Curated (`curated_get_issue`) OR generic (`issue` + GETBYID/GET) form.""" + if GET_OP_RE.search(object_name or "") or GET_OP_RE.search(node_name or ""): + return True + return ( + (object_name or "").strip().lower() == GENERIC_OBJECT + and (method or "").strip().upper() in GENERIC_GET_METHODS + ) + + +def find_get_issue_nodes(root: ET.Element) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution Jira Get-Issue op. + + Scans every descendant, not a fixed tag list (registry templates may emit + a connector activity as sendTask, serviceTask, or a plain task) -- + mirrors bpmn_live.index_runtime_connectors' own scanning discipline. + """ + found = [] + for node in root.iter(): + context = connector_context(node) + if context.get("connectorKey") != JIRA_KEY: + continue + if not any( + token in ET.tostring(node, encoding="unicode") for token in (ACTIVITY_TYPE,) + ): + continue + node_name = node.attrib.get("name", "") + if is_get_issue_node(node_name, context.get("objectName", ""), context.get("method", "")): + found.append(node) + return found + + +def collect_output_haystack(variables_data: object) -> str: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- it mirrors Flow's own + assert_outputs_contain(), which flattens the whole outputs payload. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return "\n".join(str(v) for v in leaves).lower() + + +def main() -> None: + seed_path = Path("seed.json") + if not seed_path.is_file(): + _fail("seed.json is missing from the sandbox (pre_run seed did not run)") + seed = json.loads(seed_path.read_text(encoding="utf-8")) + issue_key = seed["issue_key"] + + bpmn_path = find_bpmn_file(NAME_HINT) + raw = Path(bpmn_path).read_text(encoding="utf-8") + if JIRA_KEY not in raw: + _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") + print(f"OK: bpmn references {JIRA_KEY}") + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + get_issue_nodes = find_get_issue_nodes(root) + if not get_issue_nodes: + _fail("bpmn does not reference a Jira Get-Issue connector node (Intsvc.ActivityExecution)") + if issue_key not in raw: + _fail(f"bpmn does not reference the seeded key {issue_key!r} (agent must read it from seed.json)") + print(f"OK: bpmn references a Get-Issue op and the seeded key {issue_key}") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "JiraGetIssueLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + haystack = collect_output_haystack(variables_data) + if seed["summary"].lower() not in haystack: + _fail( + f"outputs do not contain the seeded issue summary {seed['summary']!r}\n" + f"outputs: {haystack[:1000]}" + ) + print("OK: bpmn outputs contain the seeded issue summary") + print("PASS: all JiraGetIssue checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py new file mode 100644 index 0000000000..1e9aa2405d --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py @@ -0,0 +1,351 @@ +#!/usr/bin/env python3 +"""JiraLifecycle (BPMN): structural + live + tenant checks for a +multi-instance-loop-and-gateway process. + +Ported from Flow `e2e/jira_lifecycle/_shared/check_jira_lifecycle.py`: same +scenario (a manual-start process iterates a seeded batch of issues and, per +item, creates a Jira issue then routes on the item's `priority` through a +branch node to a branch-specific Add-Comment), translated from a JSON node +walk + inline `flow debug` payload to an XML walk over the BPMN loop/gateway +constructs plus the BPMN live-debug surface (`_shared/bpmn_live.py`, per +LIVE-ADDENDUM's canonical pattern: ephemeral solution import, `bpmn debug`, +`debug-instance variables-all`/`incidents`). + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. Every confirmed key is recorded to ``.created_keys`` as soon as it is +seen, so post_run teardown deletes it even if a later assertion fails. + +Assertion map (Flow -> BPMN): + F check_jira_lifecycle.py:75 JIRA_KEY not in raw or '"nodes"' not in raw + ('"nodes"' marker dropped -- XML has no JSON + "nodes" key, see I below) + -> JIRA_KEY not in raw text of the .bpmn + F check_jira_lifecycle.py:79 assert_flow_has_node_type(["core.logic.loop"]) + -> has_multi_instance_loop(root): a + bpmn:multiInstanceLoopCharacteristics element + is present anywhere in the process + F check_jira_lifecycle.py:80 assert_flow_has_any_node_type(["core.logic.switch", + "core.logic.decision"]) + -> has_conditional_gateway(root): a + bpmn:exclusiveGateway element together with + at least one bpmn:conditionExpression is + present anywhere in the process + F check_jira_lifecycle.py:84 run_debug(timeout=600) implicitly requires + finalStatus == "Completed" (flow_check.run_debug + raises on a non-Completed status internally; + `bpmn debug` returns only an instance id, so the + check is explicit here) + -> FinalStatus in COMPLETED_STATUSES and + debug-instance incidents is empty + F check_jira_lifecycle.py:90-92 cands = regex findall of the project's + issue-key pattern over get_last_debug_raw() + (the whole inline flow debug payload text) + -> same regex over json.dumps(variables_data) + (the whole debug-instance variables-all + payload text) -- the create nodes' + responses land there regardless of how the + process mapped its outputs, mirroring + Flow's own comment + F check_jira_lifecycle.py:108 marker not in comment_blob -> branch routed + the comment correctly + -> same check, same jira_is.get_issue() field + F check_jira_lifecycle.py:115 missing = [s for s in want_marker if s not in + found] -> the loop created every seeded issue + -> same check + I locate/parse .bpmn (file exists, well-formed XML, project + directory resolved) + -> bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 + pin of the imported bytes against the submitted file -- + `bpmn debug` runs against an imported project, unlike + `flow debug`, which runs directly against the discovered + project directory + -> LIVE-ADDENDUM canonical live pattern + (mirrors + e2e/customer_escalation_triage/check_customer_escalation_behavior.py + and _shared/check_jira_get_issue.py) + I Flow imports the shared `_shared/jira_is.py` helper + (connection_id() / get_issue()); no BPMN equivalent shared + module exists yet and BATCH1-ADDENDUM asks graders not to + add a new one while several tasks port in parallel, so + the same two `uip is resources run` calls are inlined + below (byte-identical operation names/queries to the + task's own `_setup/jira_is.py`) + -> _connection_id() / _get_issue() below + DROPPED require_no_private_connector_values / require_sequence_integrity + / require_di_for_visible_elements / connection-binding checks + / per-node connector-operation classification -- not in + Flow (Flow's own structural check never classifies the + Create-Issue/Add-Comment node ops either, deferring + entirely to the LIVE tenant re-read); the `bpmn validate` + criterion covers structure +""" + +from __future__ import annotations + +import json +import os +import re +import subprocess +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# …/uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + get_ci, + incident_records, + payload_data, + run_cli, + sha256, +) + +JIRA_KEY = "uipath-atlassian-jira" +# Same tenant target as the task's own _setup/jira_is.py (and the pilot +# e2e/jira_get_issue/_setup/jira_is.py) -- a shared tenant fixture, not Flow +# vocabulary, kept verbatim per LIVE-ADDENDUM. +FOLDER_PATH = "Shared/uipath-maestro-flow" +CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" +NAME_HINT = "JiraLifecycle" + +LIVE_RUN_DIR = Path("jira-lifecycle-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes a literal timeout=720, so it prices at +# bpmn_live.debug_budget(720) == 720 (one attempt, no backoff). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in jira_lifecycle.yaml documents the arithmetic: +# 90 (solution init) + 180 (solution import) + 720 (debug) +# + 120 (variables-all) + 120 (incidents) = 1230 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1290 +# Flow's own criterion timeout (1320) already covers this (30s to spare), so +# it is kept verbatim rather than raised. +# +# The 720 below is passed as a literal (not this comment's named constant) so +# test_criterion_budgets.py's static AST pricer -- which only recognizes +# literal timeout=/retries=/backoff_seconds= arguments on the run_debug(...) +# call itself -- can price it. + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _local(tag: str) -> str: + return tag.rsplit("}", 1)[-1] + + +def has_multi_instance_loop(root: ET.Element) -> bool: + return any(_local(el.tag) == "multiInstanceLoopCharacteristics" for el in root.iter()) + + +def has_conditional_gateway(root: ET.Element) -> bool: + """A bpmn:exclusiveGateway with at least one bpmn:conditionExpression + anywhere in the process -- same any-of level of specificity as Flow's own + ``assert_flow_has_any_node_type(["core.logic.switch", "core.logic.decision"])``, + which likewise only checks node-type presence, not branch wiring.""" + has_gateway = any(_local(el.tag) == "exclusiveGateway" for el in root.iter()) + has_condition = any(_local(el.tag) == "conditionExpression" for el in root.iter()) + return has_gateway and has_condition + + +def _run(*args: str) -> dict: + out = subprocess.run( + ["uip", *args, "--output", "json"], + capture_output=True, text=True, timeout=120, + ).stdout + return json.loads(out) + + +def _connection_id() -> str: + folder_key = _run("or", "folders", "get", FOLDER_PATH)["Data"]["Key"] + conns = _run( + "is", "connections", "list", JIRA_KEY, "--folder-key", folder_key, "--refresh" + )["Data"] + return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) + + +def _get_issue(conn_id: str, key: str, project: str, issuetype_id: str) -> dict | None: + """Return the issue's `fields` dict (includes `summary` and `comment`), + or None if it doesn't exist (404).""" + env = _run( + "is", "resources", "run", "get", JIRA_KEY, "curated_get_issue", + "--connection-id", conn_id, + "--query", f"project={project}&issuetype={issuetype_id}&issueId={key}", + ) + if env.get("Result") == "Failure": + return None + return env["Data"].get("fields", {}) + + +def _record_key(key: str) -> None: + """Append a confirmed key to .created_keys (dedup) for teardown.""" + kf = Path(".created_keys") + seen = set(kf.read_text().split()) if kf.is_file() else set() + if key not in seen: + with kf.open("a") as f: + f.write(key + "\n") + + +def main() -> None: + seed = json.loads(Path("seed.json").read_text()) + issues = seed["issues"] + project = seed["project_key"] + issuetype_id = seed["issuetype_id"] + # summary -> expected comment marker for that item's branch + want_marker = { + i["summary"]: (seed["escalated_marker"] if i["priority"] == "High" else seed["routine_marker"]) + for i in issues + } + + # 1. STRUCTURAL ---------------------------------------------------------- + bpmn_path = find_bpmn_file(NAME_HINT) + raw = Path(bpmn_path).read_text(encoding="utf-8") + if JIRA_KEY not in raw: + _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") + print(f"OK: bpmn references {JIRA_KEY}") + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + if not has_multi_instance_loop(root): + _fail("bpmn does not contain a bpmn:multiInstanceLoopCharacteristics loop") + if not has_conditional_gateway(root): + _fail("bpmn does not contain a bpmn:exclusiveGateway with a conditionExpression") + print("OK: bpmn contains a multi-instance loop and a conditional exclusive gateway") + + # 2. LIVE ------------------------------------------------------------------ + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "JiraLifecycleLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log", timeout=720 + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + # Every CE- key that appears anywhere in the variables-all payload -- + # the create nodes' responses land in the runtime scopes/element outputs + # regardless of how the process mapped them, mirroring Flow's own + # whole-payload scan. + variables_text = json.dumps(variables_data) + cands = list(dict.fromkeys(re.findall(rf"\b{re.escape(project)}-\d+\b", variables_text))) + if not cands: + _fail(f"no issue key (e.g. {project}-123) in debug-instance variables-all -- the loop created nothing") + print(f"OK: candidate keys from debug: {cands}") + + # 3. TENANT ------------------------------------------------------------ + conn = _connection_id() + found: dict[str, str] = {} # summary -> key, for issues that are ours + for key in cands: + fields = _get_issue(conn, key, project, issuetype_id) + if not fields: + continue + _record_key(key) # real issue this run created -- always clean it up + summary = fields.get("summary") + if summary in want_marker: + found[summary] = key + marker = want_marker[summary] + comment_blob = json.dumps(fields.get("comment")) + if marker not in comment_blob: + _fail( + f"issue {key} ({summary!r}) is missing its expected branch " + f"comment {marker!r} -- the gateway routed it to the wrong " + f"branch (or no comment was posted)" + ) + + missing = [s for s in want_marker if s not in found] + if missing: + _fail( + f"the loop did not create every seeded issue -- missing {missing}; " + f"created and matched: {list(found.values())}" + ) + + print(f"OK: all {len(want_marker)} issues created and each carries its correct branch comment") + print("PASS: all JiraLifecycle checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py new file mode 100644 index 0000000000..5b5cd67bec --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py @@ -0,0 +1,283 @@ +#!/usr/bin/env python3 +"""JiraSearchTriage (BPMN): structural + live + tenant checks for a +JQL-search-driven triage process. + +Ported from Flow `e2e/jira_search_triage/_shared/check_jira_search_triage.py`: +same scenario (a manual-start process searches Jira by a seeded JQL and, for +each match, adds a triage comment via a loop), translated from a JSON node +walk + inline `flow debug` payload to an XML walk over the registry-driven +`Intsvc.ActivityExecution` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) plus the +BPMN live-debug surface (`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical +pattern: ephemeral solution import, `bpmn debug`, `debug-instance incidents`). +The loop construct is `bpmn:multiInstanceLoopCharacteristics` (see +skills/uipath-maestro-bpmn/references/structural-bpmn.md), the BPMN +translation of Flow's `core.logic.loop` node type. + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. The two issues are created/cleaned up by seed_jira.py / +teardown_jira.py; this check only reads the tenant through the process it +grades, never directly. + +Assertion map (Flow → BPMN): + F check_jira_search_triage.py:40 JIRA_KEY not in raw ('"nodes"' marker dropped -- XML has no JSON "nodes" key, see I below) + → JIRA_KEY not in raw text of the .bpmn + F check_jira_search_triage.py:43 SEARCH_OP_RE.search(raw) (Search-Issues-by-JQL op referenced) + → find_search_issue_nodes(): a sendTask carrying Intsvc.ActivityExecution + whose connectorKey is uipath-atlassian-jira and whose objectName/method + classify as Search Issues by JQL (curated `issue_search_get` OR a generic + GET node whose serialized XML contains a `jql` token -- catalog checked + via `uip is activities list uipath-atlassian-jira --output json`) + F check_jira_search_triage.py:45 assert_flow_has_node_type(["core.logic.loop"]) + → has_loop(): at least one bpmn:multiInstanceLoopCharacteristics element + (PORTING-BRIEF construct-translation table: Loop → multiInstanceLoopCharacteristics) + F check_jira_search_triage.py:48 run_debug(timeout=600) implicitly requires finalStatus == "Completed" + (flow_check.run_debug raises on a non-Completed status internally; + `bpmn debug` returns only an instance id, so the check is explicit here) + → FinalStatus in COMPLETED_STATUSES and debug-instance incidents is empty + F check_jira_search_triage.py:51-60 conn = jira_is.connection_id(); for key in issue_keys: re-read + assert + marker in fields["comment"] + → identical tenant re-read against the same jira_is.py helper, unchanged + I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) + → bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported + bytes against the submitted file -- `bpmn debug` runs against an imported project, + unlike `flow debug`, which runs directly against the discovered project directory + → LIVE-ADDENDUM canonical live pattern (mirrors + e2e/jira_get_issue/check_jira_get_issue.py) + T curated OR generic entity-CRUD classification (BATCH1-ADDENDUM); GET/GETBYID + equivalence not needed here (search is GET-only), but the same dual-form tolerance + (curated objectName vs generic object+method) applies + → is_search_issue_node() + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- not in Flow; the + `bpmn validate` criterion covers structure. No structural check for the Add-Comment + node either -- Flow's own grader never asserts one (it proves the comment landed via + the tenant re-read instead), so none is added here. + +No output-value assertion is made (unlike the jira_get_issue port): Flow's own +grader never reads `flow debug`'s output payload for this task, only the +tenant re-read, so `debug-instance variables-all` is not called here. +""" + +from __future__ import annotations + +import json +import os +import re +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +HERE = os.path.dirname(os.path.abspath(__file__)) # …/uipath-maestro-bpmn/_shared +# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(HERE)) # noqa: E402 +# This task's own _setup/ in the repo (not the sandbox mount -- this checker +# runs from $REFERENCE_DIR, so it reads jira_is.py straight off disk here, +# the same way check_customer_escalation_behavior.py reads its own HERE/_setup). +# No new _shared module is added (BATCH1-ADDENDUM: agents write in parallel, +# do not add new shared modules) -- this imports the verbatim per-task copy. +sys.path.insert(0, os.path.join(HERE, "..", "e2e", "jira_search_triage", "_setup")) # noqa: E402 + +from _shared.bpmn_check import elements, find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + run_cli, + sha256, +) +import jira_is # noqa: E402 + +JIRA_KEY = "uipath-atlassian-jira" +SEARCH_OP_RE = re.compile(r"search[\s_-]?issues?|issue_search_get|search-issues-by-jql", re.IGNORECASE) +GENERIC_SEARCH_OBJECTS = {"issue", "issues"} +GENERIC_SEARCH_METHODS = {"GET"} +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +NAME_HINT = "JiraSearchTriage" + +LIVE_RUN_DIR = Path("jira-search-triage-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, incidents) and the tenant +# re-read through jira_is.py (whose `_run` helper hardcodes a 120s subprocess +# timeout per call) are not priced by that guard, so their sum is added by +# hand here and the criterion `timeout:` in jira_search_triage.yaml documents +# the arithmetic: +# 90 (solution init) + 180 (solution import) + 480 (debug) + 120 (incidents) +# + 240 (connection_id: 2 tenant calls @ up to 120s each) +# + 240 (2 x get_issue re-read @ up to 120s each, one per seeded issue) +# = 1350 + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1410 +# Flow's own criterion timeout (1320) does not cover the extra ephemeral- +# solution import + live tenant re-read plumbing this BPMN sequence needs +# beyond Flow's single inline `flow debug` call, so it is raised to 1440 +# (the one sanctioned deviation from "criteria identical" per LIVE-ADDENDUM's +# Budgets section -- a property of the CLI surface, not of what is graded). + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def is_search_issue_node(node_name: str, object_name: str, method: str, node_xml: str) -> bool: + """Curated (`issue_search_get`) OR generic (GET + a `jql` token anywhere + in the node) form. The `jql` token requirement keeps a generic GET node + from being mistaken for search: BATCH1-ADDENDUM's dual-form tolerance + classifies by objectName/method, but Search Issues by JQL has no distinct + generic object name of its own (unlike Get Issue's `issue`+GETBYID).""" + if SEARCH_OP_RE.search(object_name or "") or SEARCH_OP_RE.search(node_name or ""): + return True + return ( + (object_name or "").strip().lower() in GENERIC_SEARCH_OBJECTS + and (method or "").strip().upper() in GENERIC_SEARCH_METHODS + and "jql" in node_xml.lower() + ) + + +def find_search_issue_nodes(root: ET.Element) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution Jira Search-Issues op. + + Scans every descendant, not a fixed tag list (registry templates may emit + a connector activity as sendTask, serviceTask, or a plain task) -- + mirrors bpmn_live.index_runtime_connectors' own scanning discipline. + """ + found = [] + for node in root.iter(): + context = connector_context(node) + if context.get("connectorKey") != JIRA_KEY: + continue + node_xml = ET.tostring(node, encoding="unicode") + if ACTIVITY_TYPE not in node_xml: + continue + node_name = node.attrib.get("name", "") + if is_search_issue_node(node_name, context.get("objectName", ""), context.get("method", ""), node_xml): + found.append(node) + return found + + +def has_loop(root: ET.Element) -> bool: + return bool(elements(root, "multiInstanceLoopCharacteristics")) + + +def main() -> None: + seed_path = Path("seed.json") + if not seed_path.is_file(): + _fail("seed.json is missing from the sandbox (pre_run seed did not run)") + seed = json.loads(seed_path.read_text(encoding="utf-8")) + + bpmn_path = find_bpmn_file(NAME_HINT) + raw = Path(bpmn_path).read_text(encoding="utf-8") + if JIRA_KEY not in raw: + _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") + print(f"OK: bpmn references {JIRA_KEY}") + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + if not find_search_issue_nodes(root): + _fail("bpmn does not reference a Jira Search-Issues-by-JQL connector node (Intsvc.ActivityExecution)") + if not has_loop(root): + _fail("bpmn does not contain a multiInstanceLoopCharacteristics loop over the search results") + print("OK: bpmn references a JQL search op and a loop construct") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "JiraSearchTriageLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + conn = jira_is.connection_id() + marker = seed["processed_comment"] + for key in seed["issue_keys"]: + fields = jira_is.get_issue(conn, key) + if not fields: + _fail(f"seeded issue {key} not found on re-read") + if marker not in json.dumps(fields.get("comment")): + _fail( + f"issue {key} is missing the triage comment {marker!r} — the " + "search-driven loop did not comment it" + ) + print(f"OK: all {len(seed['issue_keys'])} matched issues carry the triage comment") + print("PASS: all JiraSearchTriage checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py new file mode 100644 index 0000000000..ea2ebab203 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py @@ -0,0 +1,322 @@ +#!/usr/bin/env python3 +"""SlackEmojiListTest (BPMN): connector-mode HTTP fallback + live checks. + +Ported from Flow `connector_features/slack-http-fallback/ +check_slack_http_fallback.py`: same scenario (the Slack catalog connector, +``uipath-salesforce-slack``, has no native activity for "list a team's custom +emoji" -- Slack's ``emoji.list`` endpoint -- so the skill must fall back to a +connector-mode HTTP request that reuses the existing Slack connection's +managed auth, then the process must debug green), translated from a JSON +node/``inputs.detail`` walk to an XML walk over the registry-driven +``Intsvc.ActivityExecution`` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §"Connectionless +vs connector HTTP") plus the BPMN live-debug surface (``_shared/ +bpmn_live.py``, per LIVE-ADDENDUM's canonical pattern: ephemeral solution +import, ``bpmn debug``, ``debug-instance incidents``). + +Two subcommands (subcommand-dispatched, matching Flow's own checker): + + check_fallback Structural -- the emoji list is built as a connector-mode + HTTP node bound to the Slack connector (NOT a curated + native activity for something else), targeting the Slack + ``emoji.list`` endpoint. + check_debug Runtime -- ``uip maestro bpmn debug`` finishes with a + completed final status and no incidents, against the live + Slack connection. + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. + +Assertion map (Flow → BPMN): + F check_slack_http_fallback.py:_is_slack_http_fallback (a fallback node exists) + → find_fallback_nodes(): an element carrying Intsvc.ActivityExecution + or Intsvc.HttpExecution whose connectorKey context field equals + uipath-salesforce-slack + F check_slack_http_fallback.py:_is_slack_http_fallback (HTTP-shaped, not a native op) + → discriminated by the emoji.list endpoint match below, not by a + node `type` string: BPMN's registry wrapper is the SAME + Intsvc.ActivityExecution tag for both a curated native activity + (e.g. Get Channel Info) and an HTTP fallback, unlike Flow's node + `type`, which differs per shape (see GUESS) + F check_slack_http_fallback.py:EMOJI_ENDPOINT blob search (json.dumps(fallback_nodes).lower()) + → references_emoji_endpoint(): 'emoji.list' found anywhere across + every uipath:input name/value/text at any depth under the node, + plus the node's raw XML as a lenient fallback -- mirrors Flow's own + whole-node-blob substring search + F check_slack_http_fallback.py:check_debug (run_debug(timeout=300); flow_check.run_debug raises + internally on a non-Completed status) + → FinalStatus in COMPLETED_STATUSES and debug-instance incidents is + empty (bpmn debug returns only an instance id, so the check is + explicit here) + I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) + → bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported + bytes against the submitted file -- `bpmn debug` runs against an imported project, + unlike `flow debug`, which runs directly against the discovered project directory + → LIVE-ADDENDUM canonical live pattern (mirrors + e2e/jira_get_issue and multi_node/slack_channel_description) + T curated (Intsvc.ActivityExecution) OR the connector-authenticated form of + Intsvc.HttpExecution -- accept either wrapper tag (BATCH1-ADDENDUM; same + tolerance and same open GUESS as check_non_catalog_http_fallback.py) + → ACTIVITY_TYPES tuple checked via has_type() + T collect uipath:input elements at any depth under the node + → all_inputs() uses `.//uipath:input` + T endpoint value found in a context field, a sibling uipath:input's own + value/text, or inside the target="body" JSON payload (the skill does not pin + where the endpoint lands) + → references_emoji_endpoint() + +GUESS (flag for reviewer): same open question as check_non_catalog_http_fallback.py -- +registry-workflow.md documents `Intsvc.HttpExecution`'s `mode` context field as +hardcoded to "manual" (connectionless) with no documented connector-authenticated +alternative; the skill's own contract splits connector-mode HTTP as +`Intsvc.ActivityExecution` (a connector object/operation, here reused via a +generic "http-request" passthrough objectName under the Slack connectorKey -- +see the real CI-passing SpotifyProfileTest.bpmn fixture, which authors exactly +this shape for the non-catalog case) and reserves `Intsvc.HttpExecution` for +connectionless/manual calls only. To stay faithful to both the porting brief +and the skill's documented contract without inventing a hard requirement on one +wrapper tag, this checker classifies purely by `connectorKey` (+ the emoji.list +endpoint match), and accepts either wrapper tag carrying them. + +No Flow assertions dropped: node existence, the HTTP-fallback shape (translated +via connectorKey since BPMN's node `type` cannot distinguish curated vs raw the +way Flow's node.type string does), and the emoji.list endpoint substring all +have a BPMN counterpart above. check_debug's only requirement in Flow is that +`flow debug` completes (finalStatus Completed); no output-content assertion is +made there, so none is added here either. + +Usage (from a task's run_command, cwd = sandbox root): + python3 $REFERENCE_DIR/_shared/check_slack_http_fallback.py check_fallback + python3 $REFERENCE_DIR/_shared/check_slack_http_fallback.py check_debug +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import NS, fail, find_bpmn_file, parse_bpmn, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + run_cli, + sha256, +) + +NAME_HINT = "SlackEmojiListTest" +SLACK_KEY = "uipath-salesforce-slack" +# Slack endpoint that lists a team's custom emoji. Bare token, so both '/emoji.list' +# and 'emoji.list' forms satisfy the check, mirroring Flow's own tolerance. +EMOJI_ENDPOINT = "emoji.list" +ACTIVITY_TYPES = ("Intsvc.ActivityExecution", "Intsvc.HttpExecution") + +LIVE_RUN_DIR = Path("slack-emoji-list-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock check_debug can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the call +# below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps are not priced by that guard, so their sum is +# added by hand here and the check_debug criterion `timeout:` in +# slack_http_fallback.yaml documents the arithmetic: +# 90 (solution init) + 180 (solution import) + 480 (debug) + 120 (incidents) +# = 870 + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 930 +# Flow's own criterion timeout (720) does not fit the extra CLI steps a BPMN +# live grade needs (solution init/import, a separate incidents read), so it +# is raised to 930 -- the one sanctioned deviation from "criteria identical" +# (LIVE-ADDENDUM: a property of the CLI surface, not of what is graded). + + +def has_type(el: ET.Element, token: str) -> bool: + return token in ET.tostring(el, encoding="unicode") + + +def all_inputs(el: ET.Element) -> list[ET.Element]: + # `.//` walks every uipath:input under the node at any depth -- agents + # sometimes nest body/query/path inputs inside uipath:context. + return el.findall(".//uipath:input", NS) + + +def node_blob(el: ET.Element) -> str: + parts = [ET.tostring(el, encoding="unicode")] + for inp in all_inputs(el): + parts.append(inp.attrib.get("name") or "") + parts.append(inp.attrib.get("value") or "") + parts.append(inp.text or "") + return "\n".join(parts).lower() + + +def is_slack_connector_node(el: ET.Element) -> bool: + """True when ``el`` is itself the connector-activity host element (a + sendTask/serviceTask/plain task carrying its OWN + ``extensionElements/uipath:activity``) for the Slack connector. + + ``connector_context`` (bpmn_live.py) resolves the activity via a + DIRECT-child path (``./extensionElements/activity``), not a recursive + ``.//`` search -- root.iter() walks every ancestor of the real host + element too (the process, the definitions root, the host's own + extensionElements/activity wrapper), and each of those "contains" the + same connectorKey/type tokens somewhere in its serialized subtree. Only + the direct-child lookup correctly isolates the one true host element + instead of matching every ancestor as well. + """ + context = connector_context(el) + if context.get("connectorKey", "").strip().lower() != SLACK_KEY: + return False + return any(has_type(el, token) for token in ACTIVITY_TYPES) + + +def references_emoji_endpoint(el: ET.Element) -> bool: + return EMOJI_ENDPOINT in node_blob(el) + + +def find_slack_connector_nodes(root: ET.Element) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution/HttpExecution node + whose connectorKey is the Slack connector. Scans every descendant, not a + fixed tag list (registry templates may emit a connector activity as + sendTask, serviceTask, or a plain task).""" + return [node for node in root.iter() if is_slack_connector_node(node)] + + +# ── subcommand: check_fallback ────────────────────────────────────────────── +def check_fallback() -> None: + path, root = parse_bpmn(NAME_HINT) + + slack_nodes = find_slack_connector_nodes(root) + if not slack_nodes: + seen = sorted( + { + connector_context(n).get("connectorKey", "") + for n in root.iter() + if connector_context(n).get("connectorKey") + } + ) + fail( + f"No {SLACK_KEY!r} connector node found. The catalog connector has " + "no native 'list custom emoji' activity, so the process must fall " + "back to a connector-mode HTTP request bound to the Slack " + f"connection. connectorKey values seen: {seen}" + ) + print(f"OK: {len(slack_nodes)} {SLACK_KEY!r} connector node(s) present") + + fallback_nodes = [n for n in slack_nodes if references_emoji_endpoint(n)] + if not fallback_nodes: + fail( + f"No {SLACK_KEY!r} connector node targets the {EMOJI_ENDPOINT!r} " + "endpoint (expected the '/emoji.list' path, which lists a team's " + "custom emoji)." + ) + print( + f"OK: {len(fallback_nodes)} Slack connector node(s) target the " + f"'{EMOJI_ENDPOINT}' endpoint" + ) + print(f"OK: {path} -- all Slack HTTP-fallback structural checks passed") + + +# ── subcommand: check_debug ────────────────────────────────────────────────── +def check_debug() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "SlackEmojiListTestLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print( + "OK: uip maestro bpmn debug finished with FinalStatus=%s (no incidents)" + % final_status + ) + + +DISPATCH = { + "check_fallback": check_fallback, + "check_debug": check_debug, +} + + +def main() -> None: + if len(sys.argv) < 2 or sys.argv[1] not in DISPATCH: + fail(f"usage: {os.path.basename(sys.argv[0])} {{{'|'.join(DISPATCH)}}}") + DISPATCH[sys.argv[1]]() + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + sys.exit(f"FAIL: {error}") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py new file mode 100644 index 0000000000..71c9f6915d --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py @@ -0,0 +1,346 @@ +#!/usr/bin/env python3 +"""SlackWeatherPipeline (BPMN): structural + live checks. + +Ported from Flow `multi_node/slack_weather_pipeline/_shared/check_slack_weather_pipeline.py` +(via `tests/tasks/uipath-maestro-flow/_shared/check_slack_weather_pipeline.py`): +same scenario (a manual-start process reads the #office-bellevue Slack +channel description, extracts the city, fetches weather for that city from +open-meteo, and decides warm/cold), translated from a JSON node walk + inline +`flow debug` payload to an XML walk over the registry-driven +`Intsvc.ActivityExecution`/`Intsvc.HttpExecution` connector shells (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) plus the +BPMN live-debug surface (`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical +pattern: ephemeral solution import, `bpmn debug`, `debug-instance +variables-all`/`incidents`). Mirrors the structure of the CI-passing +`_shared/check_channel_description.py` and `_shared/check_jira_get_issue.py` +(same connector tenant, same live sequence and budget arithmetic). + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. Nothing is created against the tenant by this scenario (it only reads +a channel's description and calls a public weather API), so there is no +side-effect record to tear down -- matching Flow's own grader, which has no +teardown either. + +Assertion map (Flow → BPMN): + F check_slack_weather_pipeline.py:23 assert_flow_uses_connector_target('uipath-salesforce-slack') + → find_connector_nodes(): any element carrying + Intsvc.ActivityExecution whose connectorKey context field + equals uipath-salesforce-slack. No operation filter -- Flow's + own assertion has none either (it does not care whether the + Slack node reads the channel description via a curated or + generic op), so this port keeps the same breadth rather than + narrowing it. Mirrors check_channel_description.py's + find_connector_nodes() (same connector, same tenant folder). + F check_slack_weather_pipeline.py:24 assert_flow_has_api_node_targeting(['open-meteo', 'openmeteoapis']) + → find_weather_node(): any element carrying Intsvc.HttpExecution + OR Intsvc.ActivityExecution whose own uipath:activity + extension (not the node's full serialized subtree, which + would falsely inherit a descendant match onto every ancestor) + contains 'open-meteo' or 'openmeteoapis' (case-insensitive). Keeps + Flow's own tolerance for either a raw HTTP node or a curated + connector node -- the BPMN skill's node-selection ladder may + pick either, and PORTING-BRIEF's construct table designates + Intsvc.HttpExecution manual mode as the expected shape without + requiring it (Flow does not gate on manual vs. connected mode + either). + F check_slack_weather_pipeline.py:28-29 payload = run_debug(timeout=240); implicitly requires + finalStatus == 'Completed' (flow_check.run_debug raises on a + non-Completed status internally; `bpmn debug` returns only an + instance id, so the check is explicit here) + → FinalStatus in COMPLETED_STATUSES and debug-instance + incidents is empty + F check_slack_weather_pipeline.py:31 assert_output_nonempty(payload, 'weatherVerdict') + + lines 32-37 exactly one of ALLOWED_VERDICTS found + → collect_output_haystack(): value leaves of the root scope's + variables AND every element's Outputs in `debug-instance + variables-all` (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read + back null even when mapped correctly, so the search is not + scoped to one declared output variable -- LIVE-ADDENDUM's + translation rule for any Flow output-value assertion). + Widening from Flow's own name-scoped lookup carries no + practical false-positive risk here: the searched strings are + the full literal multi-word verdict phrases ('warm office + today' / 'cold office today'), not a loose substring, so + neither the Slack street-address text nor the raw Open-Meteo + JSON body (numeric fields only) can coincidentally match. The + exact-one-hit check is otherwise verbatim Flow logic. + I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) + → bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported + bytes against the submitted file -- `bpmn debug` runs against an imported project, + unlike `flow debug`, which runs directly against the discovered project directory + → LIVE-ADDENDUM canonical live pattern (mirrors + check_channel_description.py / check_jira_get_issue.py) + DROPPED the HTTP-proxy fallback branch of assert_flow_uses_connector_target (a + `core.action.http.v2` node with bodyParameters.targetConnector) -- a legacy Flow + accommodation for connector-backed flows authored before native connector node + types existed. The BPMN skill's registry enrichment always emits + Intsvc.ActivityExecution for a connector activity (registry-workflow.md §3), so no + analogous construct exists to translate. + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- not in Flow; the + `bpmn validate` criterion covers structure. +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_channel_description.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + BPMN_NS, + CheckFailure, + UIPATH_NS, + connector_context, + get_ci, + incident_records, + payload_data, + q, + root_scope, + run_cli, + sha256, +) + +SLACK_CONNECTOR_KEY = "uipath-salesforce-slack" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +HTTP_TYPE = "Intsvc.HttpExecution" +WEATHER_HINTS = ("open-meteo", "openmeteoapis") +NAME_HINT = "SlackWeatherPipeline" + +ALLOWED_VERDICTS = ("warm office today", "cold office today") + +LIVE_RUN_DIR = Path("slack-weather-pipeline-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in slack_weather_pipeline.yaml documents the +# arithmetic (mirrors check_channel_description.py / check_jira_get_issue.py): +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (600) does not fit the extra CLI steps a BPMN +# live grade needs (solution init/import, separate variables-all/incidents +# reads), so it is raised to 1050 -- the one sanctioned deviation from +# "criteria identical" (LIVE-ADDENDUM: a property of the CLI surface, not of +# what is graded). + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution targeting connector_key. + + No operation filter: Flow's own assertion (`assert_flow_uses_connector_target`) + only requires SOME connector node for the key, not a specific op, so this + mirrors that breadth. Scans every descendant, not a fixed tag list (registry + templates may emit a connector activity as sendTask, serviceTask, or a plain + task) -- mirrors bpmn_live.index_runtime_connectors' own scanning discipline. + """ + found = [] + for node in root.iter(): + context = connector_context(node) + if context.get("connectorKey") != connector_key: + continue + if ACTIVITY_TYPE not in ET.tostring(node, encoding="unicode"): + continue + found.append(node) + return found + + +def find_weather_node(root: ET.Element) -> list[ET.Element]: + """Every element that is an API-capable node actually targeting open-meteo. + + Mirrors Flow's assert_flow_has_api_node_targeting: an API-capable node is + one carrying Intsvc.HttpExecution (a manual/connectionless HTTP call) or + Intsvc.ActivityExecution (a curated connector, in case the skill's + node-selection ladder picks one for the weather call instead), and it + targets the service only when a weather hint appears anywhere in that + node's OWN uipath:activity extension. + + Scoped to the node's own `./extensionElements/activity` child (not a + blind substring search over the node's full serialized subtree): an + ancestor (bpmn:process, bpmn:definitions) serializes its whole descendant + tree, so a naive `ET.tostring(node)` scan would have every ancestor of the + real weather node "inherit" both the type token and the hint text and + falsely match too (caught by a synthetic-fixture gate before this port + shipped). Scoping to the immediate `uipath:activity` child keeps a Slack + connector node (also Intsvc.ActivityExecution) or a Script node that + merely mentions the service from satisfying this gate, same as + connector_context()'s own scoping discipline. + """ + found = [] + for node in root.iter(): + activity = node.find(f"./{q(BPMN_NS, 'extensionElements')}/{q(UIPATH_NS, 'activity')}") + if activity is None: + continue + raw = ET.tostring(activity, encoding="unicode") + if HTTP_TYPE not in raw and ACTIVITY_TYPE not in raw: + continue + if any(hint in raw.lower() for hint in WEATHER_HINTS): + found.append(node) + return found + + +def collect_output_haystack(variables_data: object) -> str: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- it mirrors Flow's own assert_output_nonempty() + widened per LIVE-ADDENDUM's translation rule for output-value assertions. + Element Outputs include a connector's nested `response` object, which + _leaves() flattens along with everything else. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return "\n".join(str(v) for v in leaves).lower() + + +def main() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + connector_nodes = find_connector_nodes(root, SLACK_CONNECTOR_KEY) + if not connector_nodes: + _fail( + f"bpmn does not reference a {SLACK_CONNECTOR_KEY} connector node " + f"({ACTIVITY_TYPE})" + ) + print(f"OK: bpmn references a {SLACK_CONNECTOR_KEY} connector node") + + weather_nodes = find_weather_node(root) + if not weather_nodes: + _fail( + f"bpmn does not reference an API-capable node ({HTTP_TYPE} or " + f"{ACTIVITY_TYPE}) targeting one of {WEATHER_HINTS}" + ) + print(f"OK: bpmn references an API node targeting open-meteo") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "SlackWeatherPipelineLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + haystack = collect_output_haystack(variables_data) + hits = [v for v in ALLOWED_VERDICTS if v in haystack] + if len(hits) != 1: + found = "both verdicts" if len(hits) > 1 else "neither verdict" + _fail( + f"outputs must contain exactly one of {list(ALLOWED_VERDICTS)}; " + f"found {found}\noutputs: {haystack[:1000]}" + ) + print(f"OK: bpmn outputs carry {hits[0]!r}") + print("PASS: all SlackWeatherPipeline checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py new file mode 100644 index 0000000000..2f3dbd60d5 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py @@ -0,0 +1,170 @@ +#!/usr/bin/env python3 +"""Data-grounded Test Manager Create+Get round-trip (BPMN). + +Ported from Flow `connector_features/testmanager_crud_grounded/check.py` and +its sibling criterion's +``_shared/flow_contains.py 'uipath-uipath-testmanager.create-test-case' 'uipath-uipath-testmanager.get-test-case'``. + +Flow's ``check.py`` proves a REAL create->read round-trip through the +connector by comparing two files the run itself produced -- the pre_run +seed (``seed.json``) and the agent's own report of what it sent/received +(``result.json``) -- without parsing the ``.flow`` artifact at all. That +comparison is file-format-agnostic, so it ports unchanged: ``round_trip()`` +and ``execution_evidence()`` below are Flow's ``main()`` verbatim, modulo the +process/flow vocabulary in messages. + +The one Flow assertion that DOES read the artifact -- the sibling +``flow_contains`` criterion's two node-type substrings -- is retargeted here +from Flow's per-operation JSON node ``type`` string to the registry-driven +``Intsvc.ActivityExecution`` sendTask shell, classified by objectName + +method exactly as ``_shared/check_testmanager_testcase_lifecycle.py`` does +(TestCase/POST = create, TestCase/GETBYID|GET = get; see BATCH1-ADDENDUM.md +and `uip is activities list uipath-uipath-testmanager --output json`, rows +CreateTestCase/TestCase/POST and GetTestCase/TestCase/GETBYID). That +classification is copied here (not imported), restricted to the two +operations this task's Flow prompt actually asks for -- per +BATCH1-ADDENDUM.md's "reuse their node-classification helpers by copying the +code (no new shared modules yet)". + +Assertion map (Flow -> BPMN): + F check.py:29-32 created_name == seed["name"] -> round_trip() (default) + F check.py:31-32 retrieved_name == seed["name"] -> round_trip() (default) + F check.py:33-34 id present -> round_trip() (default) + F check.py:24-26 created_name/retrieved_name/id all present -> execution_evidence() (--execution-evidence) + F criterion 2 flow_contains 'uipath-uipath-testmanager.create-test-case' + 'uipath-uipath-testmanager.get-test-case' -> node_types() (--node-types) + I locate/read seed.json, result.json -> load_json() + I locate/parse .bpmn (node-types mode only) -> parse_bpmn() + T objectName+method classification (no per-op node + type; see check_testmanager_testcase_lifecycle.py) -> OPERATIONS / context_value() + T GETBYID/GET equivalence for "get the test case" -> OPERATIONS methods set + +Usage (from a task's run_command, cwd = sandbox root): + python3 $REFERENCE_DIR/_shared/check_testmanager_crud_grounded.py --node-types + python3 $REFERENCE_DIR/_shared/check_testmanager_crud_grounded.py --execution-evidence + python3 $REFERENCE_DIR/_shared/check_testmanager_crud_grounded.py +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import NS, elements, parse_bpmn # noqa: E402 + +CONNECTOR_KEY = "uipath-uipath-testmanager" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" + +# (label, objectName, {acceptable method values, lowercased}) -- copied from +# _shared/check_testmanager_testcase_lifecycle.py's OPERATIONS, restricted to +# the two operations this task's Flow prompt asks for (create + get only; +# update/delete/execute are out of scope for this port). +OPERATIONS = [ + ("create a test case", "TestCase", {"post"}), + ("get the test case", "TestCase", {"getbyid", "get"}), +] + + +def fail(msg: str) -> None: + print(f"FAIL: {msg}") + sys.exit(1) + + +def load_json(path: str, what: str) -> dict: + try: + return json.load(open(path, encoding="utf-8")) + except OSError: + fail(f"{path} missing ({what})") + except json.JSONDecodeError as exc: + fail(f"{path} is not valid JSON: {exc}") + return {} # unreachable: fail() exits, but satisfies static analysis + + +def has_type(el: ET.Element, token: str) -> bool: + return token in ET.tostring(el, encoding="unicode") + + +def context_value(task: ET.Element, name: str) -> str: + for inp in task.findall(".//uipath:input", NS): + if inp.attrib.get("name") == name: + return (inp.attrib.get("value") or inp.text or "").strip() + return "" + + +def connector_tasks(root: ET.Element) -> list[ET.Element]: + return [ + task + for task in elements(root, "sendTask") + if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == CONNECTOR_KEY + ] + + +def node_types() -> None: + """F criterion 2: BPMN carries a create AND a get Test Manager connector node.""" + path, root = parse_bpmn() + tasks = connector_tasks(root) + if not tasks: + fail(f"no bpmn:sendTask carrying {ACTIVITY_TYPE} for connector key {CONNECTOR_KEY!r}") + + missing: list[str] = [] + for label, object_name, methods in OPERATIONS: + match = next( + ( + t + for t in tasks + if context_value(t, "objectName") == object_name + and context_value(t, "method").lower() in methods + ), + None, + ) + if match is None: + missing.append(f"{label} (objectName={object_name!r}, method in {sorted(methods)})") + else: + print( + f"OK: {label} -> {match.attrib.get('id', '?')} " + f"(objectName={object_name!r}, method={context_value(match, 'method')!r})" + ) + if missing: + fail("missing Test Manager create/get connector node(s):\n " + "\n ".join(missing)) + print(f"OK: create and get Test Manager connector nodes present in {path}") + + +def execution_evidence() -> None: + """F check.py:24-26: result.json records all three fields (weak presence check).""" + load_json("seed.json", "pre_run did not run") + res = load_json("result.json", "agent did not record the create/get round-trip") + if not all(res.get(key) for key in ("created_name", "retrieved_name", "id")): + fail("result.json is missing created_name, retrieved_name, or id") + print("OK: execution result recorded") + + +def round_trip() -> None: + """F check.py:29-34: retrieved value equals the unique seeded value (strict).""" + seed = load_json("seed.json", "pre_run did not run") + res = load_json("result.json", "agent did not record the create/get round-trip") + name = seed["name"] + if res.get("created_name") != name: + fail(f"created_name {res.get('created_name')!r} != seeded {name!r}") + if res.get("retrieved_name") != name: + fail(f"retrieved_name {res.get('retrieved_name')!r} != seeded {name!r} — value did not round-trip") + if not res.get("id"): + fail("no id captured from the create response") + print(f"OK: real round-trip verified — created & retrieved {name!r} (id {res['id']})") + + +def main() -> None: + args = sys.argv[1:] + if "--node-types" in args: + node_types() + elif "--execution-evidence" in args: + execution_evidence() + else: + round_trip() + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py new file mode 100644 index 0000000000..1a1f40d8cf --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py @@ -0,0 +1,273 @@ +#!/usr/bin/env python3 +"""BellevueWeather (BPMN): a weather-HTTP node is present and live output +contains one branch message. + +Ported from Flow `multi_node/bellevue_weather/_shared` (via +`uipath-maestro-flow/_shared/check_weather_flow.py`): same scenario (fetch +today's Bellevue weather from Open-Meteo and branch the summary message on +temperature), translated from a JSON node-type scan + inline `flow debug` +payload to an XML scan over the registry-driven `Intsvc.HttpExecution` +managed-HTTP shell (see skills/uipath-maestro-bpmn/references/structural-bpmn.md, +references/registry-workflow.md) plus the BPMN live-debug surface +(`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical pattern: ephemeral +solution import, `bpmn debug`, `debug-instance variables-all`/`incidents`). +The canonical live grader this file's plumbing is modeled on is +`_shared/check_jira_get_issue.py`. + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. + +Assertion map (Flow -> BPMN): + F check_weather_flow.py:20-21 assert_flow_has_any_node_type( + ["core.action.http", "custom-codereval-openmeteoapis"]) + -> any element carrying an Intsvc.HttpExecution + uipath:activity wrapper. BPMN has no curated + Open-Meteo Integration Service connector, so + only the managed-HTTP construct from + PORTING-BRIEF's construct-translation table + remains; the Flow grader's connector-fallback + branch has nothing to translate to. + F check_weather_flow.py:22 run_debug(timeout=240) implicitly requires + finalStatus == "Completed" (flow_check.run_debug + raises on a non-Completed status internally; + `bpmn debug` returns only an instance id, so the + check is explicit here) + -> FinalStatus in COMPLETED_STATUSES and + debug-instance incidents is empty + F check_weather_flow.py:23-24 assert_outputs_contain(payload, + ["nice day", "bring a jacket"], require_all=False) + -> either verdict string found among the root + scope's variable leaves AND every element's + Outputs in `debug-instance variables-all` + (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read + back null even when mapped correctly, so the + search is not scoped to one declared output + variable) + I locate/parse .bpmn (file exists, well-formed XML, project + directory resolved) -> bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 + pin of the imported bytes against the submitted file -- + `bpmn debug` runs against an imported project, unlike + `flow debug`, which runs directly against the discovered + project directory -> LIVE-ADDENDUM canonical live pattern + (mirrors check_jira_get_issue.py) + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- + not in Flow; the `bpmn validate` criterion covers structure + +Flow's grader does not assert a Script node or a Decision node exists (only +the weather-API node type and the branch output are graded), so this checker +does not add a structural check for the exclusiveGateway either -- adding one +would be a BPMN-only requirement the Flow prompt/grader never had. +""" + +from __future__ import annotations + +import os +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# .../uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import ( # noqa: E402 + find_bpmn_file, + has_typed_uipath_extension, + resolve_project, +) +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +BPMN_NS = "http://www.omg.org/spec/BPMN/20100524/MODEL" +ACTIVITY_TYPE = "Intsvc.HttpExecution" +NAME_HINT = "BellevueWeather" +VERDICTS = ("nice day", "bring a jacket") + +LIVE_RUN_DIR = Path("bellevue-weather-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in bellevue_weather.yaml documents the arithmetic: +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (600) does not cover this larger live-debug +# surface, so it is raised to 1050 -- the one sanctioned deviation +# LIVE-ADDENDUM allows, because the budget is a property of the CLI surface, +# not of what is graded. + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def find_http_execution_nodes(root: ET.Element) -> list[ET.Element]: + """Every element carrying an Intsvc.HttpExecution uipath:activity wrapper. + + Scans every descendant, not a fixed tag list (registry templates may emit + the managed-HTTP activity as sendTask, serviceTask, or a plain task) -- + mirrors bpmn_live.index_runtime_connectors' own scanning discipline. Skips + ``bpmn:extensionElements`` nodes themselves: ``has_typed_uipath_extension`` + matches an element whose OWN direct children include a matching + ``uipath:activity`` (the wrapper task) as well as an ``extensionElements`` + node (whose direct child literally is that ``uipath:activity``), which + would otherwise double-count every match once per node. + """ + return [ + el + for el in root.iter() + if el.tag != f"{{{BPMN_NS}}}extensionElements" + and has_typed_uipath_extension(el, "activity", ACTIVITY_TYPE) + ] + + +def collect_output_haystack(variables_data: object) -> str: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- it mirrors Flow's own + assert_outputs_contain(), which flattens the whole outputs payload. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return "\n".join(str(v) for v in leaves).lower() + + +def main() -> None: + bpmn_path = find_bpmn_file(NAME_HINT) + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + http_nodes = find_http_execution_nodes(root) + if not http_nodes: + _fail( + f"{bpmn_path} has no element carrying an {ACTIVITY_TYPE} uipath:activity " + "wrapper (no managed-HTTP weather node found)" + ) + print(f"OK: bpmn has a managed-HTTP node ({ACTIVITY_TYPE})") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "BellevueWeatherLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {incidents_list}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + haystack = collect_output_haystack(variables_data) + if not any(verdict in haystack for verdict in VERDICTS): + _fail( + f"outputs do not contain either verdict string {list(VERDICTS)!r}\n" + f"outputs: {haystack[:1000]}" + ) + print("OK: bpmn outputs contain a weather branch message") + print("PASS: all BellevueWeather checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py new file mode 100644 index 0000000000..d56cf9816e --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py @@ -0,0 +1,225 @@ +#!/usr/bin/env python3 +"""WebhookSelfTest (BPMN): structural check for a two-branch self-testing process. + +Ported from Flow `connector_trigger/webhook_waitfor_parallel.yaml`'s +`check_webhook_waitfor_parallel.py`. The process must be: + + manual start (bpmn:startEvent, no event definition) + -> fan-out into TWO branches + |-- branch 1: bpmn:receiveTask carrying the registry Intsvc.WaitForEvent + | wrapper, bound to the HTTP Webhook connector (connectorKey + | uipath-http-webhook) -> end + `-- branch 2: bpmn:sendTask carrying the registry Intsvc.HttpExecution + wrapper, manual (connectionless) GET whose url is that connection's + webhook URL, with nothing in headers or query parameters -> end + +Branch 2's GET hits the webhook URL, which delivers the event that completes +branch 1's wait -- so the process self-triggers at runtime. This checker +validates the static two-branch shape only; it does not run `bpmn debug`. + +Assertion map (Flow -> BPMN): + F check_webhook_waitfor_parallel.py:74-85 start trigger fans out into >=2 branches -> fan_out_point() + F check_webhook_waitfor_parallel.py:87-98 HTTP Webhook Wait-for-event node exists -> wait_for_event_nodes() + F check_webhook_waitfor_parallel.py:100-144 manual GET to webhook URL, no headers/query -> http_get_nodes() + F check_webhook_waitfor_parallel.py:146-157 both branch tails reach an End node -> reachable() vs end_ids + I locate/parse .bpmn -> parse_bpmn() + T Flow's `core.trigger.*` marker for "start trigger preserved" -> a manual bpmn:startEvent + (no event definition) -- the fan-out must originate downstream of it, not replace it. + T Flow's raw start-trigger fan-out (>=2 outgoing edges straight off the trigger node) -> + the documented BPMN construct for a parallel fork (structural-bpmn.md "Parallel (AND): + fork = one in, many out"): accept either (a) the manual start itself carrying >=2 + outgoing sequence flows, or (b) a bpmn:parallelGateway downstream of the manual start + with >=2 outgoing sequence flows -- whichever the skill's authoring produces, mirroring + the same curated-or-generic dual tolerance the batch's connector checks use elsewhere. + T Flow's node-`type` substring match (`uipath.connector.event` + `uipath-http-webhook`) -> + bpmn:receiveTask carrying the registry Intsvc.WaitForEvent wrapper (uipath:event, per + references/registry-workflow.md's OOTB extension-type table) with context connectorKey + == "uipath-http-webhook" -- the same connector-key literal Flow's own node-type marker + embeds. + T Flow's `core.action.http` / `core.action.http.v2` manual-GET-to-webhook-URL node shape -> + bpmn:sendTask carrying the registry Intsvc.HttpExecution wrapper (uipath:activity, per + the registry's own xmlTemplate: context fields mode/method/url/headers/parameters/body) + with mode=manual, method=GET, url containing "webhook", and no populated "headers" or + "parameters" context field -- the same "nothing in headers or query" rule Flow enforced, + read from the BPMN context fields instead of a JSON body dict. + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding check (not in Flow; the + `bpmn validate` criterion in the task YAML covers structure, and this connector has + no real tenant connection id to leak in a draft-authoring eval) +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) +from _shared.bpmn_check import ( # noqa: E402 + NS, + attr, + elements, + fail, + has_typed_uipath_extension, + parse_bpmn, +) +from _shared.graph import reachable # noqa: E402 + +BPMN_NS = NS["bpmn"] +WAIT_TYPE = "Intsvc.WaitForEvent" +HTTP_TYPE = "Intsvc.HttpExecution" +CONNECTOR_KEY = "uipath-http-webhook" + + +def manual_start_events(root: ET.Element) -> list[ET.Element]: + """bpmn:startEvent with no event-definition child -- the closest BPMN analog + of Flow's `core.trigger.*` marker: a plain, non-connector manual start.""" + out = [] + for s in elements(root, "startEvent"): + if not any( + child.tag.startswith(f"{{{BPMN_NS}}}") and child.tag.endswith("EventDefinition") + for child in s + ): + out.append(s) + return out + + +def node_inputs(el: ET.Element) -> list[ET.Element]: + return el.findall(".//uipath:input", NS) + + +def context_value(el: ET.Element, name: str) -> str: + for inp in node_inputs(el): + if inp.attrib.get("name") == name: + return inp.attrib.get("value") or (inp.text or "") + return "" + + +def _is_empty_json_field(value: str) -> bool: + """True when a headers/parameters context field carries nothing -- absent, + blank, `{}`/`[]`, or `null` (an unfilled registry template placeholder is + not the same as a populated one).""" + text = (value or "").strip() + if not text: + return True + if text.lower() == "null": + return True + try: + parsed = json.loads(text) + except json.JSONDecodeError: + return False # non-empty, non-JSON text is a populated value + if isinstance(parsed, dict): + return not parsed + if isinstance(parsed, list): + return not parsed + return False + + +def fan_out_point(root: ET.Element, start_id: str) -> str | None: + """Id of the node from which >=2 independent branches originate: either the + manual start itself, or a bpmn:parallelGateway downstream of it (the + documented parallel-fork construct).""" + downstream = reachable(root, start_id) + candidates = [start_id, *(attr(gw, "id") for gw in elements(root, "parallelGateway"))] + for node_id in candidates: + if node_id != start_id and node_id not in downstream: + continue + out_flows = [f for f in elements(root, "sequenceFlow") if attr(f, "sourceRef") == node_id] + if len(out_flows) >= 2: + return node_id + return None + + +def wait_for_event_nodes(root: ET.Element) -> list[ET.Element]: + """bpmn:receiveTask carrying the registry Intsvc.WaitForEvent wrapper, + bound to the HTTP Webhook connector.""" + out = [] + for task in elements(root, "receiveTask"): + if not has_typed_uipath_extension(task, "event", WAIT_TYPE): + continue + if context_value(task, "connectorKey") != CONNECTOR_KEY: + continue + out.append(task) + return out + + +def http_get_nodes(root: ET.Element) -> list[ET.Element]: + """bpmn:sendTask carrying Intsvc.HttpExecution, manual GET to a webhook + URL, with nothing in headers or query parameters.""" + good = [] + for task in elements(root, "sendTask"): + if not has_typed_uipath_extension(task, "activity", HTTP_TYPE): + continue + mode = context_value(task, "mode").lower() + method = context_value(task, "method").upper() + url = context_value(task, "url") + if mode != "manual" or method != "GET": + continue + if "webhook" not in url.lower(): + continue + headers = context_value(task, "headers") + params = context_value(task, "parameters") + if not _is_empty_json_field(headers): + fail(f"HTTP node must not set headers; found: {headers!r}") + if not _is_empty_json_field(params): + fail(f"HTTP node must not set query parameters; found: {params!r}") + good.append(task) + return good + + +def main() -> None: + path, root = parse_bpmn("WebhookSelfTest") + + starts = manual_start_events(root) + if not starts: + fail( + "no manual bpmn:startEvent (no event definition) -- the wait-for-event and " + "HTTP-request branches must fan out from the manual start, not replace it" + ) + start_id = attr(starts[0], "id") + + if fan_out_point(root, start_id) is None: + fail( + "manual start does not fan out into >=2 branches -- expected either the start " + "event itself or a downstream bpmn:parallelGateway to carry >=2 outgoing " + "sequence flows for the wait-for-event and HTTP-request branches" + ) + print(f"OK: manual start {start_id!r} fans out into >=2 branches") + + event_nodes = wait_for_event_nodes(root) + if not event_nodes: + fail( + f"no bpmn:receiveTask carrying {WAIT_TYPE} bound to connectorKey " + f"{CONNECTOR_KEY!r} (HTTP Webhook wait-for-event)" + ) + print("OK: HTTP Webhook wait-for-event receiveTask present") + + http_nodes = http_get_nodes(root) + if not http_nodes: + fail( + f"no bpmn:sendTask carrying {HTTP_TYPE} configured as a manual GET to the " + "webhook URL (mode=manual, method=GET, url containing 'webhook', " + "no populated headers/parameters context field)" + ) + print("OK: manual GET HttpExecution sendTask to webhook URL, no headers/query") + + end_ids = {attr(e, "id") for e in elements(root, "endEvent")} + if not end_ids: + fail("no end event") + + for label, node in (("wait-for-event", event_nodes[0]), ("http-request", http_nodes[0])): + node_id = attr(node, "id") + reach = reachable(root, node_id) + if not (reach & end_ids): + fail(f"{label} branch does not reach an end event") + print("OK: both branches terminate at an end event") + + print( + f"OK: {path} fans a manual start into an HTTP Webhook wait-for-event branch " + "and a manual-GET-to-webhook-URL branch, both terminating at an end event" + ) + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml new file mode 100644 index 0000000000..5a3f4fa870 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml @@ -0,0 +1,100 @@ +task_id: skill-bpmn-datafabric-smoke-error +description: > + Smoke: attempt a record creation against a non-existent entity — expect a + structured 4xx and zero side effects on FlowCodeEvalEntity. Static-validate + only, same as Flow's own grader (its docstring notes the check is + entity-binding only, topology is not parsed): this port does not run + `bpmn debug` to observe the runtime 4xx or replay a live re-read of + FlowCodeEvalEntity. + Ported from Flow `connector_features/datafabric_connector/smoke_error.yaml`; + connector activities are modeled as bpmn:sendTask nodes carrying the registry + Intsvc.ActivityExecution wrapper instead of Flow connector nodes, and the + grader accepts either the curated per-operation objectName or the generic + entity-CRUD form the connector skill may emit for the same operation. +tags: [uipath-maestro-bpmn, smoke, "mode:build", "lifecycle:generate", connector, negative, uipath-uipath-dataservice] + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../../_setup + mount_point: _setup + - type: template_dir + path: ../_setup + mount_point: _setup + +run_limits: + max_turns: 100 + turn_timeout: 1500 + task_timeout: 2400 + +reference: + directory: ../../.. + +pre_run: + - command: 'python3 "_setup/preflight_connections.py" uipath-uipath-dataservice' + timeout: 120 + - command: 'python3 "_setup/ensure_entity.py" "_setup/flow_code_eval_entity.entity.json"' + timeout: 60 + +initial_prompt: | + Build this with the UiPath Data Service Integration Service connector + (`uipath-uipath-dataservice`) activities for every entity operation. + + Build a UiPath Maestro BPMN process that: + + 1. Attempts to create a record on entity 'NonExistentEntity' with + title='ErrorTestRecord', score=5.0. Expect an error — do not swallow it. + 2. In a parallel branch (so it runs even after the failure), Query + FlowCodeEvalEntity for title='ErrorTestRecord'. + 3. In the same parallel branch, Query FlowCodeEvalEntity with no filter. + + Use the Data Service Integration Service connector activities for these + operations. + + The process is not complete until `uip maestro bpmn validate` passes. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: command_executed + description: "Advisory: agent queried the registry for Create Entity Record on the non-existent entity" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+maestro\s+bpmn\s+registry\s+get\s+Intsvc\.ActivityExecution\b[^\n]*--object-name\s+"?(CreateEntityRecordCurated|CreateEntityRecord_V3|NonExistentEntity)' + min_count: 1 + weight: 1.5 + pass_threshold: 0.0 + - type: command_executed + description: "Advisory: agent queried the registry for the follow-up Query Entity Records nodes" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+maestro\s+bpmn\s+registry\s+get\s+Intsvc\.ActivityExecution\b[^\n]*--object-name\s+"?(QueryEntityRecordsCurated|QueryEntityRecords_V3|FlowCodeEvalEntity)' + min_count: 1 + weight: 2.0 + pass_threshold: 0.0 + - type: command_executed + description: "BPMN file was validated" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+maestro\s+bpmn\s+validate' + min_count: 1 + weight: 3.0 + pass_threshold: 1.0 + - type: run_command + description: "Create targets NonExistentEntity; Queries target FlowCodeEvalEntity" + command: "python3 $REFERENCE_DIR/_shared/check_df_smoke_error.py" + timeout: 30 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml new file mode 100644 index 0000000000..3f428821f4 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml @@ -0,0 +1,121 @@ +task_id: skill-bpmn-generic-dynamic-node +description: > + Connector feature: validate a generic (dynamic) connector node end-to-end. A + generic activity encodes only the operation in its registry type; the object + is supplied dynamically at authoring time, so the agent must resolve the + object name and set it via `registry get Intsvc.ActivityExecution + --object-name `. Uses ServiceNow's generic "List All Records" + activity (API `objectName: "acr_user"`) as the concrete generic activity, + bound to the tenant's ServiceNow connection, then runs `bpmn debug`. + Exercises connector discovery (including connections in non-default + folders), generic-activity object-name resolution, connection binding, and + live execution. The `acr_user` table is empty in the codereval tenant, so + the call legitimately returns `[]`; the checker asserts the node is a + generic list activity (an `Intsvc.ActivityExecution` node with + `objectName: "acr_user"` and operation/method classifying it as List, the + registry's only form for this activity — it has no curated alternative), + `FinalStatus: "Completed"` with no incidents, and an array-typed output + (empty allowed) rather than requiring rows. + Connection note: ServiceNow developer instances hibernate after inactivity. + If a run fails with a connection error, log into the ServiceNow account to + wake the instance, then re-run. + + Ported from Flow `connector_features/generic_dynamic_node/generic_dynamic_node.yaml`; + the connector node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper (context fields connectorKey/objectName/ + operation/method — registry-confirmed via `registry get ... --object-name + acr_user`) instead of a Flow connector node with a node-type slug, + `flow debug`'s single-call inline payload becomes an ephemeral-solution + import + `bpmn debug` + `debug-instance variables-all`/`incidents` read, and + the array-output check searches the root scope's Globals plus every + element's Outputs instead of one inline debug payload's flow-output + globals. The live-check criterion timeout is raised from Flow's 720s to + 1050s: the BPMN sequence adds ephemeral solution init/import and two + separate `debug-instance` reads that flow's single inline `flow debug` call + did not need (see check_generic_dynamic_node.py's budget comment) — the one + sanctioned deviation from "criteria identical" per LIVE-ADDENDUM, since the + budget is a property of the CLI surface, not of what is graded. +tags: [uipath-maestro-bpmn, e2e, "mode:operate", "lifecycle:generate", "shape:single-node", connector, uipath-servicenow-servicenow] + +agent: + type: claude-code + permission_mode: acceptEdits + allowed_tools: ["Skill", "Bash", "Read", "Write", "Edit", "Glob", "Grep"] + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +run_limits: + task_timeout: 2490 + max_turns: 120 + turn_timeout: 1200 + +reference: + directory: ../.. + +pre_run: + - command: 'python3 "_setup/preflight_connections.py" uipath-servicenow-servicenow' + timeout: 120 + +initial_prompt: | + Create a UiPath Maestro BPMN process project named "AcrUserList" inside a + solution of the same name, with a manual start. + The process must call ServiceNow to list all records in the Acr User + object, then surface the returned records as a process output variable. + + Bind it to the tenant's ServiceNow connection. Then debug the process to + confirm the connector call completes. + + Use the connection present in Shared/uipath-maestro-flow. + + Do NOT substitute a mock for the + connector. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. + +success_criteria: + # ── BPMN file validity ───────────────────────────────────────────────── + - type: run_command + description: "uip maestro bpmn validate passes on the BPMN file" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + # ── Agent invoked debug ──────────────────────────────────────────────── + - type: command_executed + description: "Advisory: live-v1 agent ran bpmn debug" + tool_name: "Bash" + command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+(maestro\s+)?bpmn\s+debug' + min_count: 1 + weight: 1.5 + pass_threshold: 0.0 + + # ── Execution: connector calls ServiceNow and surfaces an array output ── + - type: run_command + description: "ServiceNow list-all-records node on acr_user executes and surfaces an array output" + command: "python3 $REFERENCE_DIR/_shared/check_generic_dynamic_node.py" + timeout: 1050 + expected_exit_code: 0 + weight: 6.0 + pass_threshold: 1.0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml new file mode 100644 index 0000000000..18954a7fbc --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml @@ -0,0 +1,97 @@ +task_id: skill-bpmn-jdbc-databricks-query +description: > + Databricks-via-JDBC coverage (Maestro BPMN connector, special SDK case): + builds a Maestro BPMN process whose Execute Query Synchronously activity + (registry `Intsvc.ActivityExecution`, connectorKey `uipath-uipath-jdbc`, + objectName `query`) runs a complex aggregate SQL query (GROUP BY / HAVING / + AVG / ORDER BY — expressible only via raw SQL, not the generic record + activities) against the `employees` table on a Databricks database, + exposing the result as a process output. Grades structurally: the .bpmn + parses as XML and wires the JDBC gateway connector as a real bpmn:sendTask + with the registry Intsvc.ActivityExecution wrapper and the + ExecuteQuerySynchronously operation (objectName `query`, method `POST`) — + not the native Databricks connector, not an HTTP fallback. Exercises the + disambiguation that Databricks SQL has no native connector — it must + resolve to the JDBC gateway. Tenant prerequisite: a Databricks-backed + `uipath-uipath-jdbc` connection reaching `sandbox.globalmart` (employees + seeded per the spartacus connector-sanity fixture). REQUIRES cloud auth AND + that connection. + Ported from Flow + `connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml`; the + JSON node/edge walk becomes an XML walk over the registry-driven + Intsvc.ActivityExecution wrapper, and Flow's node-type-suffix disambiguation + guard becomes a connectorKey + objectName/method context-field match. +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, uipath-uipath-jdbc] +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +run_limits: + expected_turns: 40 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 900 + +reference: + directory: ../.. + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "DatabricksQuery" + inside a solution of the same name, with a manual start. + + It should run a SQL query against a Databricks database using the Database + Hub (JDBC) connector, and expose the query result as a process output. Run + an aggregate query over the `employees` table that groups by department, + keeps only departments with two or more employees, and orders them by + average salary. + + Discover the Databricks-backed Database Hub (JDBC) connection in the + `Shared/uipath-maestro-flow` folder from the tenant — do NOT hand-author or + invent a connection id. If the JDBC connector isn't available in your + registry, stop rather than falling back to a generic HTTP request. + + Validate the final .bpmn file and debug it. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: command_executed + description: "Advisory: live-v1 agent ran bpmn debug" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+(maestro\s+)?bpmn\s+debug' + min_count: 1 + weight: 1.5 + pass_threshold: 0.0 + + - type: run_command + description: "DatabricksQuery checks: valid BPMN wires the JDBC Execute-Query node with a bound connection, not the native Databricks connector or an HTTP fallback" + command: "python3 $REFERENCE_DIR/_shared/check_databricks_query.py" + timeout: 600 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml new file mode 100644 index 0000000000..0dedb79542 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml @@ -0,0 +1,98 @@ +task_id: skill-bpmn-slack-http-fallback +description: > + E2E test: a catalog connector (Slack, uipath-salesforce-slack) has no native + activity for "list a team's custom emoji" (Slack's emoji.list). The skill must + fall back to a connector-mode HTTP-request node that reuses the existing Slack + connection's managed auth, then the process must debug green. + Exercises the no-native-activity -> managed-HTTP fallback path end-to-end: + structural check confirms the fallback shape, runtime check confirms + `uip maestro bpmn debug` completes against the live Slack connection. + + Ported from Flow `connector_features/slack-http-fallback/slack_http_fallback.yaml`; + the HTTP-fallback node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper with connectorKey uipath-salesforce-slack + targeting the emoji.list endpoint (registry-workflow.md §"Connectionless vs + connector HTTP" -- see the grader's GUESS note on the open question of + whether a connector-authenticated Intsvc.HttpExecution form also exists), + `flow debug`'s single-call inline payload becomes an ephemeral-solution + import + `bpmn debug` + `debug-instance incidents` read, and the check_debug + criterion's timeout is raised from Flow's 720s to 930s to cover the extra + CLI steps (`solution init`/`solution projects import`, a separate + `incidents` read) that BPMN's live surface needs beyond Flow's single + `flow debug` call (arithmetic in the grader's docstring). +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, uipath-salesforce-slack, "feature:http"] + +run_limits: + expected_turns: 32 + task_timeout: 2400 + max_turns: 100 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +reference: + directory: ../.. + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "SlackEmojiListTest" + inside a solution of the same name, with a manual start that lists the custom + emoji for a team. Wire the manual start to the node that + lists the emoji. + + Use the Slack connection in the `Shared/uipath-maestro-flow` folder. + + Validate the final BPMN file. Once it validates, debug it. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: command_executed + description: "Advisory: live-v1 agent invoked bpmn debug" + tool_name: "Bash" + command_pattern: "(uip|\\$UIP)\\s+maestro\\s+bpmn\\s+debug" + min_count: 1 + weight: 1.5 + pass_threshold: 0.0 + + - type: run_command + description: "Emoji list falls back to an HTTP-request node bound to the Slack connector (no native activity exists for it) and targets the emoji.list endpoint" + command: "python3 $REFERENCE_DIR/_shared/check_slack_http_fallback.py check_fallback" + timeout: 30 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "uip maestro bpmn debug finishes with a completed final status and no incidents against the live Slack connection" + command: "python3 $REFERENCE_DIR/_shared/check_slack_http_fallback.py check_debug" + timeout: 930 + expected_exit_code: 0 + weight: 6.0 + pass_threshold: 1.0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/_setup/seed.py b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/_setup/seed.py new file mode 100644 index 0000000000..9c67f19034 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/_setup/seed.py @@ -0,0 +1,16 @@ +#!/usr/bin/env python3 +"""Shared seed for data-grounded TM connector CRUD tasks. +Writes seed.json (unique name + project) into the sandbox. Staged into the +sandbox via sandbox.template_sources and run as ./_setup/seed.py. +""" +import json +import os +import uuid + +seed = { + "name": f"DataEval-{uuid.uuid4().hex[:8]}", + "project_key": os.environ.get("TM_EVAL_PROJECT_KEY", "HEALTH"), +} +with open("seed.json", "w", encoding="utf-8") as fh: + json.dump(seed, fh) +print(f"seeded name: {seed['name']} (project {seed['project_key']})") diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml new file mode 100644 index 0000000000..f55d8e4d1a --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml @@ -0,0 +1,107 @@ +# Ported from Flow `connector_features/testmanager_crud_grounded.yaml`. +# Flow's task carries `skip: true` (pending agent-reliability confirmation). +# Per BATCH1-ADDENDUM.md: Test Manager `skip: true` is NOT carried over -- +# the codereval Test Manager connection exists, and a skipped port yields no +# signal -- this port runs so the BPMN suite measures it. +task_id: skill-bpmn-testmanager-crud-grounded +description: > + Data-grounded (Maestro BPMN): agent builds a process that creates a Test + Case via the uipath-uipath-testmanager connector node with a unique seeded + name, gets it back, outputs the retrieved name, and debugs (executes) the + process — then records the round-trip. Graded on the connector artifacts, + successful validation, and retrieved == created through the executed + process; registry-refresh telemetry is advisory. + Ported from Flow `connector_features/testmanager_crud_grounded.yaml`; the + connector node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper (create = TestCase/POST, get = + TestCase/GETBYID|GET) instead of Flow's two per-operation connector node + types, and Flow's `skip: true` is not carried over (see header comment). +tags: + - uipath-maestro-bpmn + - e2e + - "mode:build" + - "lifecycle:generate" + - "shape:multi-node" + - connector + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: _setup + mount_point: _setup + +run_limits: + task_timeout: 2400 + max_turns: 60 + turn_timeout: 1200 + +pre_run: + - command: "python3 _setup/seed.py" + timeout: 30 + +reference: + directory: ../.. + +initial_prompt: | + Build a UiPath Maestro BPMN process (inside a project of the same name) that uses the UiPath Test Manager Integration Service connector. + + Read seed.json for a unique `name` and a `project_key`, then: + 1. Create a test case with that exact name under that project. + 2. Read it back so the create and read actually execute, and capture the name that comes back. + 3. Record result.json with `created_name` (the name you sent), `retrieved_name` (the name returned), and `id`. + + The process is not complete until `uip maestro bpmn validate` passes. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: command_executed + description: "Advisory: live-v1 agent refreshed the node manifest" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+maestro\s+bpmn\s+registry\s+pull' + min_count: 1 + weight: 1.0 + pass_threshold: 0.0 + + - type: run_command + description: "The process contains Test Manager create and get connector nodes" + command: "python3 $REFERENCE_DIR/_shared/check_testmanager_crud_grounded.py --node-types" + timeout: 30 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 1.0 + + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 1.0 + + - type: run_command + description: "Execution produced the seeded create/get round-trip result" + command: "python3 $REFERENCE_DIR/_shared/check_testmanager_crud_grounded.py --execution-evidence" + timeout: 30 + expected_exit_code: 0 + weight: 1.5 + pass_threshold: 1.0 + + - type: run_command + description: "Data-grounded: retrieved name == created (seeded) name — real data round-tripped through the executed process" + command: "python3 $REFERENCE_DIR/_shared/check_testmanager_crud_grounded.py" + timeout: 30 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml b/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml new file mode 100644 index 0000000000..a6c4d931f9 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml @@ -0,0 +1,117 @@ +task_id: skill-bpmn-webhook-waitfor-parallel +description: > + E2E self-testing process: a manual start fans out into two parallel + branches. Branch 1 is a mid-flow Wait-for-event node (bpmn:receiveTask, + registry `Intsvc.WaitForEvent`) bound to the HTTP Webhook connector + (connectorKey `uipath-http-webhook`). Branch 2 is a Managed HTTP Request + (bpmn:sendTask, registry `Intsvc.HttpExecution`, manual mode, GET) whose + URL is the webhook URL of that same HTTP Webhook connection, with nothing + in headers or query. The GET self-delivers the event that completes the + wait at runtime. + Exercises the connector-trigger plugin's "Wait for events" variant, + parallel branch fan-out from the manual start, manual-mode HTTP, and the + CLI webhook-URL wiring. Validates the static two-branch shape only (no + `bpmn debug`). + Ported from Flow `connector_trigger/webhook_waitfor_parallel.yaml`; the + Flow JSON node/edge walk becomes an XML walk over the registry-driven + `Intsvc.WaitForEvent` / `Intsvc.HttpExecution` wrappers, and the trigger's + raw outgoing-edge fan-out becomes a manual-start-or-parallelGateway fan-out + (BPMN forks via a dedicated gateway node; either shape is accepted). +tags: + [ + uipath-maestro-bpmn, + integration, + "mode:build", + "lifecycle:generate", + "shape:multi-node", + connector, + "feature:trigger", + "feature:http", + wait-for-event, + uipath-http-webhook, + ] + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +run_limits: + expected_turns: 35 + turn_timeout: 1200 + +reference: + directory: ../.. + +initial_prompt: | + Create a UiPath Maestro BPMN process named "WebhookSelfTest" with a manual + start. + + The manual start must fan out into TWO parallel branches: + + - Branch 1: a mid-flow node that waits for an HTTP Webhook + event (the HTTP Webhook connector). For its connection, use the HTTP + Webhook connection in the `Shared/uipath-maestro-flow` folder. Then End + this branch. + + - Branch 2: a Managed HTTP Request node configured as a GET request + whose URL is the webhook URL of that same HTTP Webhook connection. + Do NOT put anything in the headers or the query parameters — just the GET to + the webhook URL. Then End this branch. + + After building, validate the process. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: command_executed + description: "Solution-creation telemetry (report only; the BPMN artifact is graded below)" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+(solution\s+(new|init)|maestro\s+bpmn\s+init)' + min_count: 1 + weight: 1.0 + pass_threshold: 0.0 + + - type: run_command + description: "WebhookSelfTest BPMN artifact exists" + command: "find . -iname 'WebhookSelfTest*.bpmn' -not -path '*/node_modules/*' 2>/dev/null | grep -q ." + timeout: 30 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 1.0 + + - type: command_executed + description: "Webhook-URL retrieval telemetry (report only; the final artifact grades URL wiring)" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+is\s+webhooks\s+config' + min_count: 1 + weight: 1.0 + pass_threshold: 0.0 + + - type: run_command + description: "Process validates successfully" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 1.5 + pass_threshold: 1.0 + + - type: run_command + description: "Two-branch structure: wait-for-event + manual-GET-to-webhook-URL (no headers/query), both reaching End" + command: "python3 $REFERENCE_DIR/_shared/check_webhook_waitfor_parallel.py" + timeout: 120 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/jira_is.py new file mode 100644 index 0000000000..e3e92f5589 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/jira_is.py @@ -0,0 +1,144 @@ +#!/usr/bin/env python3 +"""Minimal live Jira helper for this task — wraps `uip is resources run`. + +Self-contained (no shared module). Assumes `uip` is on PATH and logged in and +that the connection + CE project exist — this is a tenant-gated e2e task. + +The connection is scoped to the curated single-record ops, so we create by +body / get by id / delete by id — never a JQL search. +""" + +from __future__ import annotations + +import json +import re +import subprocess + +CONNECTOR = "uipath-atlassian-jira" +FOLDER_PATH = "Shared/uipath-maestro-flow" +FOLDER_NAME = "uipath-maestro-flow" # leaf of FOLDER_PATH, as reported by connections list +CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" +PROJECT_KEY = "CE" # "Coder Eval" project on uipath-sandbox-380 +ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project + + +def _issue_not_found(env: dict) -> bool: + """True only for an ISSUE-SPECIFIC not-found from the Jira operation — proof the + requested issue key is absent. Requires BOTH a structured HTTP 404 AND an + issue-scoped signal (the provider phrases a missing issue as e.g. "Issue does + not exist ..."). A bare ``"404"``/``"not found"`` substring, or a 404 that + refers to a missing connection/activity/other prerequisite (no issue mention), + is NOT accepted — so a prerequisite failure can't masquerade as a confirmed + issue deletion. Being strict is the safe direction here: a false negative only + makes teardown print WARN, while a false positive would leak the CE issue.""" + blob = json.dumps(env) + structured_404 = bool( + re.search(r"status code ['\"]?404\b", blob, re.I) + or re.search(r'"providerErrorCode"\s*:\s*404\b', blob) + or re.search(r'"statusCode"\s*:\s*"?404\b', blob) + ) + # Require the PROVIDER's error message to say the issue is absent — NOT a bare + # `issueId` token, which every get/delete request echoes in its own query and so + # would also appear on a connection/activity 404. Only Jira's own "Issue does + # not exist"-style message proves the requested issue key is gone. + issue_specific = bool( + re.search(r"issue\s+(does\s+not\s+exist|not\s+found|no\s+longer\s+exists|is\s+not\s+found)", blob, re.I) + or re.search(r"(does\s+not\s+exist|not\s+found|no\s+longer\s+exists).{0,40}\bissue\b", blob, re.I) + ) + return structured_404 and issue_specific + + +def _run(*args: str) -> dict: + out = subprocess.run( + ["uip", *args, "--output", "json"], + capture_output=True, text=True, timeout=120, + ).stdout + # Tolerate diagnostic/log lines the CLI may print before the JSON envelope. + i = out.find("{") + return json.loads(out[i:] if i > 0 else out) + + +def connection_id() -> str: + # Resolve by (name, folder) across all folders — avoids depending on + # `uip or folders get` (which can return a Failure envelope in CI) while + # still scoping to the target folder so a same-named connection in another + # folder can't be picked by accident. + conns = _run("is", "connections", "list", CONNECTOR, "--all-folders", "--refresh")["Data"] + by_name = [c for c in conns if c["Name"] == CONNECTION_NAME] + scoped = [c for c in by_name if c.get("Folder") == FOLDER_NAME] + if scoped: + return scoped[0]["Id"] + # Only accept a name-only match when NO candidate reports folder metadata + # (older CLI / env). If folders ARE reported but none is the target folder, + # refuse to guess — a same-named connection elsewhere could be the wrong + # Jira account. Fail the prerequisite instead. + if any(c.get("Folder") for c in by_name): + raise SystemExit( + f"FAIL: no {CONNECTOR} connection named {CONNECTION_NAME!r} in folder " + f"{FOLDER_NAME!r}; candidates in folders {[c.get('Folder') for c in by_name]}" + ) + if not by_name: + raise SystemExit(f"FAIL: no {CONNECTOR} connection named {CONNECTION_NAME!r}") + return by_name[0]["Id"] + + +def myself(conn_id: str) -> str: + """Return the connection user's Atlassian accountId.""" + return _run( + "is", "resources", "run", "get", CONNECTOR, "myself", + "--connection-id", conn_id, + )["Data"]["accountId"] + + +def create_issue(conn_id: str, summary: str) -> str: + body = {"fields": {"project": {"key": PROJECT_KEY}, "issuetype": {"id": ISSUETYPE_ID}, "summary": summary}} + return _run( + "is", "resources", "run", "create", CONNECTOR, "curated_create_issue", + "--connection-id", conn_id, "--body", json.dumps(body), + )["Data"]["key"] + + +def get_issue(conn_id: str, key: str) -> dict | None: + """Return the issue's `fields` dict, or None if it doesn't exist (404).""" + env = _run( + "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", + "--connection-id", conn_id, + "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", + ) + if env.get("Result") == "Failure": + return None + return env["Data"].get("fields", {}) + + +def issue_absent(conn_id: str, key: str) -> bool: + """True ONLY when a tenant read CONFIRMS the issue does not exist (not-found / + 404). False when it exists OR when the read itself failed (transient 5xx / + auth) — so teardown never treats an ambiguous read as proof of deletion. + Distinct from :func:`get_issue`, which collapses every failure to ``None``.""" + env = _run( + "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", + "--connection-id", conn_id, + "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", + ) + if str(env.get("Result", "")).lower() != "failure": + return False # a successful read means the issue still exists + return _issue_not_found(env) + + +def delete_issue(conn_id: str, key: str) -> bool: + """Delete an issue by key. Returns True only when deletion is CONFIRMED — + either a success envelope, or a not-found/404 (already gone). Returns False + for any other Failure envelope (transient 5xx / auth) so the caller can retry + or report instead of silently leaking the issue in the shared CE project.""" + env = _run( + "is", "resources", "run", "delete", CONNECTOR, "issue", + "--connection-id", conn_id, "--query", f"issueId={key}", + # The CLI never prompts and REFUSES an irreversible delete without this + # flag ("Confirmation required … Re-run with --yes"). Without it every + # teardown since 08-19 printed WARN and left its ticket in the CE project. + "--yes", + ) + if str(env.get("Result", "")).lower() != "failure": + return True + # A Failure envelope: only a structured 404 (issue absent) is a confirmed gone. + return _issue_not_found(env) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/seed.py b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/seed.py new file mode 100644 index 0000000000..dc6a603af4 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/seed.py @@ -0,0 +1,48 @@ +#!/usr/bin/env python3 +"""pre_run: seed a Sev1 escalation case for the Jira-ticket outcome eval. + +No live issue is created here — the agent's flow creates it when the grader runs +`flow debug`. The unique tag lives in the correlationId, which the flow must echo +into the Jira issue summary, so the check can prove the ticket the flow created +(not a fabricated key) really exists in Jira. +""" + +from __future__ import annotations + +import json +import secrets +from pathlib import Path + +import jira_is + +tag = secrets.token_hex(4) +correlation = f"ESC-JIRA-{tag}" +seed = { + "tag": tag, + "project_key": jira_is.PROJECT_KEY, + "issuetype_id": jira_is.ISSUETYPE_ID, + # Jira's design-time schema marks fields.reporter.id as required for this + # project/issue type; pin it so agents never have to guess or ask. + "reporter_id": jira_is.myself(jira_is.connection_id()), + "correlationId": correlation, + "inputs": { + "senderEmail": "jane.doe@acmecorp.com", + "senderDomain": "acmecorp.com", + "subject": "Production down: checkout API returning 500s", + "body": "Critical urgent outage, all users blocked.", + "customerTier": "Enterprise", + "productionDown": True, + "workaroundAvailable": False, + "hasAttachments": False, + "customerMatchStatus": "single", + "isDuplicate": False, + "correlationId": correlation, + }, + "expected": {"severity": "Sev1", "caseKey": correlation}, + # Classification the Script must compute but the prompt does not map to a named + # End out — verified against the Script node's intermediate output. Sev1 (prod + # down, no workaround) ⇒ engineeringNeeded true. + "expected_script": {"engineeringNeeded": True}, +} +Path("seed.json").write_text(json.dumps(seed, indent=2) + "\n", encoding="utf-8") +print(f"OK: wrote seed (correlationId={correlation}, project={jira_is.PROJECT_KEY}, issuetype={jira_is.ISSUETYPE_ID})") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/teardown_jira.py new file mode 100644 index 0000000000..2e744bb3a4 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/_setup/teardown_jira.py @@ -0,0 +1,30 @@ +#!/usr/bin/env python3 +"""post_run: delete every issue the run created (keys in `.created_keys`). +Idempotent and never fails the task.""" + +import sys +from pathlib import Path + +import jira_is + +try: + kf = Path(".created_keys") + keys = kf.read_text().split() if kf.is_file() else [] + if keys: + conn = jira_is.connection_id() + for key in keys: + # Verify the delete actually happened; retry once on an unconfirmed + # (transient) failure, then confirm via a tenant reread before giving + # up. Only claim success on a confirmed deletion / not-found. + ok = jira_is.delete_issue(conn, key) + if not ok: + ok = jira_is.delete_issue(conn, key) + if not ok and jira_is.issue_absent(conn, key): + ok = True # tenant read CONFIRMS a 404 (not just an ambiguous failure) + print(f"OK: deleted {key}" if ok + else f"WARN: could NOT confirm deletion of {key} — may be leaked in CE project") + else: + print("OK: nothing to delete") +except Exception as e: # noqa: BLE001 — teardown must not fail the task + print(f"WARN: teardown ignored error: {e}") +sys.exit(0) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml new file mode 100644 index 0000000000..83999d87a0 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml @@ -0,0 +1,142 @@ +task_id: skill-bpmn-e2e-escalation-jira-ticket +description: > + E2E live Jira coverage for the escalation BPMN process — the agent builds a + manual-start escalation-triage UiPath Maestro BPMN process that classifies + severity and creates a real Jira ticket for the escalation. The grader seeds + a Sev1 case (unique correlationId), imports the exact submitted project into + an ephemeral solution, runs `uip maestro bpmn debug --inputs`, and verifies + the OUTCOME by re-reading the created key from Jira: the issue exists and + its summary carries the seeded correlationId — proof the process created + THIS run's ticket, not a fabricated key. The created issue is cleaned up in + post_run. A manual start is used because the Outlook email-received trigger + is not reliably debug-testable (see outlook_trigger_inbox / customer_escalation_triage). + + Tenant prerequisite: a `uipath-atlassian-jira` connection in folder + `Shared/uipath-maestro-flow` reaching the `CE` / "Coder Eval" project (issue + type Task). Targets live in the task's `_setup/jira_is.py`. + + Ported from Flow `uipath-maestro-flow/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml`; + same scenario and grading, translated to the BPMN CLI surface (ephemeral + solution import, sha256-pinned, plus `debug-instance variables-all`/`incidents` + in place of Flow's inline debug payload) following + `e2e/customer_escalation_triage/`'s live pattern, adapted to grade only + Flow's Jira assertions (no Slack step — Flow's task has none either). +tags: [uipath-maestro-bpmn, e2e, mode:build, lifecycle:generate, shape:multi-node, connector, feature:escalation, path-to-ga, outcome-graded] + +run_limits: + expected_turns: 55 + max_turns: 120 + # turn_timeout kept at Flow's verbatim 1200s. Grading sum: command_executed + # (no timeout: field -> priced at the suite's 30s default) + the validate + # loop (180s) + the behavior criterion (1600s, see check_escalation_jira_ticket.py + # for why Flow's own 1600s already covers the added BPMN plumbing) = 1810s. + # task_timeout must cover turn_timeout + grading (1200 + 1810 = 3010s); raised + # to 3100s for headroom. This is the one sanctioned run_limits deviation + # (LIVE-ADDENDUM: "run_limits.task_timeout must cover the agent's turn budget + # plus all grading"). + task_timeout: 3100 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + - type: template_dir + path: _setup + mount_point: _setup + +pre_run: + - command: "python3 _setup/seed.py" + timeout: 60 + +reference: + directory: ../.. + +post_run: + # Repo-standard solution sweep first (both the agent's own build solution + # and the grader's ephemeral live-eval solution live under the sandbox CWD). + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 + - command: "python3 _setup/teardown_jira.py" + timeout: 120 + +initial_prompt: | + Create a UiPath Maestro BPMN process named "EscalationJiraTicket" with a + MANUAL start inside a solution of the same name. + It triages a customer escalation and creates a Jira ticket for it. + + Read `seed.json` in the current directory. Declare process inputs matching + the keys under its `inputs`: senderEmail, senderDomain, subject, body, + customerTier, productionDown (boolean), workaroundAvailable (boolean), + hasAttachments (boolean), customerMatchStatus, isDuplicate (boolean), + correlationId. + + Steps: + 1. Classify severity from the inputs into a declared `severity` variable + (Sev1 = Enterprise tier AND productionDown AND NOT workaroundAvailable; + Sev2 = productionDown AND workaroundAvailable; Sev3 otherwise) plus + `engineeringNeeded`. Use the classification node the skill documents as + executing at runtime. + 2. Create a Jira issue using the Atlassian Jira "Create Issue" connector + activity, authored from its registry template and Integration Service + definition. Use the `project_key`, `issuetype_id`, and `reporter_id` + from `seed.json` (`reporter_id` is the issue's reporter field). The + issue summary MUST include the `correlationId` verbatim (e.g. + "[] "). Use the Atlassian Jira + connection in the `Shared/uipath-maestro-flow` folder. If the Atlassian + Jira connector isn't in your registry, STOP rather than falling back to + a generic HTTP request. + 3. End the process, exposing these public outputs: severity, caseKey (= + the incoming correlationId), and jiraIssueKey (the created issue's + key). + + Validate the BPMN. Do NOT run or debug the process — the grader executes + it with the seeded inputs. Before starting, load the uipath-maestro-bpmn + skill and follow its workflow. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: command_executed + description: "Agent validated the generated BPMN" + tool_name: "Bash" + # `bpmn` exists only under `maestro` -- `uip bpmn validate` is an unknown + # command, so the `maestro` segment must not be optional. + command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+maestro\s+bpmn\s+validate' + min_count: 1 + weight: 1.5 + pass_threshold: 1.0 + + # Discover the BPMN file — the prompt names only the PROCESS, never the + # enclosing solution, so a hardcoded `Solution//.bpmn` + # would grade the solution directory's name (an unstated requirement) and + # score 0.0 on a valid process built at e.g. + # `EscalationJiraTicket/EscalationJiraTicket.bpmn` (no wrapper solution + # folder). Mirrors Flow's discovery rationale (validate_flow.py). + - type: run_command + description: "Generated EscalationJiraTicket BPMN validates without errors" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "Seeded Sev1 debug run creates a real Jira ticket whose summary carries the seeded correlationId (verified by re-reading the tenant)" + command: "python3 $REFERENCE_DIR/_shared/check_escalation_jira_ticket.py" + timeout: 1600 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/test_jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/test_jira_is.py new file mode 100644 index 0000000000..d6ff7353bf --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/test_jira_is.py @@ -0,0 +1,71 @@ +"""Teardown contract for the escalation task's Jira helper.""" + +from __future__ import annotations + +import importlib.util +import json +from pathlib import Path +from types import ModuleType + +import pytest + +HERE = Path(__file__).parent + + +def _load_module() -> ModuleType: + spec = importlib.util.spec_from_file_location("escalation_jira_is", HERE / "_setup" / "jira_is.py") + assert spec and spec.loader + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +class Result: + def __init__(self, envelope: object, returncode: int = 0) -> None: + self.stdout = json.dumps(envelope) + self.stderr = "" + self.returncode = returncode + + +def _stub(monkeypatch: pytest.MonkeyPatch, jira_is: ModuleType, envelope: object) -> list[list[str]]: + seen: list[list[str]] = [] + + def fake_run(cmd, **kwargs): + seen.append(list(cmd)) + return Result(envelope) + + monkeypatch.setattr(jira_is.subprocess, "run", fake_run) + return seen + + +def test_delete_issue_confirms_and_reports_success(monkeypatch: pytest.MonkeyPatch) -> None: + """The CLI refuses an unconfirmed delete with a Failure envelope + ("Confirmation required … Re-run with --yes"), which `delete_issue` + correctly reports as NOT confirmed — so every run 08-19 → 09-01 printed + `WARN: could NOT confirm deletion` and left its ticket in CE. `--yes` is + the missing word.""" + jira_is = _load_module() + seen = _stub(monkeypatch, jira_is, {"Result": "Success", "Data": {"Value": ""}}) + assert jira_is.delete_issue("conn", "CE-1257") is True + (cmd,) = seen + assert cmd[:5] == ["uip", "is", "resources", "run", "delete"] + assert "--yes" in cmd and "issueId=CE-1257" in cmd + + +def test_delete_issue_unconfirmed_failure_is_not_a_deletion(monkeypatch: pytest.MonkeyPatch) -> None: + jira_is = _load_module() + _stub(monkeypatch, jira_is, { + "Result": "Failure", + "Message": "Confirmation required: this will delete resource 'issue' and cannot be undone.", + "Instructions": "Re-run with --yes to confirm.", + }) + assert jira_is.delete_issue("conn", "CE-1257") is False + + +def test_delete_issue_structured_404_counts_as_gone(monkeypatch: pytest.MonkeyPatch) -> None: + jira_is = _load_module() + _stub(monkeypatch, jira_is, { + "Result": "Failure", + "Message": "Request failed with status code '404': Issue does not exist or you do not have permission to see it.", + }) + assert jira_is.delete_issue("conn", "CE-1257") is True diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/_setup/seed.py b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/_setup/seed.py new file mode 100644 index 0000000000..9c25fdde89 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/_setup/seed.py @@ -0,0 +1,116 @@ +#!/usr/bin/env python3 +"""Seed path cases for the escalation-orchestrator outcome eval. + +Each case pins the flow inputs that steer one branch and the outputs the grader +asserts. `expect_slack: true` means the run must have actually posted a Slack +message (escalation alert on the escalation path, triage notice on the triage +paths) — a non-empty slackMessageId proves it. + +Every expected value below was verified by running the reference orchestrator +through `uip maestro flow debug --inputs` on codereval/alpha. The checker +iterates this list generically, so new paths (e.g. multiple-match, missing-domain, +Sev2-with-attachments) drop in here with no checker change. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from uuid import uuid4 + +_COMMON = { + "senderEmail": "jane.doe@acmecorp.com", + "senderDomain": "acmecorp.com", + "hasAttachments": False, +} + + +def _case(name, run_id, suffix, overrides, expected, expect_slack): + corr = f"ORCH-{run_id}-{suffix}" + inputs = { + **_COMMON, + "subject": "", + "body": "", + "customerTier": "Standard", + "productionDown": False, + "workaroundAvailable": False, + "customerMatchStatus": "single", + "isDuplicate": False, + "correlationId": corr, + **overrides, + } + return { + "name": name, + "expect_slack": expect_slack, + "inputs": inputs, + "expected": {**expected, "caseKey": corr}, + } + + +def build_seed() -> dict: + r = uuid4().hex[:12] + cases = [ + # ── Escalation paths (Slack escalation alert) ────────────────────── + _case( + # Deliberately STANDARD tier: the orchestrator's Sev1 is tier-independent + # (productionDown AND NOT workaroundAvailable). A classifier that wrongly + # gates Sev1 on Enterprise (e.g. copied from the slack_alert task) would + # misclassify this Standard outage and fail. + "sev1-standard-production-down", r, "SEV1", + {"subject": "Production down: checkout 500s", "body": "Critical urgent outage, all users blocked", + "customerTier": "Standard", "productionDown": True, "workaroundAvailable": False}, + {"escalationPath": "escalation", "severity": "Sev1", "engineeringNeeded": True, "responseMode": "Draft"}, + True, + ), + _case( + "sev2-degraded-with-workaround", r, "SEV2", + {"subject": "Degraded checkout", "body": "Slow but a workaround exists", + "productionDown": True, "workaroundAvailable": True}, + {"escalationPath": "escalation", "severity": "Sev2", "engineeringNeeded": True, "responseMode": "Draft"}, + True, + ), + _case( + "sev3-no-production-impact", r, "SEV3", + {"subject": "Report formatting off", "body": "The export looks wrong"}, + {"escalationPath": "escalation", "severity": "Sev3", "engineeringNeeded": False, "responseMode": "Draft"}, + True, + ), + # ── Triage paths (Slack triage notice) ───────────────────────────── + _case( + "duplicate-escalation", r, "DUP", + {"subject": "Prod down", "body": "urgent", "customerTier": "Enterprise", + "productionDown": True, "workaroundAvailable": False, "isDuplicate": True}, + {"escalationPath": "duplicate", "severity": "informational", "engineeringNeeded": False, "responseMode": "None"}, + True, + ), + _case( + "unknown-customer", r, "UNK", + {"subject": "Help", "body": "an issue", "customerMatchStatus": "none"}, + {"escalationPath": "unknown_customer", "severity": "informational", "engineeringNeeded": False, "responseMode": "None"}, + True, + ), + _case( + "missing-domain", r, "MD", + {"subject": "Help", "body": "an issue", "senderDomain": ""}, + {"escalationPath": "missing_domain", "severity": "informational", "engineeringNeeded": False, "responseMode": "None"}, + True, + ), + _case( + "multiple-matches", r, "MULTI", + {"subject": "Help", "body": "an issue", "customerMatchStatus": "multiple"}, + {"escalationPath": "multiple_matches", "severity": "informational", "engineeringNeeded": False, "responseMode": "None"}, + True, + ), + ] + return {"run_id": r, "cases": cases} + + +def main() -> None: + seed = build_seed() + path = Path("seed.json") + path.write_text(json.dumps(seed, indent=2) + "\n", encoding="utf-8") + print(f"seeded {path} with {len(seed['cases'])} path cases") + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml new file mode 100644 index 0000000000..259b46b71c --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml @@ -0,0 +1,160 @@ +task_id: skill-bpmn-e2e-escalation-orchestrator-paths +description: > + End-to-end, outcome-based test of the customer-escalation orchestration, driven + down each branch by seeded inputs. The agent builds a manual-start BPMN process + (the Outlook email-received trigger is not reliably debug-testable — see + outlook_trigger_inbox / customer_escalation_triage notes) whose branching is + input-driven so the grader can steer any path via `bpmn debug --inputs`. The + grader runs seven seeded cases — Sev1/Sev2/Sev3 escalations and all four triage + reasons (missing_domain / unknown_customer / multiple_matches / duplicate), all + deterministic from the inputs — and for each verifies the OUTCOME: the run + completes on the expected path with the expected severity/engineering/responseMode, + preserves the correlationId, and a real Slack message was posted (escalation alert + on the escalation path, triage notice on the triage paths) — confirmed from the + executed Slack sendTask's own runtime response. Salesforce is not called (match + status is an input), so no Salesforce connection is required; the Slack node is a + real connector activity with an error boundary event so a Slack hiccup cannot + fault the run. + + Ported from Flow `e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml`; + the Script/Decision/Slack-node graph becomes a scriptTask/exclusiveGateway/Slack + Intsvc.ActivityExecution sendTask graph, `flow debug` becomes a + `bpmn debug`-per-case run against one ephemeral-solution import of the exact + submitted project (bpmn debug addresses runtime evidence by instance id via + `debug-instance variables-all`, not inline variables), and the criterion timeout + is raised to cover that extra CLI surface (see the grader's own arithmetic + comment) — the graded behaviors, weights, and thresholds are unchanged. +tags: [uipath-maestro-bpmn, e2e, mode:build, lifecycle:generate, shape:multi-node, node:decision, connector, feature:escalation, path-to-ga, outcome-graded] + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + - type: template_dir + path: _setup + mount_point: _setup + +run_limits: + expected_turns: 80 + max_turns: 200 + task_timeout: 5400 + turn_timeout: 1800 + +pre_run: + - command: "python3 _setup/seed.py" + timeout: 30 + +reference: + directory: ../.. + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 + +initial_prompt: | + Build a UiPath Maestro BPMN process named "CustomerEscalationOrchestrator" + (inside a solution of the same name) with a MANUAL start event. It triages a + customer escalation and its routing is driven entirely by the process inputs + so it can be exercised deterministically. + + Process inputs (declare as public `uipath:input` variables on the manual + start event): senderEmail, senderDomain, subject, body, customerTier, + productionDown (boolean), workaroundAvailable (boolean), hasAttachments + (boolean), customerMatchStatus ("single" | "none" | "multiple"), isDuplicate + (boolean), correlationId. + + Keep the process small. Put ALL routing logic in ONE scriptTask, then use a + single exclusive gateway plus two Slack connector activities — do not build + a large graph of gateways and per-reason handler nodes. + + 1. A single scriptTask "classify" reads the inputs and returns: + - escalationPath: + senderDomain empty -> "missing_domain" + else customerMatchStatus == "none" -> "unknown_customer" + else customerMatchStatus == "multiple" -> "multiple_matches" + else isDuplicate is true -> "duplicate" + else -> "escalation" + - severity: only for "escalation" classify Sev1/Sev2/Sev3 — + Sev1 = productionDown AND NOT workaroundAvailable; + Sev2 = productionDown AND workaroundAvailable; + Sev3 = NOT productionDown. + For every non-"escalation" (triage) path, severity = "informational". + - engineeringNeeded: true for Sev1/Sev2, otherwise false. + - responseMode: "Draft" when escalationPath == "escalation", else "None". + 2. An exclusive gateway routes on whether escalationPath == "escalation". + 3. escalation branch -> post a Slack "Send Message to channel" alert via the + Slack Integration Service connector activity. + other (triage) branch -> post a Slack "Send Message to channel" triage + notice the same way. + For both, use the Slack connection named "is-sandboxes", channel + "coding-agent-testing", send as `user`, and include escalationPath + + correlationId in the message. Wire each Slack activity's error boundary + event to a small handler so a Slack failure degrades gracefully. + 4. Use TWO end events — one the escalation branch reaches, one the triage + branch reaches. On EACH end event, map `slackMessageId` from THAT + branch's own Slack activity's output (escalation end event ← escalation + Slack activity; triage end event ← triage Slack activity), so + slackMessageId is never empty regardless of which branch runs. Map the + rest as public outputs on both end events: + - escalationPath (from the classify scriptTask) + - severity + - engineeringNeeded (boolean) + - responseMode + - caseKey = the incoming correlationId + - slackMessageId = the id/ts of the Slack message posted on this branch + + Salesforce/Jira/Drive are NOT part of this task — the match status is an + input. + + Validate the process. Do NOT run or debug the process — the grader executes it with + seeded inputs. Before starting, load the uipath-maestro-bpmn skill and follow its workflow. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: command_executed + description: "Agent validated the generated BPMN" + tool_name: "Bash" + # `bpmn` exists only under `maestro` -- `uip bpmn validate` is an unknown + # command, so the `maestro` segment must not be optional. + command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+maestro\s+bpmn\s+validate' + min_count: 1 + weight: 1.5 + pass_threshold: 1.0 + + # ── The delivered BPMN validates ──────────────────────────────────────── + # Discovered, not hardcoded, mirroring the flow suite's own note: `uip + # maestro bpmn init` scaffolds `Sol//.bpmn`, so a plain + # shell loop over every *.bpmn file is used instead of a fixed path. + - type: run_command + description: "Generated CustomerEscalationOrchestrator BPMN validates without errors, independent of solution layout" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "Seeded path cases: each drives its branch and produces the expected outcome; the escalation case actually posts a Slack alert (non-empty message id)" + command: "python3 $REFERENCE_DIR/_shared/check_escalation_orchestrator_paths.py" + # 7 x debug_budget(300, retries=1) = 2100 run_debug + 7x120s variables-all + # (840) + 90s solution init + 180s solution import + 60s margin = 3270; + # rounded up to 3300. (Flow's own per-case run_debug math is identical -- + # 2100 -- but bpmn debug needs the extra ephemeral solution + # init/import/variables-all plumbing flow debug does not.) + timeout: 3300 # budget-guard: manual x7 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/_setup/seed.py b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/_setup/seed.py new file mode 100644 index 0000000000..c9e10819f7 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/_setup/seed.py @@ -0,0 +1,54 @@ +#!/usr/bin/env python3 +"""Seed a Sev1 escalation case for the Slack-alert outcome eval. + +Writes seed.json with one Enterprise / production-down / no-workaround case. +A fresh correlationId per run keeps the posted Slack message and the caseKey +assertion isolated across runs. The grader (check_escalation_slack_alert.py) +runs `flow debug --inputs ` and asserts both the classification outputs +and that a Slack message was actually posted. +""" + +from __future__ import annotations + +import json +from pathlib import Path +from uuid import uuid4 + + +def build_seed() -> dict: + run_id = uuid4().hex[:12] + return { + "run_id": run_id, + "cases": [ + { + "name": "enterprise-production-down-sev1", + "inputs": { + "senderEmail": "jane.doe@acmecorp.com", + "subject": "Production down: checkout API returning 500s", + "body": ( + "Critical outage since 09:15 UTC. All checkout requests " + "are failing with HTTP 500 and orders are blocked." + ), + "customerTier": "Enterprise", + "productionDown": True, + "workaroundAvailable": False, + "correlationId": f"E2E-{run_id}-SEV1", + }, + "expected": { + "severity": "Sev1", + "engineeringNeeded": True, + "caseKey": f"E2E-{run_id}-SEV1", + }, + } + ], + } + + +def main() -> None: + path = Path("seed.json") + path.write_text(json.dumps(build_seed(), indent=2) + "\n", encoding="utf-8") + print(f"seeded {path}") + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml new file mode 100644 index 0000000000..37b26b1507 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml @@ -0,0 +1,125 @@ +task_id: skill-bpmn-e2e-escalation-slack-alert +description: > + End-to-end, outcome-based slice of the customer-escalation orchestration. + The agent builds a manual-start escalation-triage BPMN process that + classifies severity and posts a Slack alert. A manual start is used + deliberately — the Outlook email-received trigger cannot be reliably + debug-tested (seeding a self-addressed email is flaky against the shared + mailbox; see the outlook_trigger_inbox and customer_escalation task + notes). The grader seeds a Sev1 case, imports the exact submitted project + into an ephemeral solution, runs `uip maestro bpmn debug --inputs`, and + verifies the OUTCOME via `debug-instance variables-all`/`incidents`: the + process completes, classifies Sev1 with engineering needed, preserves the + correlationId, and — the point of the test — the Slack Send Message + activity actually posted (its own response ts is exposed as a process + output). Grades the delivered Slack side effect, not "did it run". + + Ported from Flow `e2e/escalation_slack_alert/escalation_slack_alert.yaml`; + the Script node is a bpmn:scriptTask (BPMN.Variables mapping) and the + Slack activity is a bpmn:sendTask carrying Intsvc.ActivityExecution; `flow + debug`'s single-call inline payload becomes an ephemeral-solution import + (sha256-pinned) + `bpmn debug` + `debug-instance variables-all`/ + `incidents`, because `bpmn debug` returns an instance id rather than + inline variables. No Jira, no tenant re-read, and no Slack teardown — same + as the ported Flow task, which never re-reads the tenant or deletes the + message it posts. +tags: [uipath-maestro-bpmn, e2e, mode:build, lifecycle:generate, shape:multi-node, connector, feature:escalation, path-to-ga, outcome-graded] + +run_limits: + expected_turns: 55 + max_turns: 120 + task_timeout: 2600 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + - type: template_dir + path: _setup + mount_point: _setup + +pre_run: + - command: "python3 _setup/seed.py" + timeout: 30 + +reference: + directory: ../.. + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 + +initial_prompt: | + Build a UiPath Maestro BPMN process named "EscalationSlackAlert" (inside a + solution of the same name) with a MANUAL start. + + The process triages a customer escalation and posts a Slack alert with next + steps. Declare these process inputs (in variables associated with the start + event): senderEmail, subject, body, customerTier, productionDown (boolean), + workaroundAvailable (boolean), correlationId. + + Steps: + 1. A Script task classifies severity from the inputs: + - Sev1 = customerTier is "Enterprise" AND productionDown is true AND + workaroundAvailable is false. + - Sev2 = productionDown is true AND workaroundAvailable is true. + - Sev3 = everything else (informational / no production impact). + Also decide engineeringNeeded (true for Sev1 and Sev2, false for Sev3) and a + short nextSteps string. + 2. Post a Slack alert to the channel "coding-agent-testing" using the Slack + "Send Message to channel" connector activity. Use the Slack connection + named "is-sandboxes" and send as `user`. The message must include the + severity, the correlationId, and the next steps. + 3. End the process, mapping these public output variables: + - severity (Sev1 / Sev2 / Sev3) + - engineeringNeeded (boolean) + - caseKey = the incoming correlationId + - slackMessageId = the identifier of the Slack message that was posted + (from the Slack Send node's own output) + + Validate the BPMN. Do NOT run or debug the process — the grader executes it + with seeded inputs. Before starting, load the uipath-maestro-bpmn skill and + follow its workflow. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + # ── The agent validated the BPMN (convention adherence) ───────────────── + - type: command_executed + description: "Agent validated the generated BPMN" + tool_name: "Bash" + command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+maestro\s+bpmn\s+validate' + min_count: 1 + weight: 1.5 + pass_threshold: 1.0 + + # ── The delivered BPMN validates ──────────────────────────────────────── + - type: run_command + description: "Generated EscalationSlackAlert BPMN validates without errors, independent of solution layout" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + # ── OUTCOME: seeded Sev1 run posts to Slack and returns the right outputs ─ + - type: run_command + description: "Seeded Sev1 debug run completes, classifies Sev1 + engineering, preserves correlationId, and the Slack alert was actually posted (non-empty message id)" + command: "python3 $REFERENCE_DIR/_shared/check_escalation_slack_alert.py" + timeout: 1100 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py new file mode 100644 index 0000000000..90dd603348 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py @@ -0,0 +1,113 @@ +#!/usr/bin/env python3 +"""Minimal live Jira helper for this task — wraps `uip is resources run`. + +Self-contained (no shared module). Assumes `uip` is on PATH and logged in and +that the connection + CE project exist — this is a tenant-gated e2e task. + +The connection is scoped to the curated single-record ops, so we create by +body / get by id / delete by id — never a JQL search. +""" + +from __future__ import annotations + +import json +import subprocess +from typing import Any + +CONNECTOR = "uipath-atlassian-jira" +FOLDER_PATH = "Shared/uipath-maestro-flow" +CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" +PROJECT_KEY = "CE" # "Coder Eval" project on uipath-sandbox-380 +ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project + + +def _operation(args: tuple[str, ...]) -> str: + return " ".join(args[:3]) + + +def _run(*args: str) -> dict[str, Any]: + result = subprocess.run( + ["uip", *args, "--output", "json"], + capture_output=True, text=True, timeout=120, + ) + try: + envelope = json.loads(result.stdout) + except json.JSONDecodeError as exc: + raise RuntimeError( + f"uip {_operation(args)} returned invalid JSON " + f"(exit {result.returncode}, stdout length {len(result.stdout)})" + ) from exc + if not isinstance(envelope, dict): + raise RuntimeError(f"uip {_operation(args)} returned a non-object JSON envelope") + return envelope + + +def _is_transient(envelope: dict[str, Any]) -> bool: + return envelope.get("Retry") == "RetryLater" or envelope.get("ErrorCode") == "server_error" + + +def _failure_summary(envelope: dict[str, Any]) -> str: + keys = ("Result", "ErrorCode", "Retry", "StatusCode", "Message") + summary = {key: envelope[key] for key in keys if key in envelope} + return json.dumps(summary, sort_keys=True, default=str) + + +def _required_data(*args: str, retry_transient: bool = False) -> Any: + envelope = _run(*args) + if retry_transient and _is_transient(envelope): + envelope = _run(*args) + if envelope.get("Result") != "Success" or "Data" not in envelope: + raise RuntimeError( + f"uip {_operation(args)} failed: {_failure_summary(envelope)}" + ) + return envelope["Data"] + + +def connection_id() -> str: + folder = _required_data("or", "folders", "get", FOLDER_PATH, retry_transient=True) + folder_key = folder["Key"] + conns = _required_data( + "is", "connections", "list", CONNECTOR, + "--folder-key", folder_key, "--refresh", + ) + return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) + + +def myself(conn_id: str) -> str: + """Return the connection user's Atlassian accountId.""" + return _required_data( + "is", "resources", "run", "get", CONNECTOR, "myself", + "--connection-id", conn_id, + )["accountId"] + + +def create_issue(conn_id: str, summary: str) -> str: + body = {"fields": {"project": {"key": PROJECT_KEY}, "issuetype": {"id": ISSUETYPE_ID}, "summary": summary}} + return _required_data( + "is", "resources", "run", "create", CONNECTOR, "curated_create_issue", + "--connection-id", conn_id, "--body", json.dumps(body), + )["key"] + + +def get_issue(conn_id: str, key: str) -> dict | None: + """Return the issue's `fields` dict, or None if it doesn't exist (404).""" + env = _run( + "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", + "--connection-id", conn_id, + "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", + ) + if env.get("Result") == "Failure": + return None + return env["Data"].get("fields", {}) + + +def delete_issue(conn_id: str, key: str) -> None: + """Delete an issue by key. A 404 (already gone) is a no-op.""" + _run( + "is", "resources", "run", "delete", CONNECTOR, "issue", + "--connection-id", conn_id, "--query", f"issueId={key}", + # The CLI never prompts and REFUSES an irreversible delete without this + # flag ("Confirmation required … Re-run with --yes"). Without it every + # teardown since 08-19 printed WARN and left its ticket in the CE project. + "--yes", + ) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/seed_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/seed_jira.py new file mode 100644 index 0000000000..1723e465c3 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/seed_jira.py @@ -0,0 +1,24 @@ +#!/usr/bin/env python3 +"""pre_run: write seed.json with the create targets (no live issue yet — the +agent's flow creates it when the check runs `flow debug`). The unique tag in +the summary lets the check locate this run's issue.""" + +import json +import secrets +from pathlib import Path + +import jira_is + +tag = secrets.token_hex(4) +summary = f"coder-eval jira flow e2e {tag}" +seed = { + "tag": tag, + "summary": summary, + "project_key": jira_is.PROJECT_KEY, + "issuetype_id": jira_is.ISSUETYPE_ID, + # Jira's design-time schema marks fields.reporter.id as required for this + # project/issue type; pin it so agents never have to guess or ask. + "reporter_id": jira_is.myself(jira_is.connection_id()), +} +Path("seed.json").write_text(json.dumps(seed, indent=2)) +print(f"OK: wrote seed targets (summary={summary!r}, project={jira_is.PROJECT_KEY})") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py new file mode 100644 index 0000000000..91055ad017 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py @@ -0,0 +1,22 @@ +#!/usr/bin/env python3 +"""post_run: delete every issue the run created (keys in `.created_keys`). +Idempotent and never fails the task.""" + +import sys +from pathlib import Path + +import jira_is + +try: + kf = Path(".created_keys") + keys = kf.read_text().split() if kf.is_file() else [] + if keys: + conn = jira_is.connection_id() + for key in keys: + jira_is.delete_issue(conn, key) + print(f"OK: deleted {key}") + else: + print("OK: nothing to delete") +except Exception as e: # noqa: BLE001 — teardown must not fail the task + print(f"WARN: teardown ignored error: {e}") +sys.exit(0) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml new file mode 100644 index 0000000000..6cd4cc983c --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml @@ -0,0 +1,104 @@ +task_id: skill-bpmn-jira-create-issue +description: > + E2E live Jira coverage — builds a Maestro BPMN process with a manual start + and an Atlassian Jira "Create Issue" connector node, then grades by + executing the process against a real Jira sandbox connection + (`bpmn debug`) and re-reading the tenant. The project/issue-type/summary + come from `seed.json` (unique per run), so the check verifies a real issue + was created with the seeded summary, not a fabricated output. The created + issue is cleaned up in post_run. + + Tenant prerequisite: a `uipath-atlassian-jira` connection in folder + `Shared/uipath-maestro-flow` (currently the single Jira connection there), + reaching the `CE` / "Coder Eval" project (issue type Task). Targets live in + the task's `jira_is.py`. + + Ported from Flow `e2e/jira_create_issue/jira_create_issue.yaml`; the + connector node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper (`uip is activities list + uipath-atlassian-jira`: curated objectName `curated_create_issue`, method + POST, or the generic object form `issue` + POST) instead of a Flow + connector node, `flow debug`'s single-call inline payload becomes an + ephemeral-solution import + `bpmn debug` + `debug-instance + variables-all`/`incidents` read, and the created-key search scans the root + scope's variables plus every element's Outputs (and the raw variables-all + response text) instead of one inline debug payload. +tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:single-node", connector, e2e, uipath-atlassian-jira, "mode:build"] + +run_limits: + expected_turns: 40 + task_timeout: 1800 + max_turns: 120 + turn_timeout: 900 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + - type: template_dir + path: _setup + mount_point: _setup + +pre_run: + - command: "python3 _setup/seed_jira.py" + timeout: 60 + +reference: + directory: ../.. + +post_run: + # Repo-standard solution sweep: the grader keeps its ephemeral solution + # under the sandbox CWD precisely so this glob finds the .uipx. + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 + # Connector record the solution sweep cannot reach (jira_is.py precedent). + - command: "python3 _setup/teardown_jira.py" + timeout: 120 + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "JiraCreateIssue" + with a manual start, inside a solution of the same name. + It should create a Jira issue using the Atlassian Jira "Create Issue" + connector activity, and expose the new issue's key as a process output. + + Read `seed.json` in the current directory for the issue details: use the + `project_key`, `issuetype_id`, `summary`, and `reporter_id` exactly as + given (pass the summary verbatim; use `reporter_id` for the issue's + reporter field). Do not stop to ask for missing values. + + Use the Atlassian Jira connection available in the `Shared/uipath-maestro-flow` + folder. If the Atlassian Jira connector isn't available in your registry, + stop rather than falling back to a generic HTTP request. + + Validate the final BPMN file. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN validates successfully" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: run_command + description: "JiraCreateIssue checks: valid BPMN with Jira Create-Issue node using the seeded fields, debug creates a real issue, and the tenant carries the seeded summary" + command: "python3 $REFERENCE_DIR/_shared/check_jira_create_issue.py" + timeout: 1080 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py new file mode 100644 index 0000000000..9e6edb53c0 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py @@ -0,0 +1,66 @@ +#!/usr/bin/env python3 +"""Minimal live Jira helper for this task — wraps `uip is resources run`. + +Self-contained (no shared module). Assumes `uip` is on PATH and logged in and +that the connection + CE project exist — this is a tenant-gated e2e task. + +The connection is scoped to the curated single-record ops, so we create by +body / get by id / delete by id — never a JQL search. +""" + +from __future__ import annotations + +import json +import subprocess + +CONNECTOR = "uipath-atlassian-jira" +FOLDER_PATH = "Shared/uipath-maestro-flow" +CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" +PROJECT_KEY = "CE" # "Coder Eval" project on uipath-sandbox-380 +ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project + + +def _run(*args: str) -> dict: + out = subprocess.run( + ["uip", *args, "--output", "json"], + capture_output=True, text=True, timeout=120, + ).stdout + return json.loads(out) + + +def connection_id() -> str: + folder_key = _run("or", "folders", "get", FOLDER_PATH)["Data"]["Key"] + conns = _run("is", "connections", "list", CONNECTOR, "--folder-key", folder_key, "--refresh")["Data"] + return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) + + +def create_issue(conn_id: str, summary: str) -> str: + body = {"fields": {"project": {"key": PROJECT_KEY}, "issuetype": {"id": ISSUETYPE_ID}, "summary": summary}} + return _run( + "is", "resources", "run", "create", CONNECTOR, "curated_create_issue", + "--connection-id", conn_id, "--body", json.dumps(body), + )["Data"]["key"] + + +def get_issue(conn_id: str, key: str) -> dict | None: + """Return the issue's `fields` dict, or None if it doesn't exist (404).""" + env = _run( + "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", + "--connection-id", conn_id, + "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", + ) + if env.get("Result") == "Failure": + return None + return env["Data"].get("fields", {}) + + +def delete_issue(conn_id: str, key: str) -> None: + """Delete an issue by key. A 404 (already gone) is a no-op.""" + _run( + "is", "resources", "run", "delete", CONNECTOR, "issue", + "--connection-id", conn_id, "--query", f"issueId={key}", + # The CLI never prompts and REFUSES an irreversible delete without this + # flag ("Confirmation required … Re-run with --yes"). Without it every + # teardown since 08-19 printed WARN and left its ticket in the CE project. + "--yes", + ) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/seed_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/seed_jira.py new file mode 100644 index 0000000000..8afc67747c --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/seed_jira.py @@ -0,0 +1,25 @@ +#!/usr/bin/env python3 +"""pre_run: create a real seed issue and write seed.json so the flow has a live +key to read back. The unique tag is embedded in the summary; the created key is +recorded to `.created_keys` so teardown can delete it.""" + +import json +import secrets +from pathlib import Path + +import jira_is + +tag = secrets.token_hex(4) +summary = f"coder-eval jira flow e2e {tag}" +conn = jira_is.connection_id() +key = jira_is.create_issue(conn, summary) +seed = { + "tag": tag, + "summary": summary, + "issue_key": key, + "project_key": jira_is.PROJECT_KEY, + "issuetype_id": jira_is.ISSUETYPE_ID, +} +Path("seed.json").write_text(json.dumps(seed, indent=2)) +Path(".created_keys").write_text(key + "\n") +print(f"OK: seeded live issue {key} (summary={summary!r})") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py new file mode 100644 index 0000000000..91055ad017 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py @@ -0,0 +1,22 @@ +#!/usr/bin/env python3 +"""post_run: delete every issue the run created (keys in `.created_keys`). +Idempotent and never fails the task.""" + +import sys +from pathlib import Path + +import jira_is + +try: + kf = Path(".created_keys") + keys = kf.read_text().split() if kf.is_file() else [] + if keys: + conn = jira_is.connection_id() + for key in keys: + jira_is.delete_issue(conn, key) + print(f"OK: deleted {key}") + else: + print("OK: nothing to delete") +except Exception as e: # noqa: BLE001 — teardown must not fail the task + print(f"WARN: teardown ignored error: {e}") +sys.exit(0) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml new file mode 100644 index 0000000000..190283e44a --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml @@ -0,0 +1,102 @@ +task_id: skill-bpmn-jira-get-issue +description: > + E2E live Jira coverage — builds a Maestro BPMN process with a manual start + and an Atlassian Jira "Get Issue" connector node that reads a pre-seeded + issue by key, then grades by executing the process against a real Jira + sandbox connection (`bpmn debug`) and asserting the fetched summary appears + in the process outputs. The issue is created by pre_run (its key + summary + are unique per run and land in `seed.json`), so the agent must read the key + from the fixture rather than inventing one, and cleanup happens in post_run. + + Tenant prerequisite: a `uipath-atlassian-jira` connection in folder + `Shared/uipath-maestro-flow` (currently the single Jira connection there), + reaching the `CE` / "Coder Eval" project (issue type Task). Targets live in + the task's `jira_is.py`. + + Ported from Flow `e2e/jira_get_issue/jira_get_issue.yaml`; the connector + node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper (`uip is activities list + uipath-atlassian-jira`: curated objectName `curated_get_issue`, method + GETBYID) instead of a Flow connector node, `flow debug`'s single-call + inline payload becomes an ephemeral-solution import + `bpmn debug` + + `debug-instance variables-all`/`incidents` read, and grading of the fetched + summary searches the root scope's variables plus every element's Outputs + instead of one inline debug payload. +tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:single-node", connector, e2e, uipath-atlassian-jira, "mode:build"] + +run_limits: + expected_turns: 40 + task_timeout: 1800 + max_turns: 120 + turn_timeout: 900 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + - type: template_dir + path: _setup + mount_point: _setup + +pre_run: + - command: "python3 _setup/seed_jira.py" + timeout: 90 + +reference: + directory: ../.. + +post_run: + # Repo-standard solution sweep: the grader keeps its ephemeral solution + # under the sandbox CWD precisely so this glob finds the .uipx. + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 + # Connector record the solution sweep cannot reach (jira_is.py precedent). + - command: "python3 _setup/teardown_jira.py" + timeout: 120 + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "JiraGetIssue" with + a manual start, inside a solution of the same name. + It should read a single Jira issue using the Atlassian Jira "Get Issue" + connector activity, and expose the retrieved issue's summary (and status) as + process outputs. + + Read `seed.json` in the current directory for the issue to fetch: use the + `issue_key`, `project_key`, and `issuetype_id` exactly as given. + + Use the Atlassian Jira connection available in the `Shared/uipath-maestro-flow` + folder. If the Atlassian Jira connector isn't available in your registry, + stop rather than falling back to a generic HTTP request. + + Validate the final BPMN file. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN validates successfully" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: run_command + description: "JiraGetIssue checks: valid BPMN with Jira Get-Issue node referencing the seeded key, debug completes, and outputs carry the seeded summary" + command: "python3 $REFERENCE_DIR/_shared/check_jira_get_issue.py" + timeout: 1080 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py new file mode 100644 index 0000000000..e5e2b601ee --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py @@ -0,0 +1,67 @@ +#!/usr/bin/env python3 +"""Minimal live Jira helper for this task — wraps `uip is resources run`. + +Self-contained (no shared module). Assumes `uip` is on PATH and logged in and +that the connection + CE project exist — this is a tenant-gated e2e task. + +The connection is scoped to the curated single-record ops, so we create by +body / get by id / delete by id — never a JQL search. +""" + +from __future__ import annotations + +import json +import subprocess + +CONNECTOR = "uipath-atlassian-jira" +FOLDER_PATH = "Shared/uipath-maestro-flow" +CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" +PROJECT_KEY = "CE" # "Coder Eval" project on uipath-sandbox-380 +ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project + + +def _run(*args: str) -> dict: + out = subprocess.run( + ["uip", *args, "--output", "json"], + capture_output=True, text=True, timeout=120, + ).stdout + return json.loads(out) + + +def connection_id() -> str: + folder_key = _run("or", "folders", "get", FOLDER_PATH)["Data"]["Key"] + conns = _run("is", "connections", "list", CONNECTOR, "--folder-key", folder_key, "--refresh")["Data"] + return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) + + +def create_issue(conn_id: str, summary: str) -> str: + body = {"fields": {"project": {"key": PROJECT_KEY}, "issuetype": {"id": ISSUETYPE_ID}, "summary": summary}} + return _run( + "is", "resources", "run", "create", CONNECTOR, "curated_create_issue", + "--connection-id", conn_id, "--body", json.dumps(body), + )["Data"]["key"] + + +def get_issue(conn_id: str, key: str) -> dict | None: + """Return the issue's `fields` dict (includes `summary`, `status`, and + `comment`), or None if it doesn't exist (404).""" + env = _run( + "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", + "--connection-id", conn_id, + "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", + ) + if env.get("Result") == "Failure": + return None + return env["Data"].get("fields", {}) + + +def delete_issue(conn_id: str, key: str) -> None: + """Delete an issue by key. A 404 (already gone) is a no-op.""" + _run( + "is", "resources", "run", "delete", CONNECTOR, "issue", + "--connection-id", conn_id, "--query", f"issueId={key}", + # The CLI never prompts and REFUSES an irreversible delete without this + # flag ("Confirmation required … Re-run with --yes"). Without it every + # teardown since 08-19 printed WARN and left its ticket in the CE project. + "--yes", + ) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py new file mode 100644 index 0000000000..2875ac11f3 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py @@ -0,0 +1,39 @@ +#!/usr/bin/env python3 +"""pre_run: write seed.json with a small batch of issues to create and the +per-branch comment markers. No live issue is created here — the agent's flow +creates one issue per list item when the check runs `flow debug`. + +The batch mixes `priority` values so the flow's Switch node must route each +item to a different Add-Comment branch: + + priority == "High" -> comment carries `escalated_marker` + otherwise -> comment carries `routine_marker` + +Every summary and both markers embed the unique per-run `tag`, so the check can +locate exactly this run's issues (no JQL search — the connection is curated-ops +only) and tell the two branches apart. +""" + +import json +import secrets +from pathlib import Path + +import jira_is + +tag = secrets.token_hex(4) +issues = [ + {"summary": f"coder-eval jira lifecycle {tag} item1", "priority": "High"}, + {"summary": f"coder-eval jira lifecycle {tag} item2", "priority": "Low"}, + {"summary": f"coder-eval jira lifecycle {tag} item3", "priority": "High"}, +] +seed = { + "tag": tag, + "project_key": jira_is.PROJECT_KEY, + "issuetype_id": jira_is.ISSUETYPE_ID, + "issues": issues, + "escalated_marker": f"ESCALATED {tag}", + "routine_marker": f"ROUTINE {tag}", +} +Path("seed.json").write_text(json.dumps(seed, indent=2)) +highs = sum(1 for i in issues if i["priority"] == "High") +print(f"OK: wrote {len(issues)} seed issues ({highs} High / {len(issues) - highs} other), tag={tag}") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py new file mode 100644 index 0000000000..91055ad017 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py @@ -0,0 +1,22 @@ +#!/usr/bin/env python3 +"""post_run: delete every issue the run created (keys in `.created_keys`). +Idempotent and never fails the task.""" + +import sys +from pathlib import Path + +import jira_is + +try: + kf = Path(".created_keys") + keys = kf.read_text().split() if kf.is_file() else [] + if keys: + conn = jira_is.connection_id() + for key in keys: + jira_is.delete_issue(conn, key) + print(f"OK: deleted {key}") + else: + print("OK: nothing to delete") +except Exception as e: # noqa: BLE001 — teardown must not fail the task + print(f"WARN: teardown ignored error: {e}") +sys.exit(0) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml new file mode 100644 index 0000000000..ff23e971f4 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml @@ -0,0 +1,131 @@ +task_id: skill-bpmn-jira-lifecycle +description: > + E2E live Jira coverage of a composite multi-instance-loop-and-gateway BPMN + process. The agent builds ONE UiPath Maestro BPMN process (manual start) + that iterates a seeded batch of issues and, per item, creates an Atlassian + Jira issue and then branches on the item's `priority` (an exclusive + gateway) to a branch-specific "Add Comment": `High` items get the seeded + `escalated_marker`, all others the `routine_marker`. Grading imports the + submitted project into an ephemeral solution and executes it against a real + Jira sandbox connection (`bpmn debug`), then re-reads the tenant: it asserts + the loop created every seeded issue and the gateway routed each to the + correct comment branch. The batch + markers are unique per run + (`seed.json`), so a fabricated or hardcoded output cannot pass. Every + created issue is cleaned up in post_run. + + Tenant prerequisite: a `uipath-atlassian-jira` connection in folder + `Shared/uipath-maestro-flow` (currently the single Jira connection there), + reaching the `CE` / "Coder Eval" project (issue type Task). Targets live in + the task's `jira_is.py`. + + Ported from Flow `e2e/jira_lifecycle/jira_lifecycle.yaml`; the flow's + `core.logic.loop` node is modeled as a `bpmn:multiInstanceLoopCharacteristics` + loop (over `=vars.Var_Issues`, item read as `iterator[0].item`), its + Switch/Decision node as a `bpmn:exclusiveGateway` with a `conditionExpression` + on `iterator[0].item.priority`, and each connector node as a + `bpmn:sendTask` carrying the registry `Intsvc.ActivityExecution` wrapper + (`uip is activities list uipath-atlassian-jira`: curated objectName + `curated_create_issue` for Create Issue, `curated_add_comment` for Add + Comment, both method POST) instead of a Flow connector node. `flow debug`'s + single-call inline payload becomes an ephemeral-solution import + `bpmn + debug` + `debug-instance variables-all`/`incidents` read, and the loop's + created-issue keys are recovered by scanning the `variables-all` payload + text for the seeded project's issue-key pattern instead of one inline debug + payload. +tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", "node:loop", "node:switch", "node:decision", connector, e2e, uipath-atlassian-jira, "mode:build"] + +run_limits: + expected_turns: 55 + task_timeout: 2700 + max_turns: 150 + turn_timeout: 900 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + - type: template_dir + path: _setup + mount_point: _setup + +pre_run: + - command: "python3 _setup/seed_jira.py" + timeout: 60 + +reference: + directory: ../.. + +post_run: + # Repo-standard solution sweep: the grader keeps its ephemeral solution + # under the sandbox CWD precisely so this glob finds the .uipx. + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 + # Connector record the solution sweep cannot reach (jira_is.py precedent). + - command: "python3 _setup/teardown_jira.py" + timeout: 180 + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "JiraLifecycle" + with a manual start, inside a solution of the same name. + + Read `seed.json` in the current directory. It contains a list of `issues` + (each with a `summary` and a `priority`), the `project_key` and `issuetype_id` + to create issues under, and two comment bodies: `escalated_marker` and + `routine_marker`. + + Build a process that loops over the `issues` list and, for EACH item: + 1. Creates a Jira issue with the item's `summary`, using the Atlassian Jira + "Create Issue" connector activity (with `project_key` / `issuetype_id`). + 2. Branches on the item's `priority`: when it is "High", add a comment whose + body is `escalated_marker`; otherwise add a comment whose body is + `routine_marker`. Use the Atlassian Jira "Add Comment" activity against + the issue created in step 1. + + Pass every summary and comment body verbatim from seed.json. + + Use the Atlassian Jira connection available in the `Shared/uipath-maestro-flow` + folder. If the Atlassian Jira connector isn't available in your registry, stop + rather than falling back to a generic HTTP request. + + Validate the final BPMN file. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN validates successfully" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: command_executed + description: "Advisory: live-v1 agent ran bpmn debug" + tool_name: "Bash" + # `bpmn` exists only under `maestro` -- `uip bpmn debug` is an unknown + # command, so the `maestro` segment must not be optional. + command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+maestro\s+bpmn\s+debug' + min_count: 1 + weight: 1.5 + pass_threshold: 0.0 + + - type: run_command + description: "JiraLifecycle checks: valid BPMN with a multi-instance loop + exclusive gateway over the Jira connector, debug creates every seeded issue, and each carries the comment its priority branch should have posted" + command: "python3 $REFERENCE_DIR/_shared/check_jira_lifecycle.py" + timeout: 1320 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py new file mode 100644 index 0000000000..d734b13884 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py @@ -0,0 +1,65 @@ +#!/usr/bin/env python3 +"""Minimal live Jira helper for this task — wraps `uip is resources run`. + +Self-contained (no shared module). Assumes `uip` is on PATH and logged in and +that the connection + CE project exist — this is a tenant-gated e2e task. +This helper uses the create / get / delete ops to seed and verify. +""" + +from __future__ import annotations + +import json +import subprocess + +CONNECTOR = "uipath-atlassian-jira" +FOLDER_PATH = "Shared/uipath-maestro-flow" +CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" +PROJECT_KEY = "CE" # "Coder Eval" project on uipath-sandbox-380 +ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project + + +def _run(*args: str) -> dict: + out = subprocess.run( + ["uip", *args, "--output", "json"], + capture_output=True, text=True, timeout=120, + ).stdout + return json.loads(out) + + +def connection_id() -> str: + folder_key = _run("or", "folders", "get", FOLDER_PATH)["Data"]["Key"] + conns = _run("is", "connections", "list", CONNECTOR, "--folder-key", folder_key, "--refresh")["Data"] + return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) + + +def create_issue(conn_id: str, summary: str) -> str: + body = {"fields": {"project": {"key": PROJECT_KEY}, "issuetype": {"id": ISSUETYPE_ID}, "summary": summary}} + return _run( + "is", "resources", "run", "create", CONNECTOR, "curated_create_issue", + "--connection-id", conn_id, "--body", json.dumps(body), + )["Data"]["key"] + + +def get_issue(conn_id: str, key: str) -> dict | None: + """Return the issue's `fields` dict (includes `summary` and `comment`), + or None if it doesn't exist (404).""" + env = _run( + "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", + "--connection-id", conn_id, + "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", + ) + if env.get("Result") == "Failure": + return None + return env["Data"].get("fields", {}) + + +def delete_issue(conn_id: str, key: str) -> None: + """Delete an issue by key. A 404 (already gone) is a no-op.""" + _run( + "is", "resources", "run", "delete", CONNECTOR, "issue", + "--connection-id", conn_id, "--query", f"issueId={key}", + # The CLI never prompts and REFUSES an irreversible delete without this + # flag ("Confirmation required … Re-run with --yes"). Without it every + # teardown since 08-19 printed WARN and left its ticket in the CE project. + "--yes", + ) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py new file mode 100644 index 0000000000..3e3cc46838 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py @@ -0,0 +1,26 @@ +#!/usr/bin/env python3 +"""pre_run: create a couple of real issues carrying a unique tag in their +summary, and write seed.json with the JQL that selects exactly them plus the +comment the flow should stamp on each match. The created keys are recorded to +`.created_keys` so teardown deletes them regardless of the run outcome. +""" + +import json +import secrets +from pathlib import Path + +import jira_is + +tag = secrets.token_hex(4) +conn = jira_is.connection_id() +keys = [jira_is.create_issue(conn, f"coder-eval jira search-triage {tag} #{n}") for n in (1, 2)] +seed = { + "tag": tag, + "project_key": jira_is.PROJECT_KEY, + "jql": f'project = {jira_is.PROJECT_KEY} AND summary ~ "{tag}"', + "processed_comment": f"TRIAGED {tag}", + "issue_keys": keys, +} +Path("seed.json").write_text(json.dumps(seed, indent=2)) +Path(".created_keys").write_text("\n".join(keys) + "\n") +print(f"OK: seeded {len(keys)} issues {keys} (tag={tag}) for JQL search") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py new file mode 100644 index 0000000000..91055ad017 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py @@ -0,0 +1,22 @@ +#!/usr/bin/env python3 +"""post_run: delete every issue the run created (keys in `.created_keys`). +Idempotent and never fails the task.""" + +import sys +from pathlib import Path + +import jira_is + +try: + kf = Path(".created_keys") + keys = kf.read_text().split() if kf.is_file() else [] + if keys: + conn = jira_is.connection_id() + for key in keys: + jira_is.delete_issue(conn, key) + print(f"OK: deleted {key}") + else: + print("OK: nothing to delete") +except Exception as e: # noqa: BLE001 — teardown must not fail the task + print(f"WARN: teardown ignored error: {e}") +sys.exit(0) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml new file mode 100644 index 0000000000..d4082cee26 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml @@ -0,0 +1,107 @@ +task_id: skill-bpmn-jira-search-triage +description: > + E2E live Jira coverage of a JQL-search-driven triage process: a + manual-start Maestro BPMN process that searches for issues matching a + seeded JQL and, for each match, adds a triage comment. pre_run seeds two + real issues carrying a unique tag; grading runs the process (`bpmn debug`) + and asserts both seeded issues come back carrying the triage comment. + + Tenant prerequisite: a `uipath-atlassian-jira` connection in folder + `Shared/uipath-maestro-flow` reaching the `CE` / "Coder Eval" project, with + issue-search scope. Targets live in `jira_is.py`. + + Ported from Flow `e2e/jira_search_triage/jira_search_triage.yaml`; the + search-and-loop-and-comment shape is modeled as a bpmn:sendTask carrying + the registry Intsvc.ActivityExecution wrapper for the Search Issues by JQL + operation (`uip is activities list uipath-atlassian-jira`: curated + objectName `issue_search_get`, method GET) feeding a + bpmn:multiInstanceLoopCharacteristics loop over the matches, whose body is + an Add Comment connector activity (curated objectName + `curated_add_comment`), instead of a Flow search node feeding a + `core.logic.loop` node. `flow debug`'s single inline payload becomes an + ephemeral-solution import + `bpmn debug` + `debug-instance incidents` read. + The second criterion's timeout is raised from Flow's 1320s to 1440s: the + BPMN sequence adds the ephemeral solution init/import and an explicit + incidents read on top of Flow's single inline debug call, on the same + tenant re-read Flow itself performs (arithmetic in the grader's docstring). +tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", "node:loop", connector, e2e, uipath-atlassian-jira, "mode:build"] + +run_limits: + expected_turns: 45 + task_timeout: 2400 + max_turns: 120 + turn_timeout: 900 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + - type: template_dir + path: _setup + mount_point: _setup + +pre_run: + - command: "python3 _setup/seed_jira.py" + timeout: 90 + +reference: + directory: ../.. + +post_run: + # Repo-standard solution sweep: the grader keeps its ephemeral solution + # under the sandbox CWD precisely so this glob finds the .uipx. + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 + # Connector record the solution sweep cannot reach (jira_is.py precedent). + - command: "python3 _setup/teardown_jira.py" + timeout: 120 + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "JiraSearchTriage" + with a manual start, inside a solution of the same name. + + Read `seed.json` in the current directory for the `jql` query and the + `processed_comment` body. + + Build a process that: + 1. Searches Jira for issues matching `jql`, using the Atlassian Jira + "Search Issues by JQL" connector activity. + 2. Loops over the returned issues and adds a comment to each whose body is + `processed_comment` (pass it verbatim), using the "Add Comment" activity. + + Use the Atlassian Jira connection available in the `Shared/uipath-maestro-flow` + folder. If the Atlassian Jira connector isn't available in your registry, stop + rather than falling back to a generic HTTP request. + + Validate the final BPMN file. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN validates successfully" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: run_command + description: "JiraSearchTriage checks: valid BPMN with a JQL search feeding a loop of Add-Comment, debug completes, and both seeded issues carry the triage comment" + command: "python3 $REFERENCE_DIR/_shared/check_jira_search_triage.py" + timeout: 1440 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml new file mode 100644 index 0000000000..cc225770ad --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml @@ -0,0 +1,91 @@ +task_id: skill-bpmn-bellevue-weather +description: > + Create a UiPath Maestro BPMN process that fetches today's weather in + Bellevue from open-meteo, formats a summary, and branches on temperature: + if > 60F output 'nice day', otherwise 'bring a jacket'. Exercises a managed + HTTP activity and a decision gateway, then grades by executing the process + (`bpmn debug`) and asserting one of the two verdict strings appears in the + process outputs. + + Ported from Flow `multi_node/bellevue_weather/bellevue_weather.yaml`; the + Managed HTTP Request node (core.action.http.v2) becomes a bpmn:sendTask + carrying the registry Intsvc.HttpExecution wrapper in manual + (connectionless) mode -- BPMN has no curated Open-Meteo connector, so the + Flow grader's connector-fallback option has nothing to translate to and is + dropped -- the Decision node becomes a bpmn:exclusiveGateway, Flow's single + inline `flow debug` payload read becomes an ephemeral-solution import + + `bpmn debug` + `debug-instance variables-all`/`incidents` read, and the + live-check criterion timeout grows from 600s to 1050s to fund that larger + live-debug surface (LIVE-ADDENDUM budget rule); the graded assertions + themselves (a weather-HTTP node exists; the run's outputs contain one of + the two verdict strings) are unchanged from Flow's own grader. +tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", "node:decision", "feature:http"] + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +run_limits: + expected_turns: 32 + turn_timeout: 1200 + # task_timeout bounds turns AND grading under one watchdog (tests/README.md), + # and grading only gets what the turn did not spend. Flow's own bellevue_weather + # relies on the 1200s experiment default, which BPMN's larger live-debug + # surface (see check_weather_bpmn.py's budget comment) would not leave room + # for: 1200 (unchanged turn cap) + 180 (validate) + 1050 (weather live check) + # = 2430; rounded up for per-criterion spawn overhead. + task_timeout: 2500 + +reference: + directory: ../.. + +initial_prompt: | + Build the process inside a solution of the same name. + + Create a UiPath Maestro BPMN process project named "BellevueWeather" that + gets today's weather in Bellevue from open-meteo, formats a summary, and if + the temperature is greater than 60F returns a summary with a message field + 'nice day', otherwise the message field should be 'bring a jacket'. + Use a Managed HTTP Request activity (Intsvc.HttpExecution) in manual + (connectionless) mode to call the open-meteo API directly — do not use an + Integration Service connector. + + Validate the process. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + # ── BPMN file validity ───────────────────────────────────────────────── + - type: run_command + description: "uip maestro bpmn validate passes on the bpmn file" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + # ── Execution checks ────────────────────────────────────────────────── + - type: run_command + description: "BPMN debug runs and output contains 'nice day' or 'bring a jacket'" + command: "python3 $REFERENCE_DIR/_shared/check_weather_bpmn.py" + timeout: 1050 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml new file mode 100644 index 0000000000..b3f54670f1 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml @@ -0,0 +1,127 @@ +task_id: skill-bpmn-devcon-billing-discrepancy-detector +description: | + E2E greenfield (DevCon BillingDisputeResolution scenario): build a Maestro + BPMN process that queries the BillingDisputeERP and BillingDisputeCRM + entities as parallel branches joined by a parallel-gateway merge, then + computes an invoice overcharge; graded by validate plus one bpmn debug run + against seeded tenant data. + + Ported from Flow `multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml`; + the two Data Service reads are modeled as bpmn:sendTask nodes carrying the + registry Intsvc.ActivityExecution wrapper (connectorKey + uipath-uipath-dataservice, Query Entity Records) instead of Flow connector + nodes, the fan-out/merge as a bpmn:parallelGateway fork and join instead of + `core.logic.merge`, and `flow debug`'s single-call inline payload becomes an + ephemeral-solution import + `bpmn debug` + `debug-instance + variables-all`/`incidents` read. The live grading criterion's timeout is + raised from Flow's 600s to 1050s to cover the extra CLI steps (`solution + init`/`solution projects import`, separate `variables-all`/`incidents` + reads) that BPMN's live surface needs beyond Flow's single `flow debug` + call, matching multi_node/slack_weather_pipeline's precedent for the same + deviation. +tags: + - uipath-maestro-bpmn + - e2e + - mode:build + - lifecycle:execute + - shape:multi-node + - connector + - path-to-ga + +run_limits: + max_turns: 120 + turn_timeout: 2400 + task_timeout: 3840 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +reference: + directory: ../.. + +post_run: + - command: python3 _setup/cleanup_solutions.py + timeout: 120 + +initial_prompt: | + Build a UiPath Maestro BPMN process "BillingDiscrepancyDetector" (in a + solution of the same name) with a manual start, that checks a disputed + invoice line against the contracted amount and pulls the customer's account + tier. + + - Inputs: process variables `invoiceNumber` (string), `accountNumber` + (string), `disputedLineNumber` (number), `disputedUnitPrice` (number), + `disputedQuantity` (number). + - Look up two existing Data Service entities: + - `BillingDisputeERP` — rows for `invoiceNumber`; each has a `lineNumber` + and contracted `amount`. + - `BillingDisputeCRM` — the account row for `accountNumber`; it has an + `accountTier`. + - The two lookups are independent. Run them as parallel branches that fan out + from the start event and rejoin at a parallel-gateway join before computing + the result — do not chain them one after the other. + - Compute: invoiced = `disputedUnitPrice` x `disputedQuantity`; contracted = + the `amount` of the ERP row whose `lineNumber` equals `disputedLineNumber`; + `totalOvercharge` = invoiced minus contracted when positive, else 0; + `discrepancyCount` = 1 when overcharge is positive, else 0. + - Outputs: `totalOvercharge` (number), `discrepancyCount` (number), + `matchedInvoiceNumber` (string, first ERP row), `accountTier` (string, from + CRM). + + Build both lookups on the UiPath Data Service Integration Service connector + (`uipath-uipath-dataservice`) Query Entity Records activity, not a native + Data Fabric node: the tenants this ships to do not all enable one. + + Validate the final BPMN file — the task is not complete until `uip maestro + bpmn validate` passes. Grading then runs the process under `bpmn debug` + against live tenant data, so build it to actually run, not merely validate. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +pre_run: + - command: 'python3 "_setup/preflight_connections.py" uipath-uipath-dataservice' + timeout: 120 + +success_criteria: + - type: run_command + description: uip maestro bpmn validate passes on the BPMN file + command: python3 $REFERENCE_DIR/_shared/validate_bpmn.py + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + - type: run_command + description: Check requires a parallel-gateway join (parallel branches) and debugs once; process computed overcharge=1610, count=1, invoice MCS-2026-04872, tier Enterprise + command: python3 $REFERENCE_DIR/_shared/check_billing_discrepancy_detector.py detector + timeout: 1050 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 + - type: run_command + description: 'Advisory: packed bindings_v2.json Connection resources exist and carry no stub UUIDs' + command: python3 $REFERENCE_DIR/_shared/check_billing_discrepancy_detector.py bindings + timeout: 120 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0.0 + - type: run_command + description: 'Advisory: STRUCTURAL (covering): two DS queries (ERP + CRM) on MUTUALLY UNREACHABLE branches from the start event, converging on one bpmn:parallelGateway join that continues downstream; each filter computed from its own input; no answer literals; each output read from its own side' + command: python3 $REFERENCE_DIR/_shared/check_billing_discrepancy_detector.py advisory + timeout: 30 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0.0 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml new file mode 100644 index 0000000000..55f39c79db --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml @@ -0,0 +1,122 @@ +task_id: skill-bpmn-devcon-billing-invoice-lookup +description: | + E2E greenfield (DevCon BillingDisputeResolution scenario): build a ~4-node + Maestro BPMN process that normalizes a messy invoice number and queries the + BillingDisputeERP Data Service entity; graded by validate plus three + ephemeral-solution `bpmn debug` runs over malformed inputs. + + Ported from Flow `multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml`. + Changes: the Data Service `query-entity-records` Flow node becomes a + `bpmn:sendTask` carrying the registry `Intsvc.ActivityExecution` wrapper + (curated `QueryEntityRecordsCurated`/`QueryEntityRecords_V3`, or the generic + entity object with operation List/method GET); the connection is a + process-level `=bindings.` `resource="Connection"` binding instead of an + inline `connectionId`/`connectionFolderKey` pair; `flow debug`'s single + inline-payload call becomes, per malformed input, an ephemeral solution + import + `bpmn debug` + `debug-instance variables-all`/`incidents` read + (LIVE-ADDENDUM's canonical live pattern); and the bindings/non-stub-id check + now packs the project first (`uip maestro bpmn pack`), since BPMN + materializes `bindings_v2.json` only at pack time, unlike Flow's compiler, + which resolved the connection straight into the emitted node. The live + lookup criterion's timeout is raised from Flow's 1200s to 1590s to cover the + extra ephemeral-solution CLI round trips per case; see + `_shared/check_billing_invoice_lookup.py`'s module docstring for the + arithmetic. +tags: + - uipath-maestro-bpmn + - e2e + - "mode:build" + - "lifecycle:execute" + - "shape:multi-node" + - connector + - path-to-ga +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup +run_limits: + max_turns: 120 + turn_timeout: 2400 + task_timeout: 4455 +reference: + directory: ../.. + +pre_run: + - command: 'python3 "_setup/preflight_connections.py" uipath-uipath-dataservice' + timeout: 120 + +post_run: + - command: python3 _setup/cleanup_solutions.py + timeout: 120 + +initial_prompt: | + Build a UiPath Maestro BPMN process "BillingInvoiceLookup" (in a solution of + the same name) that looks up a disputed invoice's line items, tolerating + messy input. + + - Input: process variable `invoiceNumber` (string). Callers may omit the + "MCS-" prefix, use wrong casing, or add whitespace. + - Normalize the input (trim, uppercase, prepend `MCS-` if absent), then look + up the matching rows in the existing `BillingDisputeERP` Data Service + entity. + - Outputs: `matchedInvoiceNumber` (string, first matched row's invoiceNumber) + and `lineItemCount` (number of rows returned). + + Build the lookup on the Data Service Query Entity Records connector + activity (`uipath-uipath-dataservice`). + + Validate the final BPMN file — the task is not complete until `uip maestro + bpmn validate` passes. Grading then runs the process under debug against + live tenant data, so build it to actually run, not merely validate. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. + +success_criteria: + - type: run_command + description: uip maestro bpmn validate passes on the BPMN file + command: python3 $REFERENCE_DIR/_shared/validate_bpmn.py + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + - type: run_command + description: Check script debugs three malformed inputs (fail-fast, one ephemeral-solution bpmn debug run each); every exercised input must resolve to MCS-2026-04872 with 8 line items + command: python3 $REFERENCE_DIR/_shared/check_billing_invoice_lookup.py lookup + timeout: 1590 # 90 (solution init) + 180 (solution import) + 3 x (180 debug + 120 variables-all + 120 incidents) + 60 margin; see check_billing_invoice_lookup.py + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 + - type: run_command + description: 'Advisory: Data Service query filters server-side (whole-entity fetch + client-side filter breaks silently past the page limit)' + command: python3 $REFERENCE_DIR/_shared/check_billing_invoice_lookup.py server_side_filter + timeout: 15 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0.0 + - type: run_command + description: 'Advisory: generated bindings exist and carry no stub UUIDs (packs the project to read bindings_v2.json)' + command: python3 $REFERENCE_DIR/_shared/check_billing_invoice_lookup.py bindings + timeout: 180 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0.0 + - type: run_command + description: 'Advisory: STRUCTURAL (covering): one Data Service query node, entityName anywhere in its inputs, the filter COMPUTED from the input, no canonical/test-input literals, both outputs read from the query, and a wired connection binding' + command: python3 $REFERENCE_DIR/_shared/check_billing_invoice_lookup.py advisory + timeout: 30 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0.0 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml new file mode 100644 index 0000000000..dd5cc7c916 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml @@ -0,0 +1,87 @@ +task_id: skill-bpmn-slack-channel-description +description: > + Create a UiPath Maestro BPMN process that uses the Slack IS connector to + retrieve the channel description of #office-bellevue and outputs it. This is + an end-to-end test that exercises connector discovery, connection binding, + reference resolution, node configuration, and cloud debug execution. + + Ported from Flow `multi_node/slack_channel_description/slack_channel_description.yaml`; + the connector node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper (connectorKey uipath-salesforce-slack; + `uip is activities list uipath-salesforce-slack`: curated GetConversationInfo, + objectName ConversationsInfo_GET, method GETBYID) instead of a Flow connector + node, `flow debug`'s single-call inline payload becomes an ephemeral-solution + import + `bpmn debug` + `debug-instance variables-all`/`incidents` read, and + grading of the retrieved description searches the root scope's variables + plus every element's Outputs instead of one inline debug payload. The live + grading criterion's timeout is raised from Flow's 600s to 1050s to cover the + extra CLI steps (`solution init`/`solution projects import`, separate + `variables-all`/`incidents` reads) that BPMN's live surface needs beyond + Flow's single `flow debug` call; `run_limits` gains `task_timeout` and + `max_turns` (absent from the Flow task) so that grading window fits inside + the task watchdog. +tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", connector] + +run_limits: + expected_turns: 45 + max_turns: 120 + task_timeout: 1800 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +reference: + directory: ../.. + +pre_run: + - command: 'python3 "_setup/preflight_connections.py" uipath-salesforce-slack=uipath-maestro-flow' + timeout: 120 + +initial_prompt: | + Build the process inside a solution of the same name. + + Create a UiPath Maestro BPMN process named "SlackChannelDescription" that + retrieves the channel description of #office-bellevue and outputs it. + + Validate the final BPMN file. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. + +success_criteria: + # ── BPMN file validity ───────────────────────────────────────────────── + - type: run_command + description: "uip maestro bpmn validate passes on the BPMN file" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + # ── End-to-end execution ──────────────────────────────────────────────── + - type: run_command + description: "BPMN debug runs successfully and output contains the Bellevue office address" + command: "python3 $REFERENCE_DIR/_shared/check_channel_description.py" + timeout: 1050 + expected_exit_code: 0 + weight: 6.0 + pass_threshold: 1.0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml new file mode 100644 index 0000000000..7774572b4f --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml @@ -0,0 +1,88 @@ +task_id: skill-bpmn-slack-weather-pipeline +description: > + Multi-step pipeline: read Slack channel description, extract city with a + script, fetch weather for that city via HTTP, decide warm/cold. Exercises + chaining Slack connector → Script → HTTP → Decision → End with data flowing + between nodes. + + Ported from Flow `multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml`; + the Slack read is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper (connectorKey uipath-salesforce-slack) + instead of a Flow connector node, the weather fetch as a bpmn:sendTask + carrying Intsvc.HttpExecution in manual (connectionless) mode targeting + open-meteo, `flow debug`'s single-call inline payload becomes an + ephemeral-solution import + `bpmn debug` + `debug-instance + variables-all`/`incidents` read, and grading of the mapped `weatherVerdict` + output searches the root scope's variables plus every element's Outputs + instead of one inline debug payload. The live grading criterion's timeout + is raised from Flow's 600s to 1050s to cover the extra CLI steps (`solution + init`/`solution projects import`, separate `variables-all`/`incidents` + reads) that BPMN's live surface needs beyond Flow's single `flow debug` + call. +tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", "node:decision", connector, "feature:http"] + +run_limits: + expected_turns: 51 + task_timeout: 2490 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +reference: + directory: ../.. + + +initial_prompt: | + Build the process inside a solution of the same name. + + Create a UiPath Maestro BPMN process called "SlackWeatherPipeline". Read the + #office-bellevue Slack channel description — it holds the office's full US + mailing address (street, suite/unit, city, state, ZIP) — identify the city + (the locality, not the street or suite line). Then fetch the current weather + for that city from open-meteo. If it's above 60F the process must output the + text in variable `weatherVerdict` 'warm office today', otherwise 'cold + office today'. + + All connections for this task live in Orchestrator folder + `Shared/uipath-maestro-flow`. + + Validate the final BPMN file. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. + +success_criteria: + - type: run_command + description: "uip maestro bpmn validate passes" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN debug runs end-to-end and the named out variable weatherVerdict holds exactly 'warm office today' or 'cold office today': Slack + HTTP + decision all execute" + command: "python3 $REFERENCE_DIR/_shared/check_slack_weather_pipeline.py" + timeout: 1050 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 From a5446913a699b91c545d02cfdc0d8f2129a5a6a5 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:23:30 -0700 Subject: [PATCH 02/35] docs(bpmn): live-tier porting handoff and methodology Adds the porting brief, connector and live addenda, normalization contract, parity ledger and a handoff document describing what is done, what remains and how each port is made, so the live-tier work can resume on this branch. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/LIVE-ADDENDUM.md | 46 ++++ .../_porting/LIVE-HANDOFF.md | 75 +++++++ .../_porting/parity-ledger.md | 212 ++++++++++++++++++ 3 files changed, 333 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md create mode 100644 tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md create mode 100644 tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md new file mode 100644 index 0000000000..fc540e7cc2 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md @@ -0,0 +1,46 @@ +# Live-tier addendum — ports whose Flow grader runs `flow debug` + +Read `PORTING-BRIEF.md` (Grading contract is mandatory) and `BATCH1-ADDENDUM.md` first. This file adds the rules for ports whose Flow criteria execute the artifact. Everything here is a **T** translation of Flow's `flow_check.run_debug` + output assertions; it is not licence to add assertions Flow does not make. + +## The canonical live pattern (copy it) + +`tests/tasks/uipath-maestro-bpmn/e2e/customer_escalation_triage/check_customer_escalation_behavior.py` + `escalation_is.py` on this branch is the one CI-proven live BPMN grader. Its sequence, all via `_shared/bpmn_live.py`: + +1. `uip solution init /` under the sandbox CWD (ephemeral solution; kept under CWD so the standard post_run sweep finds the `.uipx`). +2. `uip solution projects import --solutionFile `; assert the imported `.bpmn` bytes equal the submitted ones (`sha256`). +3. `debug_data, instance_id = run_debug(imported_project_dir, inputs, log_file, timeout=…)` — `bpmn debug` returns an instance id, not inline variables. `--inputs` JSON is honoured (the escalation run seeded correlationId this way and found it in Jira). +4. `uip maestro bpmn debug-instance variables-all ` → `root_scope(variables_data)` for root variables; `element_output_records(variables_data, element_id)` for a node's `Outputs`; `connector_response_values(outputs, name)` for connector response fields. +5. `uip maestro bpmn debug-instance incidents ` → `incident_records(...)`; a completed run with incidents is a failure. +6. Side-effect ids go to a flat journal the moment they are visible; post_run replays it (teardown) — mirror the Flow task's `_setup/teardown_*.py`. + +Known runtime facts (grade around them, do not fight them): +- Element-level `Outputs` (a script task's mapped output, a connector's `response`) are reliably readable in `variables-all`. Root **public output values** have been read back as `null` even when correctly mapped (see `debug/live_debug_e2e/check_live_debug.py` docstring). So when Flow asserted "some output equals X" (`assert_output_value` / `assert_outputs_contain` over `variables.globals` + element outputs), translate to: search the value leaves of the root scope's variables AND every element's `Outputs` in `variables-all`; do not require the value on a root public output specifically. +- Variables are addressed by **id**, and the runtime may re-case ids — use `resolve_runtime_key`. +- Debug instances are ephemeral; read variables-all immediately after the run. + +## Budgets (the test suite enforces this) + +`_shared/test_criterion_budgets.py` statically prices every `run_debug(...)` call in a grader and requires the calling criterion's `timeout:` ≥ `debug_budget(timeout, retries, backoff) + bpmn_live.CRITERION_MARGIN_SECONDS` (plus the other CLI steps you run — the escalation grader sums `STEP_TIMEOUTS` and its unit test pins the YAML timeout to that sum). Keep Flow's criterion `timeout:` if it fits; if the BPMN sequence (solution init + import + debug + variables-all + incidents) needs more, raise ONLY that criterion's `timeout` and say so in the description — this is the one sanctioned deviation from "criteria identical", because the budget is a property of the CLI surface, not of what is graded. `run_limits.task_timeout` must cover the agent's turn budget plus all grading, as the escalation YAML documents. + +## Fixtures, seeds, teardown, cleanup + +- Copy the Flow task's `_setup/` scripts (seed, teardown, `jira_is.py`-style helpers) into the BPMN task's own `_setup/` and mount them the same way; change only what is suite-specific (none of the `uip is` plumbing is). Keep tenant targets (connection names, folder `Shared/uipath-maestro-flow`, project keys, channel ids) verbatim — they are shared tenant fixtures, not Flow vocabulary. +- post_run: `python3 _setup/cleanup_solutions.py` (mounted from `tests/tasks/uipath-maestro-bpmn/_setup/`) first, then the task's teardown, with Flow's timeouts or larger if the BPMN sweep has more to delete. +- Prompts keep Flow's wording about running ("Validate the flow" stays validate; if Flow's prompt told the agent to debug/run, keep that instruction with the `bpmn debug` verb). Live ports do NOT get the structural "Do NOT upload/debug/run" closing line — the grader itself runs the process, and the Flow prompt's own scope stands. + +## Assertion map tags for live ports + +Add these T rows as needed and cite the Flow helper you translate: +- `T flow_check.run_debug(inputs=…)` → ephemeral solution + `bpmn_live.run_debug(project, inputs, log)` +- `T assert_output_value / assert_outputs_contain / assert_named_equals` → value-leaf search over `variables-all` root scope + element `Outputs` (declared inputs excluded, as Flow excluded them) +- `T assert_node_type_executed / completed_node_ids_of_type` → the element's record in `variables-all` (or `debug_data.ElementExecutions[].Status == Completed`) +- `T assert_slack_message_posted / assert_connector_send_identity` → `connector_response_values` on the connector element's outputs, then the same tenant re-read Flow did +- `T assert_connector_error_handlers` → boundary error event / error path presence — only if Flow asserted it + +If a Flow assertion has no readable runtime evidence on the BPMN side (e.g. a root-output-only value that the runtime returns as null and no element output carries it), STOP and report "parked: " rather than loosening the assertion. + +## Learned on CI run 35503094182 + +- `uip maestro bpmn debug` polls at most 300 times; the wait is `300 × --poll-interval`. `bpmn_live.run_debug` now derives the interval from its `timeout` so the CLI keeps polling for the whole priced budget, and raises a clear `CheckFailure` on the CLI's poll-timeout envelope (`ErrorCode: timeout`, `Data.lastStatus`). Never pass a fixed small poll interval. +- Runtime incident 102010 with `ErrorDetails: "Value cannot be null. (Parameter 'Folder')"` on a Slack `Intsvc.ActivityExecution` means the activity lacks its `folderKey` binding — an authoring defect of the eval agent, not a grader defect. Do not widen a grader for it. +- Actions.HITL cannot pass `bpmn validate` without a deployed Action App binding (MISSING_BINDING with placeholder appId). HITL ports reach parity on every other criterion; record the validate gap in the description, do not work around it. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md new file mode 100644 index 0000000000..95a2e478ff --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -0,0 +1,75 @@ +# Flow → BPMN eval porting: live-tier handoff + +Branch: `test/bpmn-port-live`, stacked on `test/bpmn-port-connectors` (PR #3426, the structural bucket). This document is the state of the live-tier port work as of 2026-09-20 and how to continue it. Methodology files sit beside it in this directory. + +## Context + +The Flow suite (`tests/tasks/uipath-maestro-flow/`) has 131 tasks; the BPMN suite had 82. A per-task map put 62 Flow tasks in scope (`parity-ledger.md`): 29 structural (authoring + `validate`), 17 live (the Flow grader runs `flow debug`), 16 feasibility probes; 30 are Flow-only surface (IXP, evaluate, voice, conversational, bindings, native Data Fabric nodes). + +Reading the Flow graders during the loop reclassified four "structural" tasks as live (their graders debug): `slack_http_fallback`, `bellevue_weather_simulated`, `cli_dice_roller_simulated`, `slack_channel_description_simulated`; and four "live" tasks as structural (their graders never debug): `smoke_error`, `jdbc_databricks_query`, `webhook_waitfor_parallel`, `testmanager_crud_grounded` (self-reported result file, as Flow). The live bucket is therefore 21 tasks. + +## Live bucket status + +| Task | State | Evidence | +|---|---|---| +| e2e/customer_escalation_triage | green (already on main) | pre-existing | +| e2e/jira_get_issue | green | run 35501830119 | +| e2e/jira_create_issue | green | run 35503094182 | +| e2e/escalation_jira_ticket | green | run 35503094182 | +| e2e/escalation_orchestrator_paths | green (7 debug runs) | run 35503094182 | +| e2e/escalation_slack_alert | green on iteration 2 | run 35524004307; iteration 1 the agent omitted the Slack `folderKey` binding (runtime 102010) | +| multi_node/slack_channel_description | green on iteration 2 | run 35525387843; iteration 1 the agent omitted the channel parameter | +| multi_node/bellevue_weather | parked, skill gap | runs 35523787101 + 35525387843: identical runtime fault, the script task reads `temperature_2m` off an undefined HTTP response. The skill does not teach the `Intsvc.HttpExecution` response shape well enough for downstream scripts | +| e2e/jira_search_triage | parked, skill/platform gap | run 35525387843: runtime 400008 "Failed to evaluate the input collection variable for the marker element" — `multiInstanceLoopCharacteristics` over a connector response (`=vars.Var_SearchResponse.issues`) does not evaluate | +| e2e/jira_lifecycle | parked, needs live investigation | three different runtime failures in three runs (our CLI poll cap bug; instance never terminal in 720 s; `bpmn debug` exit 1 before creating an instance). Flow's own version is flaky (0.82 typical, 2/12 zero in the week's nightlies) | +| multi_node/slack_weather_pipeline | written, in CI batch 10 | run 35538279757 | +| multi_node/billing_invoice_lookup | written, in CI batch 10 | run 35538279757 | +| multi_node/billing_discrepancy_detector | written, in CI batch 10 | run 35538279757 | +| connector_features/generic_dynamic_node | written, in CI batch 10 | run 35538279757 | +| connector_features/slack_http_fallback | written, in CI batch 10 | run 35538279757 | +| connector_features/jdbc_databricks_query (structural) | written, in CI batch 10 | run 35538279757 | +| connector_trigger/webhook_waitfor_parallel (structural) | written, in CI batch 10 | run 35538279757 | +| connector_features/datafabric_connector/smoke_error (structural) | written, in CI batch 10 | run 35538279757 | +| connector_features/testmanager_crud_grounded (self-report, Flow `skip:true` dropped) | written, in CI batch 10 | run 35538279757 | +| interactive/bellevue_weather_simulated | being written (agent in flight when this doc was cut) | — | +| interactive/cli_dice_roller_simulated | being written | — | +| interactive/slack_channel_description_simulated | being written | — | + +Batch 10 results and the three interactive ports are appended in the "Batch 10 and after" section when they land; if that section is missing, read `parity-ledger.md` or re-run the batch. + +## Probe bucket (16), not started except the pilot + +`connector_features/ceql_where` is the pilot for the 12 Integration Service field-shape evals (CEQL filter, complex_array, enum, enhanced_enum, multiselect, path_params, query_params, searchable_joins, generate_schema, dtl_load_by_default ×2, paginated_reference_lookup). An agent was probing it when this doc was cut; its verdict decides the other 11. The remaining 4 probes need a published agent substitute (billing_dispute_analyst / _resolution / _writer use Flow inline agents) or a file-typed process variable (single_node/file_attachment). + +## Methodology (how each port is made) + +1. **One Sonnet subagent per task** with `PORTING-BRIEF.md` (faithfulness table, construct translation, mandatory grading contract), `BATCH1-ADDENDUM.md` (connector node forms, criterion translations, staging paths), and for live ports `LIVE-ADDENDUM.md`. Ports are born normalized: every grader assertion is tagged F (translation of a cited Flow assertion), I (artifact plumbing) or T (a listed tolerance); anything else is not written. `NORMALIZATION.md` is the pass that retro-fitted the first 15 ports to that rule. +2. **Review = mechanical cross-check + assertion map.** The cross-check (a small script used throughout; see the parity ledger notes) compares criteria type/order/weight/threshold/timeout, run_limits, tags, prompt literals, pre_run/post_run, relative paths against the Flow source. Deviations allowed: CLI verbs, grader implementation, live criterion timeouts sized to the priced BPMN CLI sequence, `task_timeout` = turn_timeout + grading + 60 when Flow's does not cover it. +3. **One CI dispatch per batch** (`gh workflow run run-coder-eval.yml --ref -f task_globs='…'`), default codex driver, alpha tenant. Results: `gh run download ` → `**/task.json` → `success_criteria_results`; artifacts under `**/00/artifacts/` hold the agent's `.bpmn` for grader regression. +4. **Iteration rule:** fix only port defects (grader over-strictness, wrong construct name); max 3 graded iterations; a repeat runtime/skill failure parks the task with evidence in the ledger. Never weaken a Flow assertion to go green. + +## Live-grader recipe (proven on 7 tasks) + +Canonical: `_shared/check_jira_get_issue.py`. Sequence via `_shared/bpmn_live.py`: `uip solution init /` under the sandbox CWD → `uip solution projects import --solutionFile <.uipx>` (assert sha256 of the imported `.bpmn` equals the submitted one) → `run_debug(project, inputs, log, timeout)` → `debug-instance variables-all ` → `debug-instance incidents `. Grade from `variables-all`: root scope Globals plus every element's `Outputs` (root public output values have read back `null`; element outputs are reliable; connector responses via `connector_response_values`). post_run `_setup/cleanup_solutions.py` sweeps the ephemeral solution. + +Budget: `_shared/test_criterion_budgets.py` prices every `run_debug` call; criterion `timeout` ≥ solution init 90 + import 180 + debug 480 + variables-all 120 + incidents 120 + margin 60 = 1050 for one debug run (per-case loops multiply the debug/variables terms; annotate `# budget-guard: manual xN` when the loop count is not a module literal). + +## Runtime and CLI facts learned (grade around them) + +- `uip maestro bpmn debug` polls at most 300 times; `run_debug` derives `--poll-interval` from its timeout and raises a clear failure on the CLI's poll-timeout envelope or a `subprocess.TimeoutExpired` (carrying the CLI's stderr log). +- Incident 102010 "Value cannot be null (Parameter 'Folder')" on a Slack activity = missing `folderKey` binding (agent defect). 102009 "Parameter '' has null or empty value" = missing activity parameter (agent defect). 400008 on a multi-instance marker = input collection over a connector response does not evaluate (platform/skill). +- Actions.HITL needs a deployed Action App to pass `bpmn validate`; HITL ports reach parity on every other criterion (decision: leave as is). +- The eval agent emits connector nodes in two forms (curated `objectName` with path/query inputs and a structured tree in `metadata`; or generic entity-CRUD with `objectName` = entity and the verb in `operation`/`method`). Graders accept both; inputs at any depth; body optional outside Create/Update. +- `_shared/validate_bpmn.py` validates every `.bpmn` in the sandbox: the old "any file validates" loop passed on a stray `bpmn init` scaffold. + +## Skill findings to report upstream + +Connector trigger parameters (Data Fabric entity, Outlook `parentFolderId`) are omitted on first attempts; `uip is triggers objects/describe` discovery is not taught; Slack `folderKey` binding omitted; `Intsvc.HttpExecution` response shape unclear to downstream scripts; multi-instance over connector output fails at runtime; no "existing solutions → ask" greenfield rule; Actions.HITL requires a tenant Action App. + +## Resuming + +1. `git checkout test/bpmn-port-live` (stacked on the PR branch; rebase after the PR merges). +2. Read batch 10's run (35538279757) if the "Batch 10 and after" section is missing; record results in `parity-ledger.md`. +3. Finish or re-spawn the three interactive live ports and the `ceql_where` probe (briefs in this directory; spawn prompts followed the pattern "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; create ; gates; report with assertion map"). +4. Dispatch new ports as one batch; iterate per the rule; park with evidence. +5. Decide the 12 field-shape probes from the `ceql_where` verdict; the 4 agent/file-typed probes need tenant fixtures first. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md new file mode 100644 index 0000000000..24f910539d --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -0,0 +1,212 @@ +# Flow → BPMN eval parity map + +Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 + +| Bucket | Count | +|---|---| +| Ported 1:1 | 21 | +| Covered by an equivalent BPMN task | 18 | +| Portable — structural (authoring + validate) | 29 | +| Portable — live (bpmn debug + tenant re-read) | 17 | +| Portable pending a feasibility probe | 16 | +| Not portable (Flow-only surface) | 30 | + +## Porting ledger (branch `test/bpmn-port-connectors`) + +| Flow task | BPMN port | CI result | Notes | +|---|---|---|---| +| `connector_features/drive_to_slack.yaml` | `connector_features/drive_to_slack/` | PASS 1/1 (run 35484984200) | pilot | +| `connector_features/datafabric_connector/smoke_create_all_types.yaml` | `…/datafabric_connector/smoke_create_all_types/` | PASS (run 35489744689, iteration 2) | grader widened to generic entity-CRUD form | +| `connector_features/datafabric_connector/integration_create_get.yaml` | `…/datafabric_connector/integration_create_get/` | PASS (run 35489744689, iteration 2) | same | +| `connector_features/datafabric_connector/contractregistry_crud_filters.yaml` | `…/datafabric_connector/contractregistry_crud_filters/` | PASS (run 35489744689, iteration 2) | same + transitive output mapping | +| `connector_features/datafabric_connector/smoke_query.yaml` | `…/datafabric_connector/smoke_query/` | PASS (run 35490499651, iteration 3; run 35490198577 ERRORed on a tenant ping timeout) | sort in ORDER BY clause | +| `connector_features/datafabric_connector/smoke_update.yaml` | `…/datafabric_connector/smoke_update/` | PASS (run 35499789502) | batch 2 | +| `connector_features/datafabric_connector/smoke_file_activities.yaml` | `…/datafabric_connector/smoke_file_activities/` | PASS (run 35499789502) | batch 2 | +| `connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml` | `…/datafabric_connector/e2e_contract_intake_pipeline/` | PASS (run 35499789502) | batch 2 | +| `connector_features/datafabric_connector/trigger_lifecycle.yaml` | `…/datafabric_connector/trigger_lifecycle/` | PASS it.3 (run 35501830119) after two grader fixes (it.1 was a real agent omission of the entity param; it.2 a grader over-strictness) | | +| `connector_features/testmanager_attachments/…` | `connector_features/testmanager_attachments/` | PASS (run 35499789502) | batch 2 | +| `connector_features/testmanager_execution_results/…` | `connector_features/testmanager_execution_results/` | PASS (run 35499789502) | batch 2 | +| `connector_features/testmanager_generic_records/…` | `connector_features/testmanager_generic_records/` | PASS (run 35499789502) | batch 2 | +| `connector_features/testmanager_requirement_lifecycle/…` | `connector_features/testmanager_requirement_lifecycle/` | PASS (run 35499789502) | batch 2 | +| `connector_features/testmanager_testset_lifecycle/…` | `connector_features/testmanager_testset_lifecycle/` | PASS (run 35499789502) | batch 2 | +| `connector_features/datafabric_connector/smoke_update_existing_flow.yaml` | `…/datafabric_connector/smoke_update_existing_flow/` | PASS (run 35500726138) | batch 3, brownfield scaffold via bpmn init | +| `connector_features/non-catalog-http-fallback/…` | `connector_features/non_catalog_http_fallback/` | PASS (run 35500726138) | batch 3 | +| `single_node/outlook_waitfor_email/…` | `single_node/outlook_waitfor_email/` | PASS (run 35500726138) | batch 3 | +| `single_node/outlook_trigger_inbox/…` | `single_node/outlook_trigger_inbox/` | PASS it.2 (run 35501830119); it.1 the agent omitted parentFolderId | advisory `uip is triggers` telemetry stays 0 (skill does not teach trigger discovery) | +| `e2e/devcon_expense_approval.yaml` | `e2e/devcon_expense_approval/` | PARTIAL 0.885 it.3 (run 35503094182) | every HITL/schema/wiring assertion passes; only `validate` fails (Actions.HITL MISSING_BINDING: needs a deployed Action App; Flow's inline quick-form has no tenant dependency). Iterations exhausted; parity-minus-validate, same platform gap as the two simulated HITL ports. | +| `e2e/jira_get_issue/…` | `e2e/jira_get_issue/` | PASS (run 35501830119) | LIVE pilot: ephemeral solution + bpmn debug + variables-all recipe works | +| `interactive/customer_escalation_simulated/…` | `interactive/customer_escalation_simulated/` | PASS 0.94 (run 35501830119) | only the advisory name check (threshold 0) missed, as in Flow | +| `interactive/expense_approval_simulated/…` | `interactive/expense_approval_simulated/` | PARTIAL 0.70 (run 35501830119) | everything passes except `validate`: Actions.HITL needs a deployed Action App binding (MISSING_BINDING with placeholder appId). Platform gap vs Flow's inline quick-form. | +| `interactive/hitl_schema_design_simulated/…` | `interactive/hitl_schema_design_simulated/` | PARTIAL 0.68 (run 35501830119) | same validate/Action App gap | +| `interactive/solution_select.yaml` | `interactive/solution_select/` | PARKED (skill gap, run 35501830119) | BPMN skill has no existing-solution selection rule; agent auto-scaffolded `WeatherAlertSolution/` without asking, exactly as predicted. Port kept in tree as the documented gap. | +| `e2e/jira_create_issue/…` | `e2e/jira_create_issue/` | PASS (run 35503094182) | live | +| `e2e/escalation_jira_ticket/…` | `e2e/escalation_jira_ticket/` | PASS (run 35503094182) | live | +| `e2e/escalation_orchestrator_paths/…` | `e2e/escalation_orchestrator_paths/` | PASS (run 35503094182) | live, 7 debug runs | +| `e2e/escalation_slack_alert/…` | `e2e/escalation_slack_alert/` | PASS it.2 (run 35524004307); it.1 the agent omitted the Slack folderKey binding (runtime 102010) | live | +| `e2e/jira_lifecycle/…` | `e2e/jira_lifecycle/` | PARKED after 3 iterations | it.1 our poll-cap bug; it.2 instance never terminal in 720s; it.3 (run 35525387843) `bpmn debug` exited 1 before creating an instance. Three different runtime failures of a multi-instance Jira loop; Flow's own version is flaky (0.82 typical, 2/12 zero). Needs a live investigation, not more retries. | +| `e2e/jira_search_triage/…` | `e2e/jira_search_triage/` | PARKED (skill gap) after 3 iterations | it.3 (run 35525387843) runtime 400008 "Failed to evaluate the input collection variable for the marker element": the multi-instance `inputCollection="=vars.Var_SearchResponse.issues"` over a connector response does not evaluate — multi-instance over connector output not taught/supported. | +| `multi_node/bellevue_weather/…` | `multi_node/bellevue_weather/` | PARKED (skill gap) after it.1+it.2 (runs 35523787101, 35525387843) | identical runtime fault both times: the script task reads `temperature_2m` off an undefined HTTP response — the skill does not teach the Intsvc.HttpExecution response shape well enough for downstream scripts. | +| `multi_node/slack_channel_description/…` | `multi_node/slack_channel_description/` | PASS it.2 (run 35525387843); it.1 the agent omitted the Slack channel parameter | live | +| `connector_trigger/trigger_with_filter.yaml` | — | PARKED (skill gap) | Flow asserts a structured `filter` tree (groupOperator + filters[], MST-8802 guard); BPMN `Intsvc.EventTrigger` declares `filter` only as an untyped object with no template placeholder and the skill says trigger properties are CLI-owned enrichment. Re-port once a persisted filter shape is documented. | +| `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | + +## Ported 1:1 (21) + +| Flow task | BPMN target / note | +|---|---| +| `edit/add_node/add_node.yaml` | edit/add_node (structural only; Flow is live) | +| `edit/add_output/add_output.yaml` | edit/add_output (structural only; Flow is live) | +| `edit/group_to_subflow/group_to_subflow.yaml` | edit/group_to_subflow (structural only; Flow is live) | +| `edit/move_node/move_node.yaml` | edit/move_node (structural only; Flow is live) | +| `edit/remove_node/remove_node.yaml` | edit/remove_node (structural only; Flow is live) | +| `edit/update_node/update_node.yaml` | edit/update_node (structural only; Flow is live) | +| `hitl/quality_01_schema_design.yaml` | hitl/quality_schema_design | +| `hitl/quality_02_result_downstream.yaml` | hitl/quality_result_downstream | +| `hitl/quality_03_boolean_decision.yaml` | hitl/quality_boolean_decision | +| `hitl/quality_04_brownfield_insert.yaml` | hitl/quality_brownfield_insert | +| `hitl/smoke_02_completed_port_wired.yaml` | hitl/smoke_completed_wired | +| `hitl/smoke_03_multi_outcome_routing.yaml` | hitl/smoke_multi_outcome_routing | +| `interactive/customer_escalation_triage/customer_escalation_triage.yaml` | e2e/customer_escalation_triage (live, this branch) | +| `multi_node/calculator/calculator.yaml` | multi_node/calculator (structural only; Flow is live) | +| `multi_node/customer_escalation/customer_escalation.yaml` | multi_node/customer_escalation | +| `multi_node/dice_roller/dice_roller.yaml` | multi_node/dice_roller (structural only; Flow is live) | +| `multi_node/feet_inches/feet_inches.yaml` | multi_node/feet_inches (structural only; Flow is live) | +| `multi_node/loop_multiply/loop_multiply.yaml` | multi_node/loop_multiply (structural only; Flow is live) | +| `multi_node/multi_city_weather/multi_city_weather.yaml` | multi_node/multi_city_weather (structural only; Flow is live) | +| `multi_node/reading_list/reading_list.yaml` | multi_node/reading_list (structural only; Flow is live) | +| `multi_node/wiki_pageviews/wiki_pageviews.yaml` | multi_node/wiki_pageviews (structural only; Flow is live) | + +## Covered by an equivalent BPMN task (18) + +| Flow task | BPMN target / note | +|---|---| +| `hitl/smoke_01_hitl_node_placed.yaml` | nodes/hitl_rpa_wrappers (HITL shell placed) | +| `single_node/api_workflow/api_workflow.yaml` | authoring/api_workflow_task (structural; Flow is live) | +| `single_node/coded_agent/coded_agent.yaml` | single_node/agent_job (same BPMN node; Flow is live) | +| `single_node/decision/decision.yaml` | author/gateway_sequence_flows (structural; Flow is live) | +| `single_node/delay/delay.yaml` | single_node/timer | +| `single_node/lowcode_agent/lowcode_agent.yaml` | single_node/agent_job (structural; Flow is live) | +| `single_node/openmeteo_weather/openmeteo_weather.yaml` | single_node/http_weather (structural; Flow is live) | +| `single_node/rpa/rpa.yaml` | single_node/rpa_job (structural; Flow is live) | +| `single_node/subflow/subflow.yaml` | single_node/subprocess | +| `single_node/switch/switch.yaml` | single_node/switch | +| `single_node/terminate/terminate.yaml` | single_node/terminate | +| `single_node/transform_filter/transform_filter.yaml` | single_node/script_task_filter | +| `single_node/transform_group_by/transform_group_by.yaml` | single_node/script_task_group_by | +| `single_node/transform_map/transform_map.yaml` | single_node/script_task_map | +| `smoke/init_validate.yaml` | smoke/author_validate | +| `smoke/merge_parallel_sync.yaml` | parallel/fork_join | +| `smoke/registry_discovery.yaml` | smoke/registry_discovery | +| `smoke/scheduled_trigger.yaml` | single_node/timer_start | + +## Portable — structural (authoring + validate) (29) + +| Flow task | BPMN target / note | +|---|---| +| `connector_features/datafabric_connector/contractregistry_crud_filters.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/integration_create_get.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/smoke_create_all_types.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/smoke_file_activities.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/smoke_query.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/smoke_update.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/smoke_update_existing_flow.yaml` | uipath-dataservice ActivityExecution payloads; validate-only | +| `connector_features/datafabric_connector/trigger_lifecycle.yaml` | Intsvc.EventTrigger Record Created/Updated + downstream ActivityExecution | +| `connector_features/drive_to_slack.yaml` | two ActivityExecution nodes; binary output chaining; validate-only | +| `connector_features/non-catalog-http-fallback/non_catalog_http_fallback.yaml` | generic HTTP connector (uipath-uipath-http) ActivityExecution | +| `connector_features/slack-http-fallback/slack_http_fallback.yaml` | connector-mode HTTP fallback (Intsvc.HttpExecution on Slack connection) | +| `connector_features/testmanager_attachments/testmanager_attachments.yaml` | one ActivityExecution per Test Manager operation; validate-only | +| `connector_features/testmanager_execution_results/testmanager_execution_results.yaml` | one ActivityExecution per Test Manager operation; validate-only | +| `connector_features/testmanager_generic_records/testmanager_generic_records.yaml` | one ActivityExecution per Test Manager operation; validate-only | +| `connector_features/testmanager_requirement_lifecycle/testmanager_requirement_lifecycle.yaml` | one ActivityExecution per Test Manager operation; validate-only | +| `connector_features/testmanager_testcase_lifecycle/testmanager_testcase_lifecycle.yaml` | one ActivityExecution per Test Manager operation; validate-only | +| `connector_features/testmanager_testset_lifecycle/testmanager_testset_lifecycle.yaml` | one ActivityExecution per Test Manager operation; validate-only | +| `connector_trigger/trigger_with_filter.yaml` | Intsvc.EventTrigger with structured filter tree | +| `e2e/devcon_expense_approval.yaml` | Actions.HITL + scriptTasks; schema-design judge | +| `interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml` | simulation harness is skill-agnostic; BPMN skill allows AskUserQuestion | +| `interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml` | simulation harness is skill-agnostic; BPMN skill allows AskUserQuestion | +| `interactive/customer_escalation_simulated/customer_escalation_simulated.yaml` | simulation harness is skill-agnostic; BPMN skill allows AskUserQuestion | +| `interactive/expense_approval_simulated/expense_approval_simulated.yaml` | simulation harness is skill-agnostic; BPMN skill allows AskUserQuestion | +| `interactive/hitl_schema_design_simulated/hitl_schema_design_simulated.yaml` | simulation harness is skill-agnostic; BPMN skill allows AskUserQuestion | +| `interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml` | simulation harness is skill-agnostic; BPMN skill allows AskUserQuestion | +| `interactive/solution_select.yaml` | existing-solution selection rule; check the BPMN skill states the same greenfield rule | +| `single_node/outlook_trigger_inbox/outlook_trigger_inbox.yaml` | Intsvc.EventTrigger startEvent with fresh parentFolderId reference resolution | +| `single_node/outlook_waitfor_email/outlook_waitfor_email.yaml` | Intsvc.WaitForEvent receiveTask with subject filter | + +## Portable — live (bpmn debug + tenant re-read) (17) + +| Flow task | BPMN target / note | +|---|---| +| `connector_features/datafabric_connector/smoke_error.yaml` | live 4xx on missing entity; debug + incidents | +| `connector_features/generic_dynamic_node/generic_dynamic_node.yaml` | generic activity + --object-name at registry get; live | +| `connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml` | JDBC ActivityExecution; live | +| `connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml` | Test Manager create+get round trip; live debug | +| `connector_trigger/webhook_waitfor_parallel.yaml` | parallelGateway + Intsvc.WaitForEvent (webhook) + HttpExecution self-trigger; live debug | +| `e2e/escalation_jira_ticket/escalation_jira_ticket.yaml` | sibling of the live escalation port on this branch; reuse escalation_is.py | +| `e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml` | exclusiveGateway paths + Orchestrator.* nodes; live debug | +| `e2e/escalation_slack_alert/escalation_slack_alert.yaml` | sibling of the live escalation port; reuse escalation_is.py | +| `e2e/jira_create_issue/jira_create_issue.yaml` | Jira ActivityExecution; live debug + tenant re-read; teardown journal | +| `e2e/jira_get_issue/jira_get_issue.yaml` | Jira ActivityExecution read-only; live debug + variables-all | +| `e2e/jira_lifecycle/jira_lifecycle.yaml` | multiInstance loop + exclusiveGateway + Jira create/comment/transition | +| `e2e/jira_search_triage/jira_search_triage.yaml` | Jira JQL search + multiInstance + add comment | +| `multi_node/bellevue_weather/bellevue_weather.yaml` | HttpExecution sendTask + scriptTask + exclusiveGateway; debug via bpmn_live | +| `multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml` | two DF reads in parallelGateway fork/join; no agent | +| `multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml` | Data Fabric via Intsvc.ActivityExecution (uipath-dataservice); no agent | +| `multi_node/slack_channel_description/slack_channel_description.yaml` | Intsvc.ActivityExecution Slack; Slack plumbing in e2e/customer_escalation_triage/escalation_is.py | +| `multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml` | HttpExecution + Slack ActivityExecution; live debug | + +## Portable pending a feasibility probe (16) + +| Flow task | BPMN target / note | +|---|---| +| `connector_features/ceql_where.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/complex_array.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/dtl_load_by_default_false.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/dtl_load_by_default_true.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/enhanced_enum.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/enum.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/generate_schema.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/multiselect.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/paginated_reference_lookup.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/path_params.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/query_params.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `connector_features/searchable_joins.yaml` | IS field-shape feature on Intsvc.ActivityExecution payload; one probe decides the group | +| `multi_node/billing_dispute_analyst/billing_dispute_analyst.yaml` | inline agent + context-grounding index: BPMN only has Orchestrator.StartAgentJob (published agent) — needs a published grounded agent on the tenant | +| `multi_node/billing_dispute_resolution/billing_dispute_resolution.yaml` | inline agent → StartAgentJob substitute | +| `multi_node/billing_resolution_writer/billing_resolution_writer.yaml` | inline agent → StartAgentJob substitute | +| `single_node/file_attachment/file_attachment.yaml` | file-typed process variable: confirm canvas variable contract supports it | + +## Not portable (Flow-only surface) (30) + +| Flow task | BPMN target / note | +|---|---| +| `bindings/idempotent_reconfigure.yaml` | tests `flow node configure` binding upsert; BPMN has no node configure. A bindings_v2.json correctness test would be new coverage, not a port | +| `bindings/multi_connector_independence.yaml` | tests `flow node configure` binding upsert; BPMN has no node configure. A bindings_v2.json correctness test would be new coverage, not a port | +| `bindings/no_duplicate_connection_bindings.yaml` | tests `flow node configure` binding upsert; BPMN has no node configure. A bindings_v2.json correctness test would be new coverage, not a port | +| `bindings/reconfigure_different_connection.yaml` | tests `flow node configure` binding upsert; BPMN has no node configure. A bindings_v2.json correctness test would be new coverage, not a port | +| `connector_features/datafabric_connector/smoke_activation_negative.yaml` | skill-routing smoke belongs to tests/tasks/activation, not a BPMN port | +| `connector_features/datafabric_connector/smoke_activation_positive.yaml` | skill-routing smoke belongs to tests/tasks/activation, not a BPMN port | +| `context-grounding/batch_transform/batch_transform.yaml` | Flow pattern node (batch transform) is Flow-only | +| `context-grounding/summarize/summarize.yaml` | Flow pattern node (deep-rag) is Flow-only | +| `conversational/conversational_chat_loop.yaml` | conversational agent loop is Flow-only | +| `evaluate/child_simulation/child_simulation_crud.yaml` | Flow evaluate capability (eval sets, simulations) has no BPMN counterpart | +| `evaluate/evaluator_type_choice.yaml` | Flow evaluate capability (eval sets, simulations) has no BPMN counterpart | +| `evaluate/inline_agent_eval/inline_agent_eval.yaml` | Flow evaluate capability (eval sets, simulations) has no BPMN counterpart | +| `evaluate/local_crud.yaml` | Flow evaluate capability (eval sets, simulations) has no BPMN counterpart | +| `evaluate/no_auto_upload.yaml` | Flow evaluate capability (eval sets, simulations) has no BPMN counterpart | +| `evaluate/simulation/simulation_crud.yaml` | Flow evaluate capability (eval sets, simulations) has no BPMN counterpart | +| `interactive/ixp_invoice_extraction_simulated/ixp_invoice_extraction_simulated.yaml` | IXP node is Flow-only | +| `ixp/e2e_01_invoice_extraction_greenfield.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/e2e_02_project_selection.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/e2e_03_project_creation_handoff/e2e_03_project_creation_handoff.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/e2e_04_build_mechanics.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/integration_handle_routing.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/routing.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/routing_listing.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/routing_negative.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/scaffold_minimal.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `ixp/scaffold_multinode.yaml` | IXP plugin is Flow-only; BPMN registry has no IXP node | +| `node_features/datafabric_native/integration_native_read_create.yaml` | native core.datafabric.* nodes are Flow-only; connector variant is covered by the DF connector ports | +| `smoke/inline_agent_robust.yaml` | inline agent is Flow-only; BPMN agents are published | +| `voice/voice_inbound_call.yaml` | voice nodes are Flow-only | +| `voice/voice_outbound_call.yaml` | voice nodes are Flow-only | From 9ad0dc62b2bea0ee7e7681d279858365ea9c34c8 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:32:33 -0700 Subject: [PATCH 03/35] test(bpmn): port bellevue_weather_simulated (live, simulated user) Faithful port of Flow interactive/bellevue_weather_simulated: same simulation persona and constraints, same five criteria and weights. Live grader runs the ephemeral solution + bpmn debug sequence, so that criterion's timeout is 1050 and task_timeout is raised to cover turns plus grading; every other run_limit is Flow's. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_weather_bpmn_simulated.py | 297 ++++++++++++++++++ .../bellevue_weather_simulated.yaml | 138 ++++++++ 2 files changed, 435 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py create mode 100644 tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py new file mode 100644 index 0000000000..a7739cefa1 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py @@ -0,0 +1,297 @@ +#!/usr/bin/env python3 +"""BellevueWeather (BPMN, simulated): a weather-HTTP node is present and live +output contains one branch message. + +Name-agnostic sibling of ``_shared/check_weather_bpmn.py`` (the committed +non-simulated Bellevue port): the two graders make exactly the same +assertions, in the same order, over the same live-debug surface. The only +difference is that this one never pins the project/file name -- the +simulated persona (see +``uipath-maestro-flow/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml``) +withholds "BellevueWeather" until asked, so a correctly-built, differently +named process must still be gradable. This mirrors how Flow's own +``check_weather_flow_simulated.py`` relates to the retired non-simulated +``check_weather_flow.py``: "Identical assertions to the retired non-simulated +original ... Name-agnostic runtime checker for the simulated variant." + +Ported from Flow `_shared/check_weather_flow_simulated.py` (itself the +name-agnostic sibling of the retired Flow `multi_node/bellevue_weather/ +check_weather_flow.py`), translated the same way `_shared/check_weather_bpmn.py` +already translates the non-simulated pair: a JSON node-type scan + inline +`flow debug` payload becomes an XML scan over the registry-driven +`Intsvc.HttpExecution` managed-HTTP shell (see +skills/uipath-maestro-bpmn/references/structural-bpmn.md, +references/registry-workflow.md) plus the BPMN live-debug surface +(`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical pattern: ephemeral +solution import, `bpmn debug`, `debug-instance variables-all`/`incidents`). +The canonical live grader this file's plumbing is modeled on is +`_shared/check_jira_get_issue.py`; the non-simulated sibling +`_shared/check_weather_bpmn.py` is the exact structural template. + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. + +Assertion map (Flow -> BPMN): + F check_weather_flow_simulated.py:20-21 + assert_flow_has_any_node_type( + ["core.action.http", "custom-codereval-openmeteoapis"]) + -> any element carrying an Intsvc.HttpExecution + uipath:activity wrapper. BPMN has no curated + Open-Meteo Integration Service connector, so + only the managed-HTTP construct from + PORTING-BRIEF's construct-translation table + remains; the Flow grader's connector-fallback + branch has nothing to translate to (same as + check_weather_bpmn.py). + F check_weather_flow_simulated.py:22 + run_debug(timeout=240) implicitly requires + finalStatus == "Completed" (flow_check.run_debug + raises on a non-Completed status internally; + `bpmn debug` returns only an instance id, so the + check is explicit here) + -> FinalStatus in COMPLETED_STATUSES and + debug-instance incidents is empty + F check_weather_flow_simulated.py:23-24 + assert_outputs_contain(payload, + ["nice day", "bring a jacket"], require_all=False) + -> either verdict string found among the root + scope's variable leaves AND every element's + Outputs in `debug-instance variables-all` + (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read + back null even when mapped correctly, so the + search is not scoped to one declared output + variable) + I locate/parse .bpmn, name-agnostic (the simulated persona + withholds the project name, mirroring Flow's own + name-agnostic glob for this variant: no fixed basename hint) + -> bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 + pin of the imported bytes against the submitted file -- + `bpmn debug` runs against an imported project, unlike + `flow debug`, which runs directly against the discovered + project directory -> LIVE-ADDENDUM canonical live pattern + (mirrors check_jira_get_issue.py / check_weather_bpmn.py) + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- + not in Flow; the `bpmn validate` criterion covers structure + +Flow's grader does not assert a Script node or a Decision node exists (only +the weather-API node type and the branch output are graded), so this checker +does not add a structural check for the exclusiveGateway either -- adding one +would be a BPMN-only requirement the Flow prompt/grader never had. +""" + +from __future__ import annotations + +import os +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# .../uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import ( # noqa: E402 + find_bpmn_file, + has_typed_uipath_extension, + resolve_project, +) +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +BPMN_NS = "http://www.omg.org/spec/BPMN/20100524/MODEL" +ACTIVITY_TYPE = "Intsvc.HttpExecution" +VERDICTS = ("nice day", "bring a jacket") + +LIVE_RUN_DIR = Path("bellevue-weather-simulated-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in bellevue_weather_simulated.yaml documents the +# arithmetic (identical to check_weather_bpmn.py's own budget): +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (600) does not cover this larger live-debug +# surface, so it is raised to 1050 -- the one sanctioned deviation +# LIVE-ADDENDUM allows, because the budget is a property of the CLI surface, +# not of what is graded. + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def find_http_execution_nodes(root: ET.Element) -> list[ET.Element]: + """Every element carrying an Intsvc.HttpExecution uipath:activity wrapper. + + Scans every descendant, not a fixed tag list (registry templates may emit + the managed-HTTP activity as sendTask, serviceTask, or a plain task) -- + mirrors bpmn_live.index_runtime_connectors' own scanning discipline. Skips + ``bpmn:extensionElements`` nodes themselves: ``has_typed_uipath_extension`` + matches an element whose OWN direct children include a matching + ``uipath:activity`` (the wrapper task) as well as an ``extensionElements`` + node (whose direct child literally is that ``uipath:activity``), which + would otherwise double-count every match once per node. + """ + return [ + el + for el in root.iter() + if el.tag != f"{{{BPMN_NS}}}extensionElements" + and has_typed_uipath_extension(el, "activity", ACTIVITY_TYPE) + ] + + +def collect_output_haystack(variables_data: object) -> str: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- it mirrors Flow's own + assert_outputs_contain(), which flattens the whole outputs payload. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return "\n".join(str(v) for v in leaves).lower() + + +def main() -> None: + # Name-agnostic: no hint. The simulated persona withholds the project + # name unless asked, so the submitted .bpmn may not be named + # "BellevueWeather*" -- find_bpmn_file() falls back to "exactly one + # .bpmn" or "the one with project.uiproj beside it" when several exist. + bpmn_path = find_bpmn_file() + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + http_nodes = find_http_execution_nodes(root) + if not http_nodes: + _fail( + f"{bpmn_path} has no element carrying an {ACTIVITY_TYPE} uipath:activity " + "wrapper (no managed-HTTP weather node found)" + ) + print(f"OK: bpmn has a managed-HTTP node ({ACTIVITY_TYPE})") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "BellevueWeatherSimulatedLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {incidents_list}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + haystack = collect_output_haystack(variables_data) + if not any(verdict in haystack for verdict in VERDICTS): + _fail( + f"outputs do not contain either verdict string {list(VERDICTS)!r}\n" + f"outputs: {haystack[:1000]}" + ) + print("OK: bpmn outputs contain a weather branch message") + print("PASS: all BellevueWeather (simulated) checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml new file mode 100644 index 0000000000..566a4eba2e --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml @@ -0,0 +1,138 @@ +# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral +# v1-base prompt, outcome-graded criteria at preserved weights, and +# arm-agnostic artifact discovery. +task_id: skill-bpmn-bellevue-weather-simulated +description: > + Bellevue weather process (HTTP -> script -> decision), but driven by a + simulated non-technical user who withholds requirements until asked. Tests + the agent's ability to clarify an ambiguous ask before building. + + Ported from Flow + `interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml`; + the persona/goal/constraints are unchanged apart from product-noun swaps + ("UiPath Flow" -> "UiPath Maestro BPMN process", "a `.flow` file" -> "a + `.bpmn` file"), the Managed HTTP construct and live-debug sequence follow + the same translation as the committed non-simulated port + (`multi_node/bellevue_weather/bellevue_weather.yaml` + + `_shared/check_weather_bpmn.py`), and the live-check criterion timeout + grows from Flow's 600s to 1050s to fund that larger live-debug surface + (ephemeral solution import + `bpmn debug` + `debug-instance + variables-all`/`incidents`, per LIVE-ADDENDUM's budget rule); `run_limits` + are otherwise Flow's verbatim, except `task_timeout`, which is raised from + 2400s to 2550s (turn_timeout 1200 + worst-case grading 1245 + 60s margin) + so the larger live-check budget still fits under the single turns+grading + watchdog. The graded assertions themselves (a weather-HTTP node exists; the + run's outputs contain one of the two verdict strings; the project was named + as elicited) are unchanged from Flow's own grader. +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", "node:decision", "feature:http", simulation] + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +run_limits: + # Ceiling on agent turns across the whole dialog, with generous headroom so a + # long build is never truncated mid-flow. The dialog length itself is bounded + # by simulation.max_turns below. Unchanged from Flow. + max_turns: 70 + turn_timeout: 1200 + # Flow's task_timeout (2400) does not cover this task's larger live-debug + # criterion (1050s vs Flow's 600s): turn_timeout (1200) + worst-case grading + # (180 validate + 1050 live check + 15 name check = 1245) + 60s margin = + # 2505; rounded up for per-criterion spawn overhead. task_timeout wraps + # turns AND grading under one watchdog (tests/README.md), so it must cover + # both, not grading alone. + task_timeout: 2550 + +# No initial_prompt — the simulator generates turn 1 from its persona + goal. + +reference: + directory: ../.. + +simulation: + enabled: true + + persona: | + You are a non-technical operations analyst. You have heard of UiPath but + have never built a process yourself. You speak in plain English, never in + code. You are friendly but impatient: if the agent dithers, prod it to + just build something. You cannot read or write code — if the agent shows + you code, say "I trust you, just make it work." + + goal: | + You want a UiPath Maestro BPMN process that checks today's weather for + the Bellevue office and tells you whether it's a nice day. + + OPENING MESSAGE: open the way you would to a colleague — a couple of short + sentences covering the outcome you want plus the rules that actually + matter to you: it is the Bellevue, Washington office; strictly above 60 + degrees Fahrenheit is a nice day and the result should say 'nice day', at + or below 60 should say 'bring a jacket'; and you want a short written + summary of the weather rather than raw numbers. Say that it has to be + built as a UiPath Maestro BPMN process — not an API workflow, not a + standalone script — inside a solution of the same name, and that you want + it validated before you sign off. Do not name files or paths and do not + use technical wording. Close by inviting the agent to ask you anything it + needs before it starts building. Keep the withheld items below to + yourself until it asks. + + WITHHELD — reveal each of these ONLY if the agent asks about it: + - The project name should be "BellevueWeather". + - Where the weather data comes from: the open-meteo public API, called + directly over HTTP — NOT an Integration Service connector. + - You do not require the project to execute before signing off. + + constraints: + - "Never paste or read code. If shown code, say 'I trust you, just make it work.'" + - "Do not mention 'HTTP', 'open-meteo', 'connector', 'node', 'script', or 'decision' unless the agent brings them up first." + - "Do not reveal the project name 'BellevueWeather' unless asked what to name it." + - "If the agent proposes getting the weather through an Integration Service connector, tell it to call the public open-meteo API directly over HTTP instead." + - "The one thing worth pushing back on: if the agent reports it built something that is not a Maestro BPMN process at all (e.g. an API workflow), tell it you specifically want a Maestro BPMN process — i.e. a `.bpmn` file in the workspace — and are not satisfied until there is one." + - "If the agent asks you anything before it starts building, answer every outstanding question in ONE message — include any withheld fact it asked about, stated plainly — and then let it build." + - "The moment the agent reports it has built the process, the conversation is over: reply with exactly <<>> on its own line. Do not ask for a rebuild, polish, re-checks, or alternative designs, and do not raise a requirement you have not mentioned yet — the grader verifies the result from here." + max_turns: 8 + stop_token: "<<>>" + stop_on_criteria_pass: false + check_criteria: end_of_dialog + n_trials: 1 + +success_criteria: + # Name-agnostic: any unambiguous generated .bpmn must validate. The exact + # project name is scored separately below. + - type: run_command + description: "uip maestro bpmn validate passes (any .bpmn)" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + # Runtime check: + - type: run_command + description: "BPMN debug runs; weather-HTTP node executed and output contains a verdict" + command: "python3 $REFERENCE_DIR/_shared/check_weather_bpmn_simulated.py" + timeout: 1050 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 + + # The project name is a WITHHELD, COSMETIC requirement (persona reveals it only + # if asked what to name it) — the process behaves identically whatever it's + # called. Non-gating: reports whether the agent elicited the exact name + # (score 1/0), but pass_threshold 0 keeps a cosmetic miss from failing the + # task and the low weight keeps it from dominating the score. + - type: run_command + description: "Project/BPMN file is named BellevueWeather (elicited from the user)" + command: "find . -iname 'BellevueWeather*.bpmn' | grep -q ." + timeout: 15 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 From 3754656640cc79920e9e4de483ee0296cf2879b4 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:32:33 -0700 Subject: [PATCH 04/35] test(bpmn): port cli_dice_roller_simulated (live, simulated user) Faithful port of Flow interactive/cli_dice_roller_simulated: same simulation, criteria, weights and criterion timeouts. Only task_timeout grows (2400 to 2800) so the multi-run live check fits under the single turns-plus-grading watchdog. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_dice_runs_simulated.py | 282 ++++++++++++++++++ .../cli_dice_roller_simulated.yaml | 154 ++++++++++ 2 files changed, 436 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py create mode 100644 tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py new file mode 100644 index 0000000000..9c605523e1 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py @@ -0,0 +1,282 @@ +#!/usr/bin/env python3 +"""DiceRoller (BPMN, simulated): a scriptTask runs and produces an integer in [1, 6]. + +Name-agnostic runtime checker for the simulated variant, modeled the same way +`_shared/check_weather_bpmn_simulated.py` relates to its own non-simulated +sibling: the simulated persona +(`uipath-maestro-flow/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml`) +withholds "DiceRoller" until asked, so a correctly-built, differently named +process must still be gradable -- no name hint is passed to `find_bpmn_file()`. + +Ported from Flow `_shared/check_dice_runs_simulated.py`, translated the same +way `_shared/check_weather_bpmn_simulated.py` and `_shared/check_jira_get_issue.py` +already translate a Flow live check: a JSON node-type scan + inline `flow +debug` payload becomes an XML scan for a `bpmn:scriptTask` element (the +`BPMN.ScriptTask` registry construct -- see +skills/uipath-maestro-bpmn/references/structural-bpmn.md "Script tasks -- +Jint authoring contract") plus the BPMN live-debug surface +(`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical pattern: ephemeral +solution import, `bpmn debug`, `debug-instance variables-all`/`incidents`). +The canonical live grader this file's plumbing is modeled on is +`_shared/check_jira_get_issue.py`. + +Note: the registry LOOKUP key is `BPMN.ScriptTask`, but a correctly-authored +scriptTask serializes its mapping `uipath:type` as `BPMN.Variables`, not the +literal string "BPMN.ScriptTask" (structural-bpmn.md: "The mapping's type +child is ``, not +`BPMN.ScriptTask`" -- any other value silently breaks script dispatch). So +this checker asserts a `bpmn:scriptTask` ELEMENT exists, never a literal +"BPMN.ScriptTask" string match, which would fail a correct build. + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. + +Assertion map (Flow -> BPMN): + F check_dice_runs_simulated.py:22 assert_flow_has_node_type(["core.action.script"]) + -> at least one bpmn:scriptTask element present + anywhere in the process (the BPMN.ScriptTask + registry construct) + F check_dice_runs_simulated.py:23 run_debug(timeout=600) implicitly requires + finalStatus == "Completed" (flow_check.run_debug + raises on a non-Completed status internally; + `bpmn debug` returns only an instance id, so the + check is explicit here) + -> FinalStatus in COMPLETED_STATUSES and + debug-instance incidents is empty + F check_dice_runs_simulated.py:24 assert_output_int_in_range(payload, 1, 6) + -> value-leaf search (never the whole JSON + payload -- see the docstring on + assert_output_int_in_range: "Extracts + integers from output values only, not from + the full debug payload") over the root + scope's Globals AND every element's Outputs + in `debug-instance variables-all` + (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read + back null even when mapped correctly, so the + search is not scoped to one declared output + variable) + I locate/parse .bpmn, name-agnostic (the simulated persona + withholds the project name) -> bpmn_check.find_bpmn_file()/ + resolve_project() + I ephemeral solution init + `solution projects import` + sha256 + pin of the imported bytes against the submitted file -- + `bpmn debug` runs against an imported project, unlike + `flow debug`, which runs directly against the discovered + project directory -> LIVE-ADDENDUM canonical live pattern + (mirrors check_jira_get_issue.py / check_weather_bpmn_simulated.py) + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- + not in Flow; the `bpmn validate` criterion covers structure + +Flow's grader does not assert anything about how the die is rolled (only that +a Script node exists and the run produces 1-6), so this checker does not +inspect the script source either -- adding a source-pattern check would be a +BPMN-only requirement the Flow prompt/grader never had. +""" + +from __future__ import annotations + +import os +import re +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# .../uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import elements, find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +LIVE_RUN_DIR = Path("dice-roller-simulated-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} +ROLL_LO, ROLL_HI = 1, 6 +INT_RE = re.compile(r"-?\d+") + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here and the +# criterion `timeout:` in cli_dice_roller_simulated.yaml documents the +# arithmetic (identical to check_jira_get_issue.py's / check_weather_bpmn_simulated.py's +# own budget): +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (1320) already covers this, so it is kept +# verbatim rather than raised. + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def collect_output_leaves(variables_data: object) -> list: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- it mirrors Flow's own + assert_output_int_in_range()/collect_outputs(), which flattens the whole + outputs payload (declared globals + element outputs) rather than the + entire debug response, so a stray digit inside an id, timestamp or other + metadata field elsewhere in the payload can never produce a false match. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return leaves + + +def find_int_in_range(leaves: list, lo: int, hi: int) -> int | None: + """First integer in [lo, hi] found in the stringified leaf values. + + Mirrors flow_check.assert_output_int_in_range's exact rule: regex over + the individual OUTPUT VALUE leaves only (never the raw variables-all JSON + blob, whose element ids, timestamps and status strings would spuriously + match a small target range like [1, 6]). + """ + haystack = "\n".join(str(v) for v in leaves) + for match in INT_RE.findall(haystack): + value = int(match) + if lo <= value <= hi: + return value + return None + + +def main() -> None: + # Name-agnostic: no hint. The simulated persona withholds the project + # name unless asked, so the submitted .bpmn may not be named + # "DiceRoller*" -- find_bpmn_file() falls back to "exactly one .bpmn" or + # "the one with project.uiproj beside it" when several exist. + bpmn_path = find_bpmn_file() + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + script_tasks = elements(root, "scriptTask") + if not script_tasks: + _fail(f"{bpmn_path} has no bpmn:scriptTask element (no Script node found)") + print(f"OK: bpmn has a scriptTask ({len(script_tasks)} found)") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "DiceRollerSimulatedLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {incidents_list}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + leaves = collect_output_leaves(variables_data) + roll = find_int_in_range(leaves, ROLL_LO, ROLL_HI) + if roll is None: + haystack = "\n".join(str(v) for v in leaves) + _fail( + f"No integer in [{ROLL_LO}, {ROLL_HI}] found in outputs\n" + f"Outputs: {haystack[:1000]}" + ) + print(f"OK: scriptTask present; dice value = {roll}") + print("PASS: all DiceRoller (simulated) checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml new file mode 100644 index 0000000000..195ee3f123 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml @@ -0,0 +1,154 @@ +# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral +# v1-base prompt, outcome-graded criteria at preserved weights, and +# arm-agnostic artifact discovery. +task_id: skill-bpmn-cli-dice-roller-simulated +description: > + Dice-roller Maestro BPMN process in interactive mode, driven by a simulated + non-technical user who withholds requirements until asked. Tests the + agent's ability to clarify ambiguous asks before building. + + Ported from Flow + `interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml`; the + persona/goal/constraints are unchanged apart from product-noun swaps + ("UiPath Flow" -> "UiPath Maestro BPMN process", "flow" -> "process", "a + `.flow` file" -> "a `.bpmn` file"), the Script node becomes a + `bpmn:scriptTask` (the `BPMN.ScriptTask` registry construct), and the + live-check criterion follows the same translation as the committed + live-simulated ports (`_shared/check_weather_bpmn_simulated.py`, + `_shared/check_jira_get_issue.py`): a JSON node-type scan + inline `flow + debug` payload becomes an XML scan for a `bpmn:scriptTask` element plus the + BPMN live-debug surface (ephemeral solution import, `bpmn debug`, + `debug-instance variables-all`/`incidents`, per LIVE-ADDENDUM's canonical + pattern). That surface's worst-case budget (90 solution init + 180 solution + import + 480 debug + 120 variables-all + 120 incidents + 60s margin = 1050s) + fits under Flow's own criterion timeout (1320s), so the criterion timeout is + kept verbatim; `run_limits.task_timeout` is raised from Flow's 2400s to + 2800s (turn_timeout 1200 + worst-case grading 1515 [180 validate + 1320 + live check + 15 name check] + 60s margin = 2775, rounded up to the nearest + 50 for per-criterion spawn overhead) so the larger live-check budget still + fits under the single turns+grading watchdog (tests/README.md). The graded + assertions themselves (a scriptTask node exists; the run's outputs contain + an integer in [1, 6]; the project was named as elicited) are unchanged from + Flow's own grader. +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", simulation] + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +run_limits: + # Ceiling on agent turns across the whole dialog, with generous headroom so a + # long build is never truncated mid-flow. The dialog length itself is bounded + # by simulation.max_turns below. Unchanged from Flow. + max_turns: 60 + turn_timeout: 1200 + # Flow's task_timeout (2400) does not cover this task's live-debug criterion + # once the BPMN ephemeral-solution surface is counted: turn_timeout (1200) + + # worst-case grading (180 validate + 1320 live check + 15 name check = 1515) + # + 60s margin = 2775; rounded up to 2800 for per-criterion spawn overhead. + # task_timeout wraps turns AND grading under one watchdog (tests/README.md), + # so it must cover both, not grading alone. + task_timeout: 2800 + +# No initial_prompt — the simulator generates turn 1 from its persona + goal. + +reference: + directory: ../.. + +simulation: + enabled: true + + persona: | + You are a non-technical operations analyst. You have heard of UiPath but + have never built a process yourself. You speak in plain English, never in + code. You are friendly but impatient: if the agent dithers, prod it to + just build something. You cannot read or write code — if the agent shows + you code, say "I trust you, just make it work." You also have no way to + open a link or type into a chat window, so anything that needs you to + drive it yourself is no good: the agent has to be able to run it and + report the number back here. + + goal: | + You want a UiPath Maestro BPMN process that rolls a single six-sided die + and shows you the number that came up. + + OPENING MESSAGE: open the way you would to a colleague — a couple of short + sentences covering the outcome you want plus the rules that matter: one + die, six sides, numbers 1 through 6, and you want the rolled number shown + as the result rather than buried somewhere nobody looks. State in this + first message that you have no way to open a link or type into a chat + window yourself, so it has to be something the agent can run and report + the number back from — do not save this for later. Say that it has + to be built as a UiPath Maestro BPMN process — not an API workflow, not a + standalone script — inside a solution of the same name, and that you want + it validated before you sign off. Do not name files or paths and do not use + technical wording. How the agent builds it is its own business. Close by + inviting the agent to ask you anything it needs before it starts building. + Keep the withheld items below to yourself until it asks. + + WITHHELD — reveal each of these ONLY if the agent asks about it: + - The project name should be "DiceRoller". + - You want the completed process to validate before you sign off. + - You want to see it actually produce a number when it runs before you sign off. + + constraints: + - "Never paste or read code. If shown code, say 'I trust you, just make it work.'" + - "Do not mention 'CLI', 'uip', 'node', or 'edge' unless the agent brings them up first." + - "Do not reveal the project name 'DiceRoller' unless asked what to name it." + - "The one thing worth pushing back on: if the agent reports it built something that is not a Maestro BPMN process at all (e.g. an API workflow), tell it you specifically want a Maestro BPMN process — i.e. a `.bpmn` file in the workspace — and are not satisfied until there is one." + - "If the agent asks you anything before it starts building, answer every outstanding question in ONE message — include any withheld fact it asked about, stated plainly — and then let it build." + - "Whatever the agent showed you actually working is the version you want kept. If it says it is swapping the process back to some other design now that the demo is done, tell it to leave the working version in place — you do not want to be handed something nobody has seen run." + - "When the agent reports it built the process, do not end yet if it hasn't shown you the result: before you sign off you want to see it actually run and produce a number 1 through 6. If it only says the process is built and validated, ask it once to run it and show you the roll. Once the agent has both built the process AND shown you a run that produced a number 1 through 6, the conversation is over: reply with exactly <<>> on its own line. Do not ask for a rebuild, polish, or alternative designs, and do not raise any requirement beyond seeing it actually work — the grader verifies the result from here." + max_turns: 8 + stop_token: "<<>>" + stop_on_criteria_pass: false + check_criteria: end_of_dialog + n_trials: 1 + +success_criteria: + # Name-agnostic: any unambiguous generated .bpmn must validate. The exact + # project name is scored separately below. + - type: run_command + description: "uip maestro bpmn validate passes (any .bpmn)" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 4.0 + pass_threshold: 1.0 + + # Runtime check (mirrors Flow's check_dice_runs_simulated.py): a + # bpmn:scriptTask exists and the live debug run's outputs contain an + # integer in [1, 6]. Supersedes a static grep for "random", which a + # process that never executes would still pass. + - type: run_command + description: "BPMN debug runs; scriptTask executed and output is an integer in [1,6]" + command: "python3 $REFERENCE_DIR/_shared/check_dice_runs_simulated.py" + # 1320 funds the ephemeral-solution live-debug surface (solution init + + # import + bpmn debug + variables-all + incidents = 990s) plus margin — + # see check_dice_runs_simulated.py's own budget comment. Kept verbatim + # from Flow, which already exceeds the 1050s this surface needs. + timeout: 1320 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + # The project name is a WITHHELD, COSMETIC requirement (persona reveals it only + # if asked what to name it) — the process behaves identically whatever it's + # called. Non-gating: reports whether the agent elicited the exact name + # (score 1/0), but pass_threshold 0 keeps a cosmetic miss from failing the + # task and the low weight keeps it from dominating the score. + - type: run_command + description: "Project/BPMN file is named DiceRoller (elicited from the user)" + command: "find . -iname 'DiceRoller*.bpmn' | grep -q ." + timeout: 15 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 From 29b0a48fc0980558421e7645e71f6ad438d89f9f Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:32:33 -0700 Subject: [PATCH 05/35] test(bpmn): port slack_channel_description_simulated (live, simulated user) Faithful port of Flow interactive/slack_channel_description_simulated with all five criteria kept (validate, advisory debug command, live grader, static channel regex, name advisory). Live criterion timeout sized to the BPMN CLI sequence (1050); everything else is Flow's verbatim. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../check_channel_description_simulated.py | 287 ++++++++++++++++++ .../slack_channel_description_simulated.yaml | 188 ++++++++++++ 2 files changed, 475 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py create mode 100644 tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py new file mode 100644 index 0000000000..04b01514d1 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py @@ -0,0 +1,287 @@ +#!/usr/bin/env python3 +"""SlackChannelDescription (simulated, BPMN): structural + live checks. + +Ported from Flow `interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml` +(via `tests/tasks/uipath-maestro-flow/_shared/check_channel_description_simulated.py`, +itself byte-identical in its assertions to the retired non-simulated Flow +original -- see that file's own docstring). Same scenario as the non-simulated +BPMN sibling (`multi_node/slack_channel_description/_shared/check_channel_description.py`): +a manual-start process retrieves the channel description of #office-bellevue via +the Slack Integration Service connector and outputs it, driven here by a +simulated non-technical user who withholds the channel and project name until +asked. Identical live sequence and Slack-node classification to the non- +simulated sibling; the only difference from that sibling is BPMN-file discovery +staying name-agnostic (the simulated persona withholds the project name, so no +hint is available -- mirrors Flow's own `flow_check._find_project`, which never +filters by name either). + +Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per +check. Nothing is created against the tenant by this scenario (it only reads a +channel's description), so there is no side-effect record to tear down -- +matching both Flow graders, neither of which has a teardown. + +Assertion map (Flow -> BPMN): + F check_channel_description_simulated.py:27 assert_flow_uses_connector_target('uipath-salesforce-slack') + -> find_connector_nodes(): any element carrying + Intsvc.ActivityExecution whose connectorKey context field + equals uipath-salesforce-slack. No operation filter -- Flow's + own assertion has none either. + F check_channel_description_simulated.py:28 run_debug(timeout=240) implicitly requires finalStatus == + "Completed" (flow_check.run_debug raises on a non-Completed + status internally; `bpmn debug` returns only an instance id, so + the check is explicit here) + -> FinalStatus in COMPLETED_STATUSES and debug-instance + incidents is empty + F check_channel_description_simulated.py:29 assert_outputs_contain(payload, ADDRESS_FRAGMENTS, require_all=True) + -> every fragment found among the root scope's variable leaves + AND every element's Outputs (incl. nested connector `response`) + in `debug-instance variables-all` (LIVE-ADDENDUM: a root PUBLIC + OUTPUT has read back null even when mapped correctly, so the + search is not scoped to one declared output variable) + I locate/parse .bpmn, name-agnostic (file exists, well-formed XML, project directory + resolved -- no name hint, since the simulated persona withholds the project name + until asked and the grader must not assume the agent used it) + -> bpmn_check.find_bpmn_file()/resolve_project() + I ephemeral solution init + `solution projects import` + sha256 pin of the imported + bytes against the submitted file -- `bpmn debug` runs against an imported project, + unlike `flow debug`, which runs directly against the discovered project directory + -> LIVE-ADDENDUM canonical live pattern (mirrors the non- + simulated sibling and e2e/jira_get_issue's checker) + DROPPED the HTTP-proxy fallback branch of assert_flow_uses_connector_target (a + `core.action.http.v2` node with bodyParameters.targetConnector) -- a legacy Flow + accommodation for connector-backed flows authored before native connector node + types existed. The BPMN skill's registry enrichment always emits + Intsvc.ActivityExecution for a connector activity (registry-workflow.md §3), so no + analogous construct exists to translate. Same drop as the non-simulated sibling. + DROPPED require_no_private_connector_values / require_sequence_integrity / + require_di_for_visible_elements / connection-binding checks -- not in Flow; the + `bpmn validate` criterion covers structure. +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET +from pathlib import Path + +# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_channel_description.py +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 + +from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared import bpmn_live # noqa: E402 +from _shared.bpmn_live import ( # noqa: E402 + CheckFailure, + connector_context, + get_ci, + incident_records, + payload_data, + root_scope, + run_cli, + sha256, +) + +CONNECTOR_KEY = "uipath-salesforce-slack" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" + +ADDRESS_FRAGMENTS = [ + "700 Bellevue Way NE", + "Suite 2000", + "Bellevue", + "WA 98004", +] + +LIVE_RUN_DIR = Path("slack-channel-description-simulated-live") +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 +COMPLETED_STATUSES = {"Completed", "Successful"} + +# Worst-case wall clock this checker can spend, priced the way +# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug +# call below passes no timeout/retries/backoff kwargs, so it prices at +# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). +# The surrounding CLI steps (solution init/import, variables-all, incidents) +# are not priced by that guard, so their sum is added by hand here, exactly as +# the non-simulated sibling's checker does for the identical sequence, and the +# criterion `timeout:` in slack_channel_description_simulated.yaml documents +# the arithmetic: +# 90 (solution init) + 180 (solution import) + 480 (debug) +# + 120 (variables-all) + 120 (incidents) = 990 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# Flow's own criterion timeout (600) does not fit the extra CLI steps a BPMN +# live grade needs (solution init/import, separate variables-all/incidents +# reads), so it is raised to 1050 -- the one sanctioned deviation from +# "criteria identical" (LIVE-ADDENDUM: a property of the CLI surface, not of +# what is graded). + + +def _fail(msg: str) -> None: + sys.exit(f"FAIL: {msg}") + + +def _leaves(value): + if isinstance(value, dict): + for v in value.values(): + yield from _leaves(v) + elif isinstance(value, list): + for v in value: + yield from _leaves(v) + elif value is not None: + yield value + + +def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: + """Every element carrying an Intsvc.ActivityExecution targeting connector_key. + + No operation filter: Flow's own assertion (`assert_flow_uses_connector_target`) + only requires SOME connector node for the key, not a specific op, so this + mirrors that breadth. Scans every descendant, not a fixed tag list (registry + templates may emit a connector activity as sendTask, serviceTask, or a plain + task) -- mirrors bpmn_live.index_runtime_connectors' own scanning discipline. + """ + found = [] + for node in root.iter(): + context = connector_context(node) + if context.get("connectorKey") != connector_key: + continue + if ACTIVITY_TYPE not in ET.tostring(node, encoding="unicode"): + continue + found.append(node) + return found + + +def collect_output_haystack(variables_data: object) -> str: + """Value leaves of the root scope's Globals AND every element's Outputs. + + A root public output has been observed to read back null even when + correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one + declared output variable -- it mirrors Flow's own assert_outputs_contain(), + which flattens the whole outputs payload. Element Outputs include a + connector's nested `response` object, which _leaves() flattens along with + everything else. + """ + leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + return "\n".join(str(v) for v in leaves).lower() + + +def main() -> None: + # Name-agnostic: the simulated persona withholds the project name until + # asked, so this grader (like Flow's simulated grader) must not assume the + # agent used "SlackChannelDescription" -- unlike the non-simulated + # sibling, which pins a NAME_HINT. + bpmn_path = find_bpmn_file() + raw = Path(bpmn_path).read_text(encoding="utf-8") + if CONNECTOR_KEY not in raw: + _fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") + print(f"OK: bpmn references {CONNECTOR_KEY}") + + try: + root = ET.parse(bpmn_path).getroot() + except ET.ParseError as exc: + _fail(f"{bpmn_path} is not well-formed XML: {exc}") + + connector_nodes = find_connector_nodes(root, CONNECTOR_KEY) + if not connector_nodes: + _fail( + f"bpmn does not reference a {CONNECTOR_KEY} connector node " + f"({ACTIVITY_TYPE})" + ) + print(f"OK: bpmn references a {CONNECTOR_KEY} connector node") + + project_dir = resolve_project(os.path.basename(bpmn_path)) + original_hash = sha256(Path(bpmn_path)) + + LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) + solution_dir = LIVE_RUN_DIR / "SlackChannelDescriptionSimulatedLiveEval" + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + solution_file = solution_files[0] + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_file), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + imported_project = solution_dir / project_dir.name + if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + + debug_data, instance_id = bpmn_live.run_debug( + imported_project, {}, LIVE_RUN_DIR / "debug.log" + ) + print(f"OK: debug completed (instance {instance_id})") + + final_status = get_ci(debug_data, "FinalStatus") + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + incidents_list = incident_records(incidents_data) + + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if incidents_list: + detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + if incidents_list is None: + raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") + if incidents_list: + raise CheckFailure(f"unexpected incidents: {incidents_list}") + print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + + haystack = collect_output_haystack(variables_data) + missing = [f for f in ADDRESS_FRAGMENTS if f.lower() not in haystack] + if missing: + _fail( + f"outputs missing address fragments {missing}; " + f"expected all of {ADDRESS_FRAGMENTS}\noutputs: {haystack[:1000]}" + ) + print("OK: bpmn outputs contain the Bellevue office address") + print("PASS: all SlackChannelDescription (simulated) checks passed") + + +if __name__ == "__main__": + try: + main() + except CheckFailure as error: + raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml new file mode 100644 index 0000000000..bb3a49776a --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml @@ -0,0 +1,188 @@ +task_id: skill-bpmn-slack-channel-description-simulated +description: > + Single-connector Slack BPMN process (read a channel's description and output + it), driven by a simulated non-technical user who withholds the channel and + project name until asked. Tests the agent's ability to clarify an ambiguous + ask before building. Executes: builds a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper for the uipath-salesforce-slack connector, + validates, then runs an ephemeral-solution import + `bpmn debug` and asserts + the fetched channel description (the Bellevue office address) lands in the + runtime outputs — needs a live Slack connection. + Ported from Flow `interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml`; + the connector node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper (connectorKey uipath-salesforce-slack; + `uip is activities list uipath-salesforce-slack`: curated GetConversationInfo, + objectName ConversationsInfo_GET, method GETBYID) instead of a Flow + connector node, `flow debug`'s single-call inline payload becomes an + ephemeral-solution import + `bpmn debug` + `debug-instance + variables-all`/`incidents` read (mirroring the CI-passing non-simulated BPMN + sibling, multi_node/slack_channel_description), and the live criterion's + timeout is raised from Flow's 600s to 1050s for the same reason that + sibling's was: the extra CLI steps (`solution init`/`solution projects + import`, separate `variables-all`/`incidents` reads) that BPMN's live + surface needs beyond Flow's single `flow debug` call. All five of Flow's + success criteria are carried over one-for-one, in Flow's order, with Flow's + weights and pass_thresholds: validate, an advisory `command_executed` on + `bpmn debug` (translating Flow's advisory on `flow debug`), the live domain + grader, a static `office-bellevue`/channel-ID check (translating Flow's + `flow_contains.py --regex` call into an inline `python3 -c` scan of every + `.bpmn` file, since BPMN has no `flow_contains.py` equivalent), and the + withheld-project-name advisory. +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", connector, simulation] + +run_limits: + # Ceiling on agent turns across the whole dialog, with generous headroom so a + # long build is never truncated mid-flow. The dialog length itself is bounded + # by simulation.max_turns below. Flow's run_limits, verbatim. + max_turns: 70 + task_timeout: 2400 + turn_timeout: 1200 + +# No initial_prompt — the simulator generates turn 1 from its persona + goal. + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + - type: template_dir + path: ../../_setup + mount_point: _setup + +reference: + directory: ../.. + +simulation: + enabled: true + + persona: | + You are a non-technical office manager. You have heard of UiPath but have + never built a process yourself. You speak in plain English, never in code. + You are friendly but impatient: if the agent dithers, prod it to just + build something. You cannot read or write code — if the agent shows you + code, say "I trust you, just make it work." + + goal: | + You want a UiPath Maestro BPMN process that grabs the description text off + one of your team's Slack channels and shows it to you. + + OPENING MESSAGE: open the way you would to a colleague — a couple of short + sentences covering the outcome you want plus the details that matter to + you: the channel is #office-bellevue in Slack, you want that channel's + description text as the result, and it has to work on its own without you + supplying the channel every time it runs. Say that it has to be built as a + UiPath Maestro BPMN process — not an API workflow, not a standalone + script — inside a solution of the same name, and that you want it + validated before you sign off. Do not name files or paths and do not use + technical wording. Close by inviting the agent to ask you anything it + needs before it starts building. Keep the withheld items below to + yourself until it asks. + + WITHHELD — reveal each of these ONLY if the agent asks about it: + - The project name should be "SlackChannelDescription". + - You do not require the project to execute before signing off. + + constraints: + - "Never paste or read code. If shown code, say 'I trust you, just make it work.'" + - "Do not mention 'connector' or 'node' unless the agent brings them up first." + - "The process must work for #office-bellevue without you supplying anything at run time. If the agent asks you to provide a channel ID when it runs, tell it to wire the #office-bellevue channel in directly so it runs unattended." + - "Do not reveal the project name 'SlackChannelDescription' unless asked what to name it." + - "The one thing worth pushing back on: if the agent reports it built something that is not a Maestro BPMN process at all (e.g. an API workflow), tell it you specifically want a Maestro BPMN process — i.e. a `.bpmn` file in the workspace — and are not satisfied until there is one." + - "If the agent asks you anything before it starts building, answer every outstanding question in ONE message — include any withheld fact it asked about, stated plainly — and then let it build." + - "The moment the agent reports it has built the process, the conversation is over: reply with exactly <<>> on its own line. Do not ask for a rebuild, polish, re-checks, or alternative designs, and do not raise a requirement you have not mentioned yet — the grader verifies the result from here." + max_turns: 8 + stop_token: "<<>>" + stop_on_criteria_pass: false + check_criteria: end_of_dialog + n_trials: 1 + +pre_run: + - command: 'python3 "_setup/preflight_connections.py" uipath-salesforce-slack=uipath-maestro-flow' + timeout: 120 + +success_criteria: + # Name-agnostic: any unambiguous generated .bpmn must validate. The exact + # project name is scored separately below. + - type: run_command + description: "uip maestro bpmn validate passes (any .bpmn)" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + # Advisory (non-gating): did the agent itself run `bpmn debug` during the + # session, translating Flow's "Advisory: live-v1 agent ran flow debug" + # command_executed check (same weight/pass_threshold/description). + - type: command_executed + description: "Advisory: live-v1 agent ran flow debug" + tool_name: "Bash" + command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+maestro\s+bpmn\s+debug' + min_count: 1 + weight: 1.5 + pass_threshold: 0.0 + + # Runtime check (mirrors the non-simulated BPMN sibling's + # check_channel_description.py, and Flow's own check_channel_description_simulated.py): + # a node targets the Slack connector AND bpmn debug retrieves the channel + # description into the runtime outputs (the Bellevue office address). + # Supersedes a static connector-key grep, which a process that never + # executes would still pass. THIS is the authoritative proof the correct + # channel was targeted — a wrong or unresolved channel fails here regardless + # of whether the process still names #office-bellevue anywhere. + # Timeout raised from Flow's 600s to 1050s — see the module docstring in + # check_channel_description_simulated.py for the arithmetic (solution init + + # import + debug + variables-all + incidents), the same raise the + # non-simulated sibling already made for the identical live sequence. + - type: run_command + description: "BPMN debug runs; Slack connector executed and output has the channel description" + command: "python3 $REFERENCE_DIR/_shared/check_channel_description_simulated.py" + timeout: 1050 + expected_exit_code: 0 + weight: 5.0 + pass_threshold: 1.0 + + # The channel #office-bellevue is a withheld requirement the persona reveals only + # when asked which channel. The runtime debug check above already PROVES the right + # channel was read (its description IS the Bellevue office address); this static + # check separately reports whether a specific channel was elicited and wired in. + # + # The Slack "Get Channel Info" activity (/ConversationsInfo/{conversationsInfoId}) + # is keyed by channel ID, so a correct process resolves "#office-bellevue" -> its + # channel ID (e.g. C0B50H7DE2F) and stores the ID — the human-readable name + # legitimately never survives into the .bpmn. A bare `office-bellevue` substring + # grep therefore FALSE-FAILS every correct implementation. So we pass on EITHER: + # (a) the process still names #office-bellevue (e.g. kept on an element label), OR + # (b) the connector node has a resolved Slack channel ID hardwired into the + # conversationsInfoId path parameter (a specific channel, not a run-time + # input the persona explicitly forbade). Case (b) is not coupled to any one + # workspace's ID — it matches the Slack channel-ID shape, and the runtime + # check above confirms it's the *correct* channel. + # Non-gating (pass_threshold 0): a legitimate ID-based encoding must never + # hard-fail the whole task — see the block comment; scored for signal only. + # Translates Flow's `flow_contains.py --regex '...'` call (identical regex) into + # an inline python3 scan of every .bpmn under the sandbox, since BPMN has no + # flow_contains.py equivalent in _shared/. + - type: run_command + description: "Flow targets the #office-bellevue channel (elicited from the user)" + command: "python3 -c 'import glob, re, sys; pat = re.compile(r\"(?i)(office-bellevue|conversationsInfoId[^A-Za-z0-9]+C[A-Z0-9]{6,})\"); paths = [p for p in glob.glob(\"**/*.bpmn\", recursive=True) if \"node_modules\" not in p]; sys.exit(0 if any(pat.search(open(p, encoding=\"utf-8\").read()) for p in paths) else 1)'" + timeout: 15 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 0 + + # The project name is a WITHHELD, COSMETIC requirement (persona reveals it + # only if asked what to name it) — the process behaves identically whatever + # it's called. Non-gating: reports whether the agent elicited the exact name + # (score 1/0), but pass_threshold 0 keeps a cosmetic miss from failing the + # task and the low weight keeps it from dominating the score. + - type: run_command + description: "Project/BPMN file is named SlackChannelDescription (elicited from the user)" + command: 'find . -name "SlackChannelDescription.bpmn" | grep -q .' + timeout: 15 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0 + +post_run: + - command: "python3 _setup/cleanup_solutions.py" + timeout: 120 From 83439304d3683da601c966ccdd562cba730dd6c0 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:32:33 -0700 Subject: [PATCH 06/35] test(bpmn): port ceql_where as the field-shape probe pilot Pilot for the twelve Integration Service field-shape evals. Same prompt, criteria and weights as Flow connector_features/ceql_where; the grader reads the CEQL where clause off the connector node's query inputs in either curated or generic form. Verdict for the other eleven is recorded in _porting/LIVE-HANDOFF.md. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_ceql_where.py | 269 ++++++++++++++++++ .../ceql_where/ceql_where.yaml | 72 +++++ 2 files changed, 341 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py new file mode 100644 index 0000000000..cc1f73bf73 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py @@ -0,0 +1,269 @@ +#!/usr/bin/env python3 +"""CEQL where (BPMN): verify the agent's planned filter JSON in +``where_detail.json`` carries a canonical CEQL filter tree per +``skills/uipath-platform/references/integration-service/activities.md`` +— section "Filter Trees (CEQL)" — and that the .bpmn file references the +registered Microsoft Entra (Azure AD) connector with the List Groups +operation, plus a Terminate end event for routing. + +Ported from Flow `connector_features/ceql_where.yaml`'s +``check_ceql_where_flow.py``: same scenario (plan a structured CEQL filter +tree for Entra's List Groups operation; build a connector node + Terminate +routing), translated from a JSON node/edge walk to an XML walk over the +registry-driven ``Intsvc.ActivityExecution`` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) and a +Terminate end event (see skills/uipath-maestro-bpmn/references/ +structural-bpmn.md, "Terminate (end events only)"). + +Why we still grade ``where_detail.json`` and not the node's live inputs: + The prompt forbids live node/connection configuration (no tenant in the + sandbox), which is what would populate the enriched `where`/`queryExpression` + input on the `Intsvc.ActivityExecution` node (registry-workflow.md §3 — + enrichment requires a live `--connection-id`). `uip maestro bpmn validate` + accepts a connector node with an empty/draft body, so requiring a fully + enriched body here would test something the prompt forbids and the CLI + doesn't enforce. This is the same rationale the Flow grader documents, and + it holds identically for BPMN: `where_detail.json` is the artifact the + prompt asks the agent to plan, so that is the artifact we grade. + +Assertion map (Flow → BPMN): + F check_ceql_where_flow.py:116-133 where_detail.json filter-tree shape → _check_where_detail() (verbatim: format-agnostic JSON check) + F check_ceql_where_flow.py:167-173 CONNECTOR_KEY referenced in flow → connector_task(root, CONNECTOR_KEY) present + F check_ceql_where_flow.py:140-154 _is_groups_operation node match → _is_groups_operation(task) + F check_ceql_where_flow.py:182 assert_flow_has_node_type(["terminate"]) → _has_terminate_end_event(root) + I locate/parse .bpmn → parse_bpmn() + T curated vs generic connector form → _is_groups_operation() accepts objectName containing "group" (any + curated spelling) OR objectName=="groups" with method GET / operation + List/list-groups (the generic form; confirmed live against the + uipath-microsoft-azureactivedirectory connector's `groups` object, + whose only describable object name is "groups" — the curated name + "ListGroups" is never a valid --object-name on its own) + DROPPED require_no_private_connector_values (not in Flow) + DROPPED require_sequence_integrity (not in Flow; `bpmn validate` is not graded here either, matching Flow) + DROPPED require_di_for_visible_elements (not in Flow) + DROPPED connection-binding check (Flow never checked connections; node is expected to stay draft, no live tenant) +""" + +from __future__ import annotations + +import glob +import json +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import NS, elements, fail, parse_bpmn # noqa: E402 + +CONNECTOR_KEY = "uipath-microsoft-azureactivedirectory" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +WHERE_DETAIL_GLOB = "**/where_detail.json" +EXPECTED_FIELD = "displayname" +EXPECTED_VALUE = "active" + + +# --- where_detail.json filter-tree checks (verbatim from Flow's +# check_ceql_where_flow.py — this artifact is a standalone JSON planning file +# unrelated to the .flow/.bpmn format, so the check does not change at all) --- + + +def _walk(node): + """Yield every dict in a nested filter tree (groups + leaves).""" + if isinstance(node, dict): + yield node + for v in node.values(): + yield from _walk(v) + elif isinstance(node, list): + for item in node: + yield from _walk(item) + + +def _leaf_field(n: dict): + return n.get("id") or n.get("fieldName") or n.get("field") or n.get("name") + + +def _leaf_value(n: dict): + v = n.get("value") + if isinstance(v, dict): + return v.get("value") + return v + + +def _looks_like_filter_tree(node) -> bool: + """A canonical filter-tree dict carries a numeric ``groupOperator`` and a + list of ``filters``. Used to locate the tree regardless of the key the + agent stored it under (e.g. top-level ``filter``, ``filterTree``, or + nested under ``plannedDetail.filter``).""" + return ( + isinstance(node, dict) + and isinstance(node.get("groupOperator"), (int, float)) + and isinstance(node.get("filters"), list) + ) + + +def _find_filter_tree(plan): + """Return the first filter-tree-shaped dict found anywhere in ``plan``. + The prompt asks the agent to capture a filter for review but does not pin + the JSON key, so accept the tree under any key.""" + for node in _walk(plan): + if _looks_like_filter_tree(node): + return node + return None + + +def _assert_filter_tree_shape(tree, *, source: str) -> None: + """Per Filter Trees (CEQL) doc: structured tree with numeric + groupOperator (0 = And, 1 = Or), at least one leaf with PascalCase + operator referencing displayName='active'. Leaves use ``id`` (canonical) + or fall back to ``fieldName``/``field``/``name`` for older shapes.""" + if not isinstance(tree, dict): + sys.exit(f"FAIL: {source} must be a filter-tree object") + + if not isinstance(tree.get("groupOperator"), (int, float)): + sys.exit( + f"FAIL: {source}.groupOperator must be a number " + "(0 = And, 1 = Or) — see Filter Trees (CEQL) doc" + ) + + filters = tree.get("filters") + if not isinstance(filters, list) or not filters: + sys.exit(f"FAIL: {source}.filters must be a non-empty list") + + leaves = [n for n in _walk(tree) if isinstance(n.get("operator"), str)] + if not leaves: + sys.exit(f"FAIL: {source} has no leaf filter with `operator`") + + fields = [_leaf_field(n) for n in leaves] + if not any(isinstance(f, str) and EXPECTED_FIELD in f.lower() for f in fields): + sys.exit( + f"FAIL: {source} leaves do not reference the displayName field " + f"(found fields: {[f for f in fields if f]})" + ) + + values = [_leaf_value(n) for n in leaves] + if not any(isinstance(v, str) and v.strip().lower() == EXPECTED_VALUE for v in values): + sys.exit( + f"FAIL: {source} has no leaf with value '{EXPECTED_VALUE}' " + f"(found values: {[v for v in values if v is not None]})" + ) + + +def _check_where_detail() -> None: + matches = glob.glob(WHERE_DETAIL_GLOB, recursive=True) + if not os.path.exists("where_detail.json") and not matches: + sys.exit("FAIL: where_detail.json not found") + path = "where_detail.json" if os.path.exists("where_detail.json") else matches[0] + try: + plan = json.load(open(path)) + except json.JSONDecodeError as e: + sys.exit(f"FAIL: {path} is not valid JSON: {e}") + + filter_tree = _find_filter_tree(plan) + if filter_tree is None: + sys.exit( + "FAIL: where_detail.json has no filter-tree object (a dict with a " + "numeric `groupOperator` and a `filters` list) under any key — " + "the prompt requires a structured CEQL filter tree" + ) + _assert_filter_tree_shape(filter_tree, source="where_detail.json filter tree") + + +# --- .bpmn structural checks --- + + +def context_inputs(task: ET.Element) -> list[ET.Element]: + return task.findall(".//uipath:input", NS) + + +def context_value(task: ET.Element, name: str) -> str: + for inp in context_inputs(task): + if inp.attrib.get("name") == name: + return inp.attrib.get("value") or (inp.text or "") + return "" + + +def has_type(el: ET.Element, token: str) -> bool: + return token in ET.tostring(el, encoding="unicode") + + +def connector_task(root: ET.Element, connector_key: str) -> ET.Element | None: + for task in elements(root, "sendTask"): + if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == connector_key: + return task + return None + + +def _is_groups_operation(task: ET.Element) -> bool: + """Match either connector form (see BATCH1-ADDENDUM "Lessons from batch 1 + CI"): a curated objectName that names the group (contains "group"), or + the generic object form — objectName=="groups" with method GET / an + operation naming List — confirmed live against the connector's + enrichment (`uip is activities list uipath-microsoft-azureactivedirectory + --output json`: ListGroups -> ObjectName "groups", MethodName "GET"; + `uip maestro bpmn registry get Intsvc.ActivityExecution --object-name + groups --operation List` -> Operation.Name "List", Curated "List Groups"). + """ + if context_value(task, "connectorKey") != CONNECTOR_KEY: + return False + + object_name = (context_value(task, "objectName") or "").lower() + if "group" in object_name: + return True + + method = (context_value(task, "method") or "").upper() + operation = (context_value(task, "operation") or "").lower() + path = (context_value(task, "path") or "").rstrip("/").lower() + return ( + object_name == "groups" + and (method == "GET" or "list" in operation) + and (not path or path.endswith("/groups")) + ) + + +def _has_terminate_end_event(root: ET.Element) -> bool: + for end in elements(root, "endEvent"): + if end.find("bpmn:terminateEventDefinition", NS) is not None: + return True + return False + + +def _check_bpmn_structure() -> None: + path, root = parse_bpmn("CeqlWhereTest") + + task = connector_task(root, CONNECTOR_KEY) + if task is None: + fail( + f"BPMN does not reference the registered Azure AD / Entra connector " + f"key {CONNECTOR_KEY!r} on a bpmn:sendTask carrying {ACTIVITY_TYPE}. " + "Display names like 'Microsoft Entra' or 'Microsoft Entra ID' are " + "NOT registry keys — confirm the registered key with " + "`uip maestro bpmn registry search`." + ) + + if not _is_groups_operation(task): + fail( + f"connector sendTask does not target the List Groups operation " + f"(objectName={context_value(task, 'objectName')!r}, " + f"method={context_value(task, 'method')!r}, " + f"operation={context_value(task, 'operation')!r})" + ) + print(f"OK: {CONNECTOR_KEY} sendTask targets List Groups") + + if not _has_terminate_end_event(root): + fail("no bpmn:endEvent with bpmn:terminateEventDefinition found") + print(f"OK: {path} has a Terminate end event") + + +def main() -> None: + _check_where_detail() + _check_bpmn_structure() + print( + f"OK: where_detail.json carries canonical CEQL filter tree on " + f"displayName='{EXPECTED_VALUE}'; BPMN targets {CONNECTOR_KEY} " + "List Groups; Terminate end event present" + ) + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml new file mode 100644 index 0000000000..856c926890 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml @@ -0,0 +1,72 @@ +task_id: skill-bpmn-ceql-where +description: > + Tests the CEQL where query IS feature — agent plans a structured filter + tree for the Microsoft Entra (Azure AD) connector's "List groups" + operation with a `displayName = "active"` filter, following the canonical + shape documented in the Filter Trees (CEQL) section of the + uipath-platform skill. The process must reference the registered connector + key (`uipath-microsoft-azureactivedirectory`) and use an exclusive gateway + + Terminate end event for routing. This is an offline structure test: + product validation is not graded because the two authoring loops persist + an unconfigured connector differently. + Ported from Flow `connector_features/ceql_where.yaml`; the .flow JSON + node/edge graph becomes a .bpmn XML `bpmn:sendTask` carrying the registry + `Intsvc.ActivityExecution` wrapper, Flow's Decision node becomes a + `bpmn:exclusiveGateway`, and Flow's Terminate node becomes a `bpmn:endEvent` + with `bpmn:terminateEventDefinition`. The `where_detail.json` planning + artifact and its canonical CEQL filter-tree shape are graded unchanged — + the offline-planning rationale (no live tenant enrichment available to + either authoring loop) applies identically to Flow and to Maestro BPMN. +tags: [uipath-maestro-bpmn, integration, connector, ceql, filter, uipath-microsoft-azureactivedirectory, "mode:build"] + +run_limits: + expected_turns: 34 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +initial_prompt: | + This sandbox has no UiPath tenant, so you can't run the live node-configuration + step. Instead, capture the filter you'd apply to the operation as where_detail.json + so it can be reviewed. + + Build a UiPath Maestro BPMN process "CeqlWhereTest" with a manual start that + lists Microsoft Entra (Azure AD) groups whose displayName equals "active". + Find the registered Entra/Azure AD connector key before wiring the node. + + Branch on the result: if the call fails, stop the process immediately; + if it succeeds, log "CeqlWhere test passed". + Produce the final .bpmn file. Do not fabricate a tenant connection merely to + make product validation pass; the generated structure and the separately + reviewable filter are the deliverables in this offline task. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "where_detail.json (found anywhere under the solution) has a canonical CEQL filter tree on displayName='active'; process references the registered Azure AD / Entra connector key with List Groups + Terminate end event" + command: "python3 $REFERENCE_DIR/_shared/check_ceql_where.py" + timeout: 60 + expected_exit_code: 0 + weight: 8.0 + pass_threshold: 1.0 From 4e9c15b75ce7213db9a42084b40fd3156ceb893e Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:33:49 -0700 Subject: [PATCH 07/35] docs(bpmn): record the four late ports and the field-shape probe verdict Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/LIVE-HANDOFF.md | 30 ++++++++++++++----- 1 file changed, 22 insertions(+), 8 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index 95a2e478ff..d5fa53adcf 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -31,15 +31,29 @@ Reading the Flow graders during the loop reclassified four "structural" tasks as | connector_trigger/webhook_waitfor_parallel (structural) | written, in CI batch 10 | run 35538279757 | | connector_features/datafabric_connector/smoke_error (structural) | written, in CI batch 10 | run 35538279757 | | connector_features/testmanager_crud_grounded (self-report, Flow `skip:true` dropped) | written, in CI batch 10 | run 35538279757 | -| interactive/bellevue_weather_simulated | being written (agent in flight when this doc was cut) | — | -| interactive/cli_dice_roller_simulated | being written | — | -| interactive/slack_channel_description_simulated | being written | — | +| interactive/bellevue_weather_simulated | written, reviewed, not run | commit 80033663f; live criterion timeout 1050, task_timeout 2550 (sanctioned) | +| interactive/cli_dice_roller_simulated | written, reviewed, not run | commit 8fc7a7692; task_timeout 2800 (sanctioned) | +| interactive/slack_channel_description_simulated | written, reviewed, not run | commit f000d026d; all five Flow criteria kept, live timeout 1050 | -Batch 10 results and the three interactive ports are appended in the "Batch 10 and after" section when they land; if that section is missing, read `parity-ledger.md` or re-run the batch. +Batch 10 results are appended in the "Batch 10 and after" section when they land; if that section is missing, read `parity-ledger.md` or re-run the batch. The three interactive ports have never been dispatched: they go in the next batch together. -## Probe bucket (16), not started except the pilot +## Probe bucket (16): pilot ported, 11 decided, 4 blocked -`connector_features/ceql_where` is the pilot for the 12 Integration Service field-shape evals (CEQL filter, complex_array, enum, enhanced_enum, multiselect, path_params, query_params, searchable_joins, generate_schema, dtl_load_by_default ×2, paginated_reference_lookup). An agent was probing it when this doc was cut; its verdict decides the other 11. The remaining 4 probes need a published agent substitute (billing_dispute_analyst / _resolution / _writer use Flow inline agents) or a file-typed process variable (single_node/file_attachment). +`connector_features/ceql_where` is ported (commit fd6312fde), not yet run. The probe confirmed the filter carrier exists: `Intsvc.ActivityExecution` enrichment for the Entra `groups` List operation exposes a `where` parameter (type `query`, `FilterBuilder`, `hasCEQL: true`), and the CI-passing Data Fabric artifact carries the same tree as a `target="query" name="queryExpression" type="json"` input. As in Flow, the sandbox has no live tenant for enrichment, so the port grades the same standalone `where_detail.json` planning artifact plus the connector node and terminate end. No `bpmn validate` gate, matching Flow. Its one review flag: the groups-operation tolerance (objectName contains `group`, or `groups` + GET/list) has no CI-passed fixture yet. + +Verdict for the other 11 field-shape evals, from that probe: + +| Eval | Verdict | Carrier | +|---|---|---| +| path_params, query_params | portable, high confidence | `target="path"` / `target="query"` inputs, proven live | +| paginated_reference_lookup | portable | same query carrier (`pageSize`, `nextPage` seen in the Entra enrichment) | +| complex_array, multiselect | portable | nested JSON in the single `target="body"` CDATA, or array-valued query inputs | +| enum | portable | any literal `uipath:input` graded against the allowed set | +| enhanced_enum, searchable_joins | plausible, unverified | needs the `metadata` json context field or a `where`/`queryExpression` carrier; verify with a live `registry get` for the target connector first | +| generate_schema | uncertain | BPMN's schema surface is the opaque `jsonSchema` output contract that the skill says not to pre-empt offline; side-artifact grading or park | +| dtl_load_by_default ×2 | uncertain | `design.loadByDefault` is discovery-time metadata, not wire XML; portable only if Flow's grader checks the wire value | + +The remaining 4 probes need a published agent substitute (billing_dispute_analyst / _resolution / _writer use Flow inline agents) or a file-typed process variable (single_node/file_attachment). ## Methodology (how each port is made) @@ -70,6 +84,6 @@ Connector trigger parameters (Data Fabric entity, Outlook `parentFolderId`) are 1. `git checkout test/bpmn-port-live` (stacked on the PR branch; rebase after the PR merges). 2. Read batch 10's run (35538279757) if the "Batch 10 and after" section is missing; record results in `parity-ledger.md`. -3. Finish or re-spawn the three interactive live ports and the `ceql_where` probe (briefs in this directory; spawn prompts followed the pattern "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; create ; gates; report with assertion map"). +3. Dispatch the four never-run ports (three interactive simulated tasks + `ceql_where`) as one batch. New ports follow the spawn pattern "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; create ; gates; report with assertion map". 4. Dispatch new ports as one batch; iterate per the rule; park with evidence. -5. Decide the 12 field-shape probes from the `ceql_where` verdict; the 4 agent/file-typed probes need tenant fixtures first. +5. Port the 7 field-shape probes marked portable; verify the carrier live before the 2 plausible ones; the 4 agent/file-typed probes need tenant fixtures first. From e08f288e18778cb402e7867af457a2266712b215 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 20 Sep 2026 14:40:32 -0700 Subject: [PATCH 08/35] test(bpmn): fix four grader defects surfaced by batch 10 Batch 10 (run 35538279757) failed four ports on the grader, not the agent: identical .bpmn copies read as ambiguity, a live grader reading its own ephemeral import as a second project, wait-for-event classified by BPMN tag instead of the Intsvc.WaitForEvent wrapper, and HttpExecution-only where the skill also teaches Intsvc.UnifiedHttpRequest. Slack's generic resource for emoji.list is emoji_list_GET, so the fallback grader accepts both spellings. All four replay green on the downloaded CI artifacts. Ledger and handoff record the batch, including the two real failures left for iteration 2. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/LIVE-HANDOFF.md | 40 +++++++++++++------ .../_porting/parity-ledger.md | 9 +++++ .../uipath-maestro-bpmn/_shared/bpmn_check.py | 25 +++++++++++- .../check_billing_discrepancy_detector.py | 4 +- .../_shared/check_billing_invoice_lookup.py | 4 +- .../_shared/check_slack_http_fallback.py | 7 +++- .../_shared/check_webhook_waitfor_parallel.py | 19 ++++++--- .../_shared/test_bpmn_check.py | 34 +++++++++++++++- 8 files changed, 115 insertions(+), 27 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index d5fa53adcf..a196d934d0 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -22,20 +22,34 @@ Reading the Flow graders during the loop reclassified four "structural" tasks as | multi_node/bellevue_weather | parked, skill gap | runs 35523787101 + 35525387843: identical runtime fault, the script task reads `temperature_2m` off an undefined HTTP response. The skill does not teach the `Intsvc.HttpExecution` response shape well enough for downstream scripts | | e2e/jira_search_triage | parked, skill/platform gap | run 35525387843: runtime 400008 "Failed to evaluate the input collection variable for the marker element" — `multiInstanceLoopCharacteristics` over a connector response (`=vars.Var_SearchResponse.issues`) does not evaluate | | e2e/jira_lifecycle | parked, needs live investigation | three different runtime failures in three runs (our CLI poll cap bug; instance never terminal in 720 s; `bpmn debug` exit 1 before creating an instance). Flow's own version is flaky (0.82 typical, 2/12 zero in the week's nightlies) | -| multi_node/slack_weather_pipeline | written, in CI batch 10 | run 35538279757 | -| multi_node/billing_invoice_lookup | written, in CI batch 10 | run 35538279757 | -| multi_node/billing_discrepancy_detector | written, in CI batch 10 | run 35538279757 | -| connector_features/generic_dynamic_node | written, in CI batch 10 | run 35538279757 | -| connector_features/slack_http_fallback | written, in CI batch 10 | run 35538279757 | -| connector_features/jdbc_databricks_query (structural) | written, in CI batch 10 | run 35538279757 | -| connector_trigger/webhook_waitfor_parallel (structural) | written, in CI batch 10 | run 35538279757 | -| connector_features/datafabric_connector/smoke_error (structural) | written, in CI batch 10 | run 35538279757 | -| connector_features/testmanager_crud_grounded (self-report, Flow `skip:true` dropped) | written, in CI batch 10 | run 35538279757 | +| multi_node/slack_weather_pipeline | FAIL 0.375, iteration 1 of 3 | run 35538279757: runtime 300501 "Slack channel office-bellevue was not found" in the agent's channel-select script; agent defect (channel exists, Flow finds it) | +| multi_node/billing_invoice_lookup | green on the graded criteria (0.91); bindings advisory fixed, not re-run | run 35538279757; grader read its own ephemeral live solution as a second project | +| multi_node/billing_discrepancy_detector | FAIL 0.30, iteration 1 of 3 | run 35538279757: Integration Services 400 "Expected a field name expression but got 'StringValue'" on the ERP query (malformed Data Service filter, agent authoring); accountTier not derived from CRM | +| connector_features/generic_dynamic_node | green | run 35538279757 | +| connector_features/slack_http_fallback | 0.76; grader fixed, not re-run | run 35538279757: debug completed; grader wanted `emoji.list`, connector generic resource is `emoji_list_GET` | +| connector_features/jdbc_databricks_query (structural) | green | run 35538279757 | +| connector_trigger/webhook_waitfor_parallel (structural) | 0.47; grader fixed, not re-run | run 35538279757: agent used intermediateCatchEvent + WaitForEvent and Intsvc.UnifiedHttpRequest, both valid | +| connector_features/datafabric_connector/smoke_error (structural) | green | run 35538279757 | +| connector_features/testmanager_crud_grounded (self-report, Flow `skip:true` dropped) | 0.89; grader fixed, not re-run | run 35538279757: two byte-identical `.bpmn` (scaffold + solution copy) | | interactive/bellevue_weather_simulated | written, reviewed, not run | commit 80033663f; live criterion timeout 1050, task_timeout 2550 (sanctioned) | | interactive/cli_dice_roller_simulated | written, reviewed, not run | commit 8fc7a7692; task_timeout 2800 (sanctioned) | | interactive/slack_channel_description_simulated | written, reviewed, not run | commit f000d026d; all five Flow criteria kept, live timeout 1050 | -Batch 10 results are appended in the "Batch 10 and after" section when they land; if that section is missing, read `parity-ledger.md` or re-run the batch. The three interactive ports have never been dispatched: they go in the next batch together. +Batch 10 landed; see "Batch 10 and after". The three interactive ports have never been dispatched. + +## Batch 10 and after + +Run 35538279757 (nine ports, one dispatch). Green: smoke_error, generic_dynamic_node, jdbc_databricks_query. Four more failed only on grader defects, all fixed on this branch and replayed green against the downloaded CI artifacts (`gh run download 35538279757`, `**/00/artifacts/`): + +1. `find_bpmn_file` with no hint now treats byte-identical `.bpmn` copies as one artifact (testmanager_crud_grounded: the agent copied its scaffold into the solution wrapper). +2. `resolve_project(exclude_under=…)`: a live grader's own `uip solution projects import` leaves an identical project under its run directory; later criteria in the same task must exclude it (both billing graders pass `LIVE_RUN_DIR`). Any future multi-criterion live grader needs the same. +3. Classify wait-for-event by the `Intsvc.WaitForEvent` wrapper, not the BPMN tag: the agent emits `bpmn:intermediateCatchEvent` + messageEventDefinition as well as `bpmn:receiveTask`, and both validate (webhook_waitfor_parallel; same lesson as trigger_lifecycle for EventTrigger). +4. Accept `Intsvc.UnifiedHttpRequest` wherever a grader accepts `Intsvc.HttpExecution`; registry-workflow.md lists both for the managed HTTP sendTask. +5. Slack's generic resource for the `emoji.list` endpoint is `emoji_list_GET`; the fallback grader matches `emoji[._]list`. + +Two real failures, one iteration spent each: billing_discrepancy_detector (Integration Services 400 on the ERP query filter, "Expected a field name expression but got 'StringValue'": the agent wrote a malformed Data Service filter; add to the skill findings as "Data Service query filter grammar") and slack_weather_pipeline (script task could not find channel `office-bellevue`, which exists and Flow's agent finds; likely channel-list pagination). + +Next dispatch, one batch: the four grader-fixed tasks for confirmation, the two real failures (iteration 2), the three interactive simulated ports and `ceql_where` (first run). Ten tasks. ## Probe bucket (16): pilot ported, 11 decided, 4 blocked @@ -78,12 +92,12 @@ Budget: `_shared/test_criterion_budgets.py` prices every `run_debug` call; crite ## Skill findings to report upstream -Connector trigger parameters (Data Fabric entity, Outlook `parentFolderId`) are omitted on first attempts; `uip is triggers objects/describe` discovery is not taught; Slack `folderKey` binding omitted; `Intsvc.HttpExecution` response shape unclear to downstream scripts; multi-instance over connector output fails at runtime; no "existing solutions → ask" greenfield rule; Actions.HITL requires a tenant Action App. +Connector trigger parameters (Data Fabric entity, Outlook `parentFolderId`) are omitted on first attempts; `uip is triggers objects/describe` discovery is not taught; Slack `folderKey` binding omitted; `Intsvc.HttpExecution` response shape unclear to downstream scripts; Data Service query filter grammar (400 "Expected a field name expression"); Slack channel lookup misses existing channels (pagination); multi-instance over connector output fails at runtime; no "existing solutions → ask" greenfield rule; Actions.HITL requires a tenant Action App. ## Resuming 1. `git checkout test/bpmn-port-live` (stacked on the PR branch; rebase after the PR merges). -2. Read batch 10's run (35538279757) if the "Batch 10 and after" section is missing; record results in `parity-ledger.md`. -3. Dispatch the four never-run ports (three interactive simulated tasks + `ceql_where`) as one batch. New ports follow the spawn pattern "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; create ; gates; report with assertion map". +2. Dispatch the ten-task batch listed under "Batch 10 and after"; record results in `parity-ledger.md` and this table. +3. New ports follow the spawn pattern "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; create ; gates; report with assertion map". 4. Dispatch new ports as one batch; iterate per the rule; park with evidence. 5. Port the 7 field-shape probes marked portable; verify the carrier live before the 2 plausible ones; the 4 agent/file-typed probes need tenant fixtures first. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index 24f910539d..8a1009a436 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -48,6 +48,15 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | `multi_node/bellevue_weather/…` | `multi_node/bellevue_weather/` | PARKED (skill gap) after it.1+it.2 (runs 35523787101, 35525387843) | identical runtime fault both times: the script task reads `temperature_2m` off an undefined HTTP response — the skill does not teach the Intsvc.HttpExecution response shape well enough for downstream scripts. | | `multi_node/slack_channel_description/…` | `multi_node/slack_channel_description/` | PASS it.2 (run 35525387843); it.1 the agent omitted the Slack channel parameter | live | | `connector_trigger/trigger_with_filter.yaml` | — | PARKED (skill gap) | Flow asserts a structured `filter` tree (groupOperator + filters[], MST-8802 guard); BPMN `Intsvc.EventTrigger` declares `filter` only as an untyped object with no template placeholder and the skill says trigger properties are CLI-owned enrichment. Re-port once a persisted filter shape is documented. | +| `connector_features/datafabric_connector/smoke_error.yaml` | `…/datafabric_connector/smoke_error/` | PASS 1.0 (run 35538279757) | batch 10, structural | +| `connector_features/generic_dynamic_node/…` | `connector_features/generic_dynamic_node/` | PASS 1.0 (run 35538279757) | batch 10, live: ServiceNow acr_user list ran, empty array as expected | +| `connector_features/jdbc_databricks_query/…` | `connector_features/jdbc_databricks_query/` | PASS 1.0 (run 35538279757) | batch 10, structural | +| `multi_node/billing_invoice_lookup/…` | `multi_node/billing_invoice_lookup/` | PASS 0.91 it.1 (run 35538279757); grader fixed, not re-run | live: all three malformed inputs normalized and queried. Only the advisory `bindings` step failed, on the grader's own ephemeral live solution being read as a second project (fixed: `resolve_project(exclude_under=[LIVE_RUN_DIR])`). | +| `connector_features/testmanager_crud_grounded/…` | `connector_features/testmanager_crud_grounded/` | 0.89 it.1 (run 35538279757); grader fixed, not re-run | self-report round-trip verified live. The node-types criterion died on two byte-identical `.bpmn` (scaffold + solution copy); `find_bpmn_file` now treats identical copies as one artifact. Replays green on the CI artifact. | +| `connector_features/slack-http-fallback/…` | `connector_features/slack_http_fallback/` | 0.76 it.1 (run 35538279757); grader fixed, not re-run | live debug completed with no incidents. The fallback check looked for `emoji.list`; the Slack connector's generic resource for that endpoint is `emoji_list_GET`, which the agent used. Tolerance `emoji[._]list` added; replays green. | +| `connector_trigger/webhook_waitfor_parallel.yaml` | `connector_trigger/webhook_waitfor_parallel/` | 0.47 it.1 (run 35538279757); grader fixed, not re-run | agent emitted the wait as `bpmn:intermediateCatchEvent` + `Intsvc.WaitForEvent` (validates) and the GET as `Intsvc.UnifiedHttpRequest` (the skill lists it beside HttpExecution). Grader now classifies by wrapper type and accepts both HTTP types; replays green. Advisory `uip is webhooks config` telemetry 0, as feared. | +| `multi_node/billing_discrepancy_detector/…` | `multi_node/billing_discrepancy_detector/` | FAIL 0.30 it.1 (run 35538279757) | validate passed; live debug raised an Integration Services 400 on `Task_QueryERP`: "Expected a field name expression but got 'StringValue'" (malformed Data Service query filter, agent authoring). Advisory: `accountTier` did not derive from the CRM query. Also hit the same `bindings` grader defect (fixed). One iteration left to spend when live work resumes. | +| `multi_node/slack_weather_pipeline/…` | `multi_node/slack_weather_pipeline/` | FAIL 0.375 it.1 (run 35538279757) | validate passed; live debug incident 300501 in script task `Task_SelectChannel`: "Slack channel office-bellevue was not found" — the agent's channel lookup did not find a channel Flow's port finds (likely list pagination/limit). Agent defect, not grader; retry when live work resumes. | | `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | ## Ported 1:1 (21) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py index 7b8b516466..bc3e8d488a 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py @@ -54,10 +54,20 @@ def find_bpmn_file(name_hint: str | None = None) -> str: projects = _project_files(paths) if len(projects) == 1: return projects[0] + # Byte-identical copies (the agent copied its scaffold into the solution + # wrapper; CI run 35538279757, testmanager_crud_grounded) are one artifact. + if len({_sha256(p) for p in paths}) == 1: + return paths[0] fail(f"multiple BPMN files found; expected one or hint match: {paths}") -def resolve_project(bpmn_name: str) -> Path: +def _sha256(path: str | Path) -> str: + import hashlib + + return hashlib.sha256(Path(path).read_bytes()).hexdigest() + + +def resolve_project(bpmn_name: str, exclude_under: Iterable[Path] = ()) -> Path: """Locate the project directory containing ``bpmn_name``. Grades the project wherever the agent placed it (top level or nested under @@ -66,8 +76,19 @@ def resolve_project(bpmn_name: str) -> Path: project unambiguously: exactly one ``bpmn_name`` with project.uiproj beside it, so a stray draft copy is never graded (``find_bpmn_file`` would silently return the alphabetically-first match). + + ``exclude_under``: directories whose contents are not candidates. A live + grader's own ephemeral solution (``uip solution projects import``) leaves a + second, byte-identical project under its run directory; a later criterion + in the same task must not read that copy as ambiguity (CI run 35538279757, + billing_invoice_lookup ``bindings``). """ - candidates = _project_files(Path.cwd().rglob(bpmn_name)) + excluded = [Path(d).resolve() for d in exclude_under] + candidates = [ + p + for p in _project_files(Path.cwd().rglob(bpmn_name)) + if not any(p.resolve().is_relative_to(d) for d in excluded) + ] if len(candidates) != 1: fail( f"expected exactly one {bpmn_name} with project.uiproj beside it, " diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py index a0c3b5e2e0..37c2f5fa85 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py @@ -558,7 +558,7 @@ def bindings() -> None: root = ET.parse(bpmn_path).getroot() needs_connection = any(connector_context(node).get("connectorKey") for node in root.iter()) - project_dir = resolve_project(os.path.basename(bpmn_path)) + project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) with tempfile.TemporaryDirectory(prefix="billing-pack-") as out_dir: packed = subprocess.run( ["uip", "maestro", "bpmn", "pack", str(project_dir), out_dir, "--output", "json"], @@ -688,7 +688,7 @@ def detector() -> None: if not find_join_gateways(root): fail("no bpmn:parallelGateway acts as a join (>=2 incoming flows)") - project_dir = resolve_project(os.path.basename(bpmn_path)) + project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) original_hash = sha256(Path(bpmn_path)) LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py index 13e6b8d792..a2d0c8f0eb 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py @@ -392,7 +392,7 @@ def lookup() -> None: fail("process declares no public uipath:input variable for the invoice number") var_name = input_names[0] - project_dir = resolve_project(os.path.basename(bpmn_path)) + project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) original_hash = sha256(Path(bpmn_path)) LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) @@ -522,7 +522,7 @@ def is_real_connection_key(value) -> bool: def bindings() -> None: bpmn_path = find_bpmn_file(NAME_HINT) - project_dir = resolve_project(os.path.basename(bpmn_path)) + project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) with tempfile.TemporaryDirectory(prefix="bpmn-eval-pack-") as output_dir: result = subprocess.run( ["uip", "maestro", "bpmn", "pack", str(project_dir), output_dir, "--output", "json"], diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py index ea2ebab203..c3fbd94087 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py @@ -95,6 +95,7 @@ import json import os +import re import sys import xml.etree.ElementTree as ET from pathlib import Path @@ -117,7 +118,11 @@ SLACK_KEY = "uipath-salesforce-slack" # Slack endpoint that lists a team's custom emoji. Bare token, so both '/emoji.list' # and 'emoji.list' forms satisfy the check, mirroring Flow's own tolerance. +# T: the Slack connector's generic (non-curated) resource for that endpoint is +# named ``emoji_list`` / ``emoji_list_GET`` -- the same endpoint, connector +# naming (CI run 35538279757: the agent's node ran to completion against it). EMOJI_ENDPOINT = "emoji.list" +EMOJI_ENDPOINT_RE = re.compile(r"emoji[._]list") ACTIVITY_TYPES = ("Intsvc.ActivityExecution", "Intsvc.HttpExecution") LIVE_RUN_DIR = Path("slack-emoji-list-live") @@ -181,7 +186,7 @@ def is_slack_connector_node(el: ET.Element) -> bool: def references_emoji_endpoint(el: ET.Element) -> bool: - return EMOJI_ENDPOINT in node_blob(el) + return EMOJI_ENDPOINT_RE.search(node_blob(el)) is not None def find_slack_connector_nodes(root: ET.Element) -> list[ET.Element]: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py index d56cf9816e..2739cf4c9f 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py @@ -69,6 +69,9 @@ BPMN_NS = NS["bpmn"] WAIT_TYPE = "Intsvc.WaitForEvent" HTTP_TYPE = "Intsvc.HttpExecution" +# registry-workflow.md lists both wrappers for the managed HTTP sendTask; the +# eval agent emitted UnifiedHttpRequest on CI run 35538279757. +HTTP_TYPES = (HTTP_TYPE, "Intsvc.UnifiedHttpRequest") CONNECTOR_KEY = "uipath-http-webhook" @@ -132,10 +135,14 @@ def fan_out_point(root: ET.Element, start_id: str) -> str | None: def wait_for_event_nodes(root: ET.Element) -> list[ET.Element]: - """bpmn:receiveTask carrying the registry Intsvc.WaitForEvent wrapper, - bound to the HTTP Webhook connector.""" + """Element carrying the registry Intsvc.WaitForEvent wrapper, bound to the + HTTP Webhook connector. Classified by the wrapper type, not the BPMN tag: + the skill teaches ``bpmn:receiveTask``, but the eval agent also emits a + validating ``bpmn:intermediateCatchEvent`` + messageEventDefinition with + the same wrapper (CI run 35538279757), as it did for Intsvc.EventTrigger + in trigger_lifecycle.""" out = [] - for task in elements(root, "receiveTask"): + for task in list(elements(root, "receiveTask")) + list(elements(root, "intermediateCatchEvent")): if not has_typed_uipath_extension(task, "event", WAIT_TYPE): continue if context_value(task, "connectorKey") != CONNECTOR_KEY: @@ -149,7 +156,7 @@ def http_get_nodes(root: ET.Element) -> list[ET.Element]: URL, with nothing in headers or query parameters.""" good = [] for task in elements(root, "sendTask"): - if not has_typed_uipath_extension(task, "activity", HTTP_TYPE): + if not any(has_typed_uipath_extension(task, "activity", t) for t in HTTP_TYPES): continue mode = context_value(task, "mode").lower() method = context_value(task, "method").upper() @@ -190,7 +197,7 @@ def main() -> None: event_nodes = wait_for_event_nodes(root) if not event_nodes: fail( - f"no bpmn:receiveTask carrying {WAIT_TYPE} bound to connectorKey " + f"no bpmn:receiveTask or intermediateCatchEvent carrying {WAIT_TYPE} bound to connectorKey " f"{CONNECTOR_KEY!r} (HTTP Webhook wait-for-event)" ) print("OK: HTTP Webhook wait-for-event receiveTask present") @@ -198,7 +205,7 @@ def main() -> None: http_nodes = http_get_nodes(root) if not http_nodes: fail( - f"no bpmn:sendTask carrying {HTTP_TYPE} configured as a manual GET to the " + f"no bpmn:sendTask carrying {' or '.join(HTTP_TYPES)} configured as a manual GET to the " "webhook URL (mode=manual, method=GET, url containing 'webhook', " "no populated headers/parameters context field)" ) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py index c03b0b09b2..760433d730 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py @@ -165,7 +165,7 @@ def test_find_bpmn_file_without_hint_prefers_the_project_file(tmp_path, monkeypa project.mkdir() (project / "Proj.bpmn").write_text("", encoding="utf-8") (project / "project.uiproj").write_text("{}", encoding="utf-8") - (tmp_path / "draft.bpmn").write_text("", encoding="utf-8") + (tmp_path / "draft.bpmn").write_text("", encoding="utf-8") monkeypatch.chdir(tmp_path) assert bpmn_check.find_bpmn_file().endswith("Proj/Proj.bpmn") @@ -373,3 +373,35 @@ def _no_cli(*args, **kwargs): monkeypatch.setattr(validate_bpmn.subprocess, "run", _no_cli) assert validate_bpmn.main([]) == 1 + + +def test_find_bpmn_file_without_hint_accepts_identical_copies(tmp_path, monkeypatch) -> None: + """Two byte-identical .bpmn files, both beside a project.uiproj: one + artifact, not ambiguity (the agent copied its scaffold into the solution + wrapper on CI run 35538279757). Differing content still fails.""" + for d in ("Proj", "ProjSolution/Proj"): + (tmp_path / d).mkdir(parents=True) + (tmp_path / d / "Proj.bpmn").write_text("", encoding="utf-8") + (tmp_path / d / "project.uiproj").write_text("{}", encoding="utf-8") + monkeypatch.chdir(tmp_path) + + assert bpmn_check.find_bpmn_file().endswith("Proj.bpmn") + + (tmp_path / "Proj" / "Proj.bpmn").write_text("", encoding="utf-8") + with pytest.raises(SystemExit): + bpmn_check.find_bpmn_file() + + +def test_resolve_project_excludes_the_live_run_copy(tmp_path, monkeypatch) -> None: + """A live grader's ephemeral solution holds an imported copy of the + project; ``exclude_under`` keeps it out of the candidate set.""" + for d in ("ProjSolution/Proj", "proj-live/ProjLiveEval/Proj"): + (tmp_path / d).mkdir(parents=True) + (tmp_path / d / "Proj.bpmn").write_text("", encoding="utf-8") + (tmp_path / d / "project.uiproj").write_text("{}", encoding="utf-8") + monkeypatch.chdir(tmp_path) + + with pytest.raises(SystemExit): + bpmn_check.resolve_project("Proj.bpmn") + resolved = bpmn_check.resolve_project("Proj.bpmn", exclude_under=[Path("proj-live")]) + assert resolved == tmp_path / "ProjSolution" / "Proj" From a7dc3fc2947fa65083d76f8e7c7aa0e9a2abb85c Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Mon, 21 Sep 2026 17:46:46 -0700 Subject: [PATCH 09/35] test(bpmn): fail cleanly when the Jira seed file is missing The two remaining Jira live graders raised FileNotFoundError when pre_run's seed did not run; jira_get_issue already reports it as a FAIL line. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_escalation_jira_ticket.py | 5 ++++- .../uipath-maestro-bpmn/_shared/check_jira_lifecycle.py | 5 ++++- 2 files changed, 8 insertions(+), 2 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py index 2a0be82903..df30859435 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py @@ -259,7 +259,10 @@ def _recover_partial_keys(project_key: str, correlation: str, raw_text: str) -> def main() -> None: - seed = json.loads(Path("seed.json").read_text(encoding="utf-8")) + seed_path = Path("seed.json") + if not seed_path.is_file(): + _fail("seed.json is missing from the sandbox (pre_run seed did not run)") + seed = json.loads(seed_path.read_text(encoding="utf-8")) correlation = seed["correlationId"] project_key = seed["project_key"] diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py index 1e9aa2405d..365ac7bc92 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py @@ -199,7 +199,10 @@ def _record_key(key: str) -> None: def main() -> None: - seed = json.loads(Path("seed.json").read_text()) + seed_path = Path("seed.json") + if not seed_path.is_file(): + _fail("seed.json is missing from the sandbox (pre_run seed did not run)") + seed = json.loads(seed_path.read_text(encoding="utf-8")) issues = seed["issues"] project = seed["project_key"] issuetype_id = seed["issuetype_id"] From 01430491f13dd01e9f15cec6aaa52160a8fdb2dd Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Mon, 21 Sep 2026 19:16:50 -0700 Subject: [PATCH 10/35] docs(bpmn): record smoke_query as a sort-field surface gap in the parity ledger Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md | 2 +- 1 file changed, 1 insertion(+), 1 deletion(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index 8a1009a436..a7d1cf2580 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -19,7 +19,7 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | `connector_features/datafabric_connector/smoke_create_all_types.yaml` | `…/datafabric_connector/smoke_create_all_types/` | PASS (run 35489744689, iteration 2) | grader widened to generic entity-CRUD form | | `connector_features/datafabric_connector/integration_create_get.yaml` | `…/datafabric_connector/integration_create_get/` | PASS (run 35489744689, iteration 2) | same | | `connector_features/datafabric_connector/contractregistry_crud_filters.yaml` | `…/datafabric_connector/contractregistry_crud_filters/` | PASS (run 35489744689, iteration 2) | same + transitive output mapping | -| `connector_features/datafabric_connector/smoke_query.yaml` | `…/datafabric_connector/smoke_query/` | PASS (run 35490499651, iteration 3; run 35490198577 ERRORed on a tenant ping timeout) | sort in ORDER BY clause | +| `connector_features/datafabric_connector/smoke_query.yaml` | `…/datafabric_connector/smoke_query/` | SKIPPED (surface gap) after PASS it.3 (run 35490499651) and two smoke-gate fails (run 35674400362 attempts 1+2) | the curated Query Entity Records template has no sort-field parameter (only `isAscending`); Flow carried `_sortFieldName`. The passing runs used an invented `sortField` input or CEQL `ORDER BY`; without a sanctioned carrier the task is `skip: true`, criteria unchanged. | | `connector_features/datafabric_connector/smoke_update.yaml` | `…/datafabric_connector/smoke_update/` | PASS (run 35499789502) | batch 2 | | `connector_features/datafabric_connector/smoke_file_activities.yaml` | `…/datafabric_connector/smoke_file_activities/` | PASS (run 35499789502) | batch 2 | | `connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml` | `…/datafabric_connector/e2e_contract_intake_pipeline/` | PASS (run 35499789502) | batch 2 | From e4da5e5c7ff517ed8fd18f907a65798995a552f7 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 14:18:04 -0700 Subject: [PATCH 11/35] test(bpmn): accept UnifiedHttpRequest in the live HTTP graders; record batch 11 registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask and the eval agent emits either, so the weather, pipeline, fallback and billing graders classify by both. Ledger and handoff record run 35783045540: four green, one infra 504, one harness stop, three agent failures at Flow-level flakiness, billing_discrepancy_detector parked. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md | 8 ++++++++ .../uipath-maestro-bpmn/_porting/parity-ledger.md | 10 ++++++++++ .../_shared/check_billing_invoice_lookup.py | 2 +- .../_shared/check_slack_http_fallback.py | 2 +- .../_shared/check_slack_weather_pipeline.py | 8 +++++--- .../uipath-maestro-bpmn/_shared/check_weather_bpmn.py | 6 ++++-- .../_shared/check_weather_bpmn_simulated.py | 6 ++++-- 7 files changed, 33 insertions(+), 9 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index a196d934d0..1cc8357e40 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -51,6 +51,14 @@ Two real failures, one iteration spent each: billing_discrepancy_detector (Integ Next dispatch, one batch: the four grader-fixed tasks for confirmation, the two real failures (iteration 2), the three interactive simulated ports and `ceql_where` (first run). Ten tasks. +## Batch 11 (after the structural PR merged) + +Branch rebased onto main (PR #3426 squashed as 345c1da8a). Run 35783045540, ten tasks. Green: slack_http_fallback, webhook_waitfor_parallel, testmanager_crud_grounded (the three batch-10 grader fixes confirmed) and cli_dice_roller_simulated (first run). billing_invoice_lookup hit a platform 504 during polling (infra, rerun). bellevue_weather_simulated's simulation stopped on turn 1 with empty agent output (harness, rerun). Real failures: billing_discrepancy_detector parked after two different malformed Data Service where clauses (skill gap: where clause from a process variable); slack_channel_description_simulated (channel_not_found), slack_weather_pipeline (wrong Slack connection bound, 401 + MISSING_BINDING) and ceql_where (CEQL string instead of the canonical tree) get one more iteration. Flow's own nightlies for these three pass 4/12, 6/12 and 11/12, so the BPMN flakiness is at parity except ceql_where. + +Grader tolerance added this round: every live grader that classified managed HTTP by `Intsvc.HttpExecution` now also accepts `Intsvc.UnifiedHttpRequest` (both listed in registry-workflow.md). Also landed on main via #3476: `bpmn_check.body_object()` reads sendTask bodies in both registry forms (one JSON blob, or one typed input per field); any new grader that reads a body must use it. + +Batch 12 = billing_invoice_lookup (rerun), bellevue_weather_simulated (rerun), slack_channel_description_simulated (it.2), slack_weather_pipeline (it.3), ceql_where (it.2). + ## Probe bucket (16): pilot ported, 11 decided, 4 blocked `connector_features/ceql_where` is ported (commit fd6312fde), not yet run. The probe confirmed the filter carrier exists: `Intsvc.ActivityExecution` enrichment for the Entra `groups` List operation exposes a `where` parameter (type `query`, `FilterBuilder`, `hasCEQL: true`), and the CI-passing Data Fabric artifact carries the same tree as a `target="query" name="queryExpression" type="json"` input. As in Flow, the sandbox has no live tenant for enrichment, so the port grades the same standalone `where_detail.json` planning artifact plus the connector node and terminate end. No `bpmn validate` gate, matching Flow. Its one review flag: the groups-operation tolerance (objectName contains `group`, or `groups` + GET/list) has no CI-passed fixture yet. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index a7d1cf2580..c2e05d3523 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -57,6 +57,16 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | `connector_trigger/webhook_waitfor_parallel.yaml` | `connector_trigger/webhook_waitfor_parallel/` | 0.47 it.1 (run 35538279757); grader fixed, not re-run | agent emitted the wait as `bpmn:intermediateCatchEvent` + `Intsvc.WaitForEvent` (validates) and the GET as `Intsvc.UnifiedHttpRequest` (the skill lists it beside HttpExecution). Grader now classifies by wrapper type and accepts both HTTP types; replays green. Advisory `uip is webhooks config` telemetry 0, as feared. | | `multi_node/billing_discrepancy_detector/…` | `multi_node/billing_discrepancy_detector/` | FAIL 0.30 it.1 (run 35538279757) | validate passed; live debug raised an Integration Services 400 on `Task_QueryERP`: "Expected a field name expression but got 'StringValue'" (malformed Data Service query filter, agent authoring). Advisory: `accountTier` did not derive from the CRM query. Also hit the same `bindings` grader defect (fixed). One iteration left to spend when live work resumes. | | `multi_node/slack_weather_pipeline/…` | `multi_node/slack_weather_pipeline/` | FAIL 0.375 it.1 (run 35538279757) | validate passed; live debug incident 300501 in script task `Task_SelectChannel`: "Slack channel office-bellevue was not found" — the agent's channel lookup did not find a channel Flow's port finds (likely list pagination/limit). Agent defect, not grader; retry when live work resumes. | +| `connector_features/slack-http-fallback/…` (batch 11) | `connector_features/slack_http_fallback/` | PASS 1.0 (run 35783045540) | confirmed after the emoji_list tolerance | +| `connector_trigger/webhook_waitfor_parallel.yaml` (batch 11) | `connector_trigger/webhook_waitfor_parallel/` | PASS 1.0 (run 35783045540) | confirmed after wrapper-type classification | +| `connector_features/testmanager_crud_grounded/…` (batch 11) | `connector_features/testmanager_crud_grounded/` | PASS 1.0 (run 35783045540) | confirmed after identical-copy tolerance | +| `interactive/cli_dice_roller_simulated/…` | `interactive/cli_dice_roller_simulated/` | PASS 1.0 first run (run 35783045540) | live, simulated user | +| `multi_node/billing_invoice_lookup/…` (batch 11) | `multi_node/billing_invoice_lookup/` | INFRA (run 35783045540): platform 504 on poll-instance-status; not counted | rerun in batch 12 | +| `multi_node/billing_discrepancy_detector/…` (it.2) | `multi_node/billing_discrepancy_detector/` | PARKED (skill gap) after it.2 (run 35783045540) | it.1 400 "Expected a field name expression but got 'StringValue'"; it.2 400 "Error parsing query: SELECT * FROM DUMMY WHERE `accountNumber`=" (variable interpolated as empty). Both: the skill does not teach how to build a Data Service where clause from a process variable. Flow passes 11/12 nightlies. | +| `connector_features/ceql_where.yaml` (it.1) | `connector_features/ceql_where/` | FAIL 0 it.1 (run 35783045540) | agent wrote the CEQL string `displayName='active'` plus a flat {field, operator, value} object, not the canonical tree (numeric groupOperator + filters[]) the prompt asks for. Iteration 2 in batch 12; if repeated, park: BPMN's sanctioned `where` carrier is a CEQL string. | +| `interactive/bellevue_weather_simulated/…` (it.1) | `interactive/bellevue_weather_simulated/` | HARNESS (run 35783045540): simulation stopped on turn 1 with an empty agent output (stop_token), no HTTP node built | rerun in batch 12; Flow's version passes 8/12 | +| `interactive/slack_channel_description_simulated/…` (it.1) | `interactive/slack_channel_description_simulated/` | FAIL 0.52 it.1 (run 35783045540) | runtime 400 channel_not_found on Get_Channel_Info (agent passed a channel the bot is not in). Flow's version fully passes 4/12 nightlies. Iteration 2 in batch 12. | +| `multi_node/slack_weather_pipeline/…` (it.2) | `multi_node/slack_weather_pipeline/` | FAIL 0 it.2 (run 35783045540) | validate MISSING_BINDING on the Slack node + runtime 401 "Invalid Organization or User secret" (wrong connection bound). it.1 was channel-not-found. Flow's version passes 6/12 nightlies. Iteration 3 in batch 12, then park. | | `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | ## Ported 1:1 (21) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py index a2d0c8f0eb..6dbad06022 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py @@ -694,7 +694,7 @@ def advisory() -> None: http_nodes = [ task.attrib.get("id") for task in (*elements(root, "sendTask"), *elements(root, "serviceTask")) - if has_typed_uipath_extension(task, "activity", "Intsvc.HttpExecution") + if any(has_typed_uipath_extension(task, "activity", t) for t in ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest")) ] if http_nodes: fail(f"the process calls Data Service over raw HTTP ({http_nodes}); use the connector action") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py index c3fbd94087..5790eb4046 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py @@ -123,7 +123,7 @@ # naming (CI run 35538279757: the agent's node ran to completion against it). EMOJI_ENDPOINT = "emoji.list" EMOJI_ENDPOINT_RE = re.compile(r"emoji[._]list") -ACTIVITY_TYPES = ("Intsvc.ActivityExecution", "Intsvc.HttpExecution") +ACTIVITY_TYPES = ("Intsvc.ActivityExecution", "Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") LIVE_RUN_DIR = Path("slack-emoji-list-live") SOLUTION_INIT_TIMEOUT = 90 diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py index 71c9f6915d..8ac0444916 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py @@ -111,7 +111,9 @@ SLACK_CONNECTOR_KEY = "uipath-salesforce-slack" ACTIVITY_TYPE = "Intsvc.ActivityExecution" -HTTP_TYPE = "Intsvc.HttpExecution" +# registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask; the eval agent emits either (CI run 35538279757). +HTTP_TYPES = ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") +HTTP_TYPE = HTTP_TYPES[0] WEATHER_HINTS = ("open-meteo", "openmeteoapis") NAME_HINT = "SlackWeatherPipeline" @@ -204,7 +206,7 @@ def find_weather_node(root: ET.Element) -> list[ET.Element]: if activity is None: continue raw = ET.tostring(activity, encoding="unicode") - if HTTP_TYPE not in raw and ACTIVITY_TYPE not in raw: + if not any(t in raw for t in HTTP_TYPES) and ACTIVITY_TYPE not in raw: continue if any(hint in raw.lower() for hint in WEATHER_HINTS): found.append(node) @@ -247,7 +249,7 @@ def main() -> None: weather_nodes = find_weather_node(root) if not weather_nodes: _fail( - f"bpmn does not reference an API-capable node ({HTTP_TYPE} or " + f"bpmn does not reference an API-capable node ({' / '.join(HTTP_TYPES)} or " f"{ACTIVITY_TYPE}) targeting one of {WEATHER_HINTS}" ) print(f"OK: bpmn references an API node targeting open-meteo") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py index 1a1f40d8cf..2dd84e733c 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py @@ -88,7 +88,9 @@ ) BPMN_NS = "http://www.omg.org/spec/BPMN/20100524/MODEL" -ACTIVITY_TYPE = "Intsvc.HttpExecution" +# registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask; the eval agent emits either (CI run 35538279757). +ACTIVITY_TYPES = ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") +ACTIVITY_TYPE = ACTIVITY_TYPES[0] NAME_HINT = "BellevueWeather" VERDICTS = ("nice day", "bring a jacket") @@ -146,7 +148,7 @@ def find_http_execution_nodes(root: ET.Element) -> list[ET.Element]: el for el in root.iter() if el.tag != f"{{{BPMN_NS}}}extensionElements" - and has_typed_uipath_extension(el, "activity", ACTIVITY_TYPE) + and any(has_typed_uipath_extension(el, "activity", t) for t in ACTIVITY_TYPES) ] diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py index a7739cefa1..7e02569e8e 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py @@ -108,7 +108,9 @@ ) BPMN_NS = "http://www.omg.org/spec/BPMN/20100524/MODEL" -ACTIVITY_TYPE = "Intsvc.HttpExecution" +# registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask; the eval agent emits either (CI run 35538279757). +ACTIVITY_TYPES = ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") +ACTIVITY_TYPE = ACTIVITY_TYPES[0] VERDICTS = ("nice day", "bring a jacket") LIVE_RUN_DIR = Path("bellevue-weather-simulated-live") @@ -166,7 +168,7 @@ def find_http_execution_nodes(root: ET.Element) -> list[ET.Element]: el for el in root.iter() if el.tag != f"{{{BPMN_NS}}}extensionElements" - and has_typed_uipath_extension(el, "activity", ACTIVITY_TYPE) + and any(has_typed_uipath_extension(el, "activity", t) for t in ACTIVITY_TYPES) ] From 5893e6c3cded9227d5532b9e3d2e162cb9876fff Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 14:37:05 -0700 Subject: [PATCH 12/35] test(bpmn): skip untyped drafts when locating the graded process; record batch 12 The agent left a bpmn-init draft under Solution/ beside its real solution (run 35785806030, slack_channel_description_simulated); both had project.uiproj. find_bpmn_file now drops candidates that carry no registry-typed node before declaring ambiguity. Ledger and handoff record batch 12: two green, two parked with runtime evidence, one grader defect fixed. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/LIVE-HANDOFF.md | 6 ++++++ .../_porting/parity-ledger.md | 5 +++++ .../uipath-maestro-bpmn/_shared/bpmn_check.py | 15 +++++++++++++++ .../_shared/test_bpmn_check.py | 18 ++++++++++++++++++ 4 files changed, 44 insertions(+) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index 1cc8357e40..cc77d10c8b 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -59,6 +59,12 @@ Grader tolerance added this round: every live grader that classified managed HTT Batch 12 = billing_invoice_lookup (rerun), bellevue_weather_simulated (rerun), slack_channel_description_simulated (it.2), slack_weather_pipeline (it.3), ceql_where (it.2). +## Batch 12 + +Run 35785806030, five tasks. Green: billing_invoice_lookup, slack_weather_pipeline (iteration 3). Parked: bellevue_weather_simulated (same `temperature_2m` response-shape fault as its non-simulated twin) and ceql_where (the agent writes the connector's sanctioned CEQL `where` string, never Flow's numeric-groupOperator tree; the tree is a Flow-skill construct). slack_channel_description_simulated failed on a grader defect: an abandoned untyped draft under `Solution/` beside the real solution; `find_bpmn_file` now drops candidates with no registry-typed node. Iteration 3 goes in batch 13 with the field-shape ports. + +Field-shape family decision: assertions on wire parameters (path, query, pagination, enum, multiselect, complex_array) port; assertions on a Flow filter-tree shape do not (ceql_where parked; enhanced_enum and searchable_joins must be checked for the same trap before porting). + ## Probe bucket (16): pilot ported, 11 decided, 4 blocked `connector_features/ceql_where` is ported (commit fd6312fde), not yet run. The probe confirmed the filter carrier exists: `Intsvc.ActivityExecution` enrichment for the Entra `groups` List operation exposes a `where` parameter (type `query`, `FilterBuilder`, `hasCEQL: true`), and the CI-passing Data Fabric artifact carries the same tree as a `target="query" name="queryExpression" type="json"` input. As in Flow, the sandbox has no live tenant for enrichment, so the port grades the same standalone `where_detail.json` planning artifact plus the connector node and terminate end. No `bpmn validate` gate, matching Flow. Its one review flag: the groups-operation tolerance (objectName contains `group`, or `groups` + GET/list) has no CI-passed fixture yet. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index c2e05d3523..bbc2602e4d 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -67,6 +67,11 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | `interactive/bellevue_weather_simulated/…` (it.1) | `interactive/bellevue_weather_simulated/` | HARNESS (run 35783045540): simulation stopped on turn 1 with an empty agent output (stop_token), no HTTP node built | rerun in batch 12; Flow's version passes 8/12 | | `interactive/slack_channel_description_simulated/…` (it.1) | `interactive/slack_channel_description_simulated/` | FAIL 0.52 it.1 (run 35783045540) | runtime 400 channel_not_found on Get_Channel_Info (agent passed a channel the bot is not in). Flow's version fully passes 4/12 nightlies. Iteration 2 in batch 12. | | `multi_node/slack_weather_pipeline/…` (it.2) | `multi_node/slack_weather_pipeline/` | FAIL 0 it.2 (run 35783045540) | validate MISSING_BINDING on the Slack node + runtime 401 "Invalid Organization or User secret" (wrong connection bound). it.1 was channel-not-found. Flow's version passes 6/12 nightlies. Iteration 3 in batch 12, then park. | +| `multi_node/billing_invoice_lookup/…` (batch 12) | `multi_node/billing_invoice_lookup/` | PASS 1.0 (run 35785806030) | live; the batch-11 504 was infra | +| `multi_node/slack_weather_pipeline/…` (it.3) | `multi_node/slack_weather_pipeline/` | PASS 1.0 it.3 (run 35785806030) | live; it.1 channel lookup, it.2 wrong connection — agent flakiness at Flow parity (Flow 6/12) | +| `interactive/bellevue_weather_simulated/…` (it.2) | `interactive/bellevue_weather_simulated/` | PARKED (skill gap) after it.2 (run 35785806030) | runtime "Cannot read property 'temperature_2m' of undefined" in the summarize script: same HttpExecution response-shape gap that parked multi_node/bellevue_weather. | +| `connector_features/ceql_where.yaml` (it.2) | `connector_features/ceql_where/` | PARKED (surface gap) after it.2 (run 35785806030) | both runs: agent writes the CEQL string `displayName='active'` (the connector's sanctioned `where` parameter) plus a flat object, never the numeric-groupOperator tree Flow's grader requires. The tree is a Flow-skill construct with no BPMN carrier. Verdict for the family: tree-shape assertions do not port; wire-parameter assertions (path, query, pagination, enum, multiselect, complex_array) do. | +| `interactive/slack_channel_description_simulated/…` (it.2) | `interactive/slack_channel_description_simulated/` | GRADER DEFECT it.2 (run 35785806030); fixed, it.3 in batch 13 | agent left an untyped draft under `Solution/` beside the real solution; `find_bpmn_file` without a hint saw two projects. It now drops candidates carrying no registry-typed node; replays to the real file. | | `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | ## Ported 1:1 (21) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py index bc3e8d488a..922cb9b50d 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py @@ -58,9 +58,24 @@ def find_bpmn_file(name_hint: str | None = None) -> str: # wrapper; CI run 35538279757, testmanager_crud_grounded) are one artifact. if len({_sha256(p) for p in paths}) == 1: return paths[0] + # An abandoned draft beside the real process: `bpmn init` leaves a + # solution wrapper whose process carries no registry-typed node, while the + # deliverable does (CI run 35785806030, slack_channel_description_simulated). + # Only a candidate with at least one `` is a process + # the task could be graded on. + typed = [p for p in projects if _has_typed_node(p)] + if len(typed) == 1: + return typed[0] fail(f"multiple BPMN files found; expected one or hint match: {paths}") +def _has_typed_node(path: str | Path) -> bool: + try: + return "uipath:type value=" in Path(path).read_text(encoding="utf-8", errors="replace") + except OSError: + return False + + def _sha256(path: str | Path) -> str: import hashlib diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py index 760433d730..39992e7b54 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py @@ -405,3 +405,21 @@ def test_resolve_project_excludes_the_live_run_copy(tmp_path, monkeypatch) -> No bpmn_check.resolve_project("Proj.bpmn") resolved = bpmn_check.resolve_project("Proj.bpmn", exclude_under=[Path("proj-live")]) assert resolved == tmp_path / "ProjSolution" / "Proj" + + +def test_find_bpmn_file_without_hint_skips_an_untyped_draft(tmp_path, monkeypatch) -> None: + """Two different .bpmn files, both beside a project.uiproj: the one with a + registry-typed node is the deliverable; the other is an abandoned draft + (CI run 35785806030). Two typed candidates stay ambiguous.""" + typed = '' + for d, body in (("Real/Proj", typed), ("ProjSolution/Proj", "")): + (tmp_path / d).mkdir(parents=True) + (tmp_path / d / "Proj.bpmn").write_text(body, encoding="utf-8") + (tmp_path / d / "project.uiproj").write_text("{}", encoding="utf-8") + monkeypatch.chdir(tmp_path) + + assert bpmn_check.find_bpmn_file().endswith("Real/Proj/Proj.bpmn") + + (tmp_path / "ProjSolution" / "Proj" / "Proj.bpmn").write_text(typed.replace("Intsvc", "BPMN"), encoding="utf-8") + with pytest.raises(SystemExit): + bpmn_check.find_bpmn_file() From 8bece67ed5ff1998ad22f8f56cddcf45c0ead375 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 14:52:45 -0700 Subject: [PATCH 13/35] test(bpmn): port the Flow path_params, query_params and enum evals Faithful ports of three Integration Service field-shape evals: same criteria, weights and run limits. A shared check_managed_http_fallback grader accepts the native connector node or the managed HTTP fallback (HttpExecution, UnifiedHttpRequest, or the generic uipath-uipath-http connector), mirroring Flow's acceptance; the Jira path parameter and the Gmail enum body are read from inputs in either registry body form. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_enum_flow.py | 197 ++++++++++++++++++ .../_shared/check_managed_http_fallback.py | 192 +++++++++++++++++ .../_shared/check_path_param_value.py | 161 ++++++++++++++ .../connector_features/enum/enum.yaml | 109 ++++++++++ .../path_params/path_params.yaml | 109 ++++++++++ .../query_params/query_params.yaml | 85 ++++++++ 6 files changed, 853 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py new file mode 100644 index 0000000000..7bb89cc944 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py @@ -0,0 +1,197 @@ +#!/usr/bin/env python3 +"""Validate the EnumTest BPMN process: structure and enum body-value wiring. + +Ported from Flow `connector_features/check_enum_flow.py`: only the +`structure` and `body_params` modes are ported (enum.yaml's own criteria +never call Flow's `control` mode -- that mode grades enhanced_enum's +Decision/Terminate control-flow shape, not enum.yaml's), translated from a +JSON `inputs.detail.bodyParameters` walk to an XML walk over the +registry-driven `Intsvc.*` connector shell (see skills/uipath-maestro-bpmn/ +references/registry-workflow.md §3 "Body shape"). + +registry-workflow.md documents the canonical hand-authored body shape as +exactly ONE `target="body"` input holding the whole request as a JSON CDATA +blob. The CLI manifest's stale `InputNotes`, however, still tell an author to +add one `uipath:input` per request field, and BATCH1-ADDENDUM's own lessons +record agents emitting both shapes for other Intsvc.ActivityExecution +parameters (curated separate path/query inputs vs. one JSON blob). Per this +task's instructions, both body forms are accepted here too: one JSON blob +input, or one typed `target="body"` input per field. + +Assertion map (Flow -> BPMN): + F check_enum_flow.py:39-44 structure: flow exists, valid JSON, -> check_structure(): locate/parse .bpmn + has nodes/edges (translation of "artifact exists and is + parseable" to the BPMN artifact shape) + F check_enum_flow.py:49-52 _EXPECTED_BODY = {"to": ..., "importance": -> EXPECTED_BODY (kept identical -- + "high"} (only `to` and `importance` are asserted, despite the only `to` and `importance`, matching + docstring's "to/subject/body/importance" -- the code is the source Flow's actual code, not its stale + of truth) docstring) + F check_enum_flow.py:63-72 _body_matches(): case-insensitive field -> body_fields_match(): same + lookup, `.strip().lower()` value comparison case-insensitive compare + F check_enum_flow.py:76-94 _check_body_params(): scan every node's -> check_body_params(): scan every node + inputs.detail.bodyParameters, keep the best (fewest-missing) partial carrying a target="body" input, keep + match for the failure message the best (fewest-missing) partial match + I locate/parse .bpmn -> parse_bpmn(name_hint) + T body read in both forms: one JSON blob, or one typed -> body_fields(): single JSON-parseable + input per field (task instruction; mirrors the CLI manifest's stale target="body" input -> its parsed + separateInputs InputNotes as a real, if non-canonical, shape) object; multiple target="body" inputs, + each name=field -> {name: value} map + T `=`-prefixed expression values pass any type/value check -> body_fields_match() treats a value + (grading-contract-wide translation tolerance for expression strings) starting with "=" as satisfying its + expected field unconditionally + +No Flow assertions dropped: `structure`'s existence/parse check and +`body_params`'s to/importance field match both have a BPMN counterpart above. +Flow's `control` mode is out of scope (enum.yaml's criteria never call it). + +Usage (from a task's run_command, cwd = sandbox root): + python3 $REFERENCE_DIR/_shared/check_enum_flow.py +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import context_inputs, elements, fail, parse_bpmn # noqa: E402 + +# Kept identical to Flow's own _EXPECTED_BODY (check_enum_flow.py:49-52): only +# `to` and `importance` are asserted, matching the code, not its docstring. +EXPECTED_BODY = { + "to": "is-test@uipath.com", + "importance": "high", # this is the enum field under test +} + +# Task-like leaf elements only -- NOT root.iter(), which also yields ancestor +# containers (bpmn:process, bpmn:definitions). An ancestor's serialized +# subtree recursively contains every descendant's inputs, so scanning it too +# would let unrelated target="body" inputs on different nodes merge into one +# fields map. +CANDIDATE_TAGS = ( + "sendTask", + "serviceTask", + "task", + "receiveTask", + "userTask", + "businessRuleTask", + "scriptTask", +) + + +def check_structure(name_hint: str) -> None: + path, root = parse_bpmn(name_hint) + tasks = [ + *elements(root, "sendTask"), + *elements(root, "serviceTask"), + *elements(root, "task"), + ] + print(f"OK: {path} is well-formed BPMN XML with {len(tasks)} task-like node(s)") + + +def body_fields(node: ET.Element) -> dict[str, str] | None: + """The node's request body as a field->value map, in either accepted form. + + Form 1: exactly one `target="body"` input whose text/value parses as a + JSON object -- the canonical single-blob shape. + Form 2: one or more `target="body"` inputs, each named after the field it + carries -- the CLI manifest's stale separateInputs shape. + """ + body_inputs = [inp for inp in context_inputs(node) if inp.attrib.get("target") == "body"] + if not body_inputs: + return None + + if len(body_inputs) == 1: + inp = body_inputs[0] + raw = (inp.text or inp.attrib.get("value") or "").strip() + if raw: + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + parsed = None + if isinstance(parsed, dict): + return {str(k): v for k, v in parsed.items()} + name = inp.attrib.get("name") + if name and name != "body": + return {name: raw} + return None + + fields: dict[str, str] = {} + for inp in body_inputs: + name = inp.attrib.get("name") + if not name: + continue + fields[name] = (inp.attrib.get("value") or inp.text or "").strip() + return fields or None + + +def body_fields_match(fields: dict[str, object]) -> list[str]: + """Missing/mismatched field descriptions; empty list means OK. + + A value beginning with "=" is an expression (`=vars.X`, `=js:...`) and + passes unconditionally -- the grading-contract-wide translation tolerance + for expression strings. + """ + lowered = {str(k).lower(): v for k, v in fields.items()} + missing: list[str] = [] + for key, expected in EXPECTED_BODY.items(): + actual = lowered.get(key.lower()) + if actual is None: + missing.append(f"{key}={expected!r} (missing)") + continue + actual_str = str(actual).strip() + if actual_str.startswith("="): + continue + if actual_str.lower() != expected.lower(): + missing.append(f"{key}={expected!r} (got {actual!r})") + return missing + + +def check_body_params(name_hint: str) -> None: + path, root = parse_bpmn(name_hint) + best_missing: list[str] | None = None + checked_any = False + candidates = [node for tag in CANDIDATE_TAGS for node in elements(root, tag)] + for node in candidates: + fields = body_fields(node) + if fields is None: + continue + checked_any = True + missing = body_fields_match(fields) + if not missing: + node_id = node.attrib.get("id", "") + print(f"OK: body payload on node {node_id!r} carries expected to/importance") + return + if best_missing is None or len(missing) < len(best_missing): + best_missing = missing + + if not checked_any: + fail( + f"No node in {path} has a target=\"body\" input. Hand-authored connector " + f"nodes must carry either one JSON target=\"body\" input holding the " + f"whole request object, or one target=\"body\" input per field." + ) + fail(f"target=\"body\" input found but missing or wrong fields: {best_missing}. BPMN: {path}") + + +CHECKS = { + "structure": check_structure, + "body_params": check_body_params, +} + + +def main() -> None: + if len(sys.argv) != 3: + fail(f"usage: check_enum_flow.py <{'|'.join(CHECKS)}>") + name_hint, check_name = sys.argv[1], sys.argv[2] + check = CHECKS.get(check_name) + if check is None: + fail(f"unknown check {check_name!r}; expected one of {sorted(CHECKS)}") + check(name_hint) + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py new file mode 100644 index 0000000000..1d214513ac --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py @@ -0,0 +1,192 @@ +#!/usr/bin/env python3 +"""Native connector use OR task-specific managed HTTP fallback (BPMN). + +Ported from Flow `connector_features/check_managed_http_fallback.py`: same +three scenarios (Jira Get Issue path params, Google Tasks query params, Gmail +Send Mail enum), translated from a whole-flow JSON blob search to an XML walk +over the registry-driven ``Intsvc.*`` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3 "Connector +enrichment" and §"Connectionless vs connector HTTP"). One shared grader +serves all three ports (`path_params.yaml`, `query_params.yaml`, `enum.yaml`) +per _porting/BATCH1-ADDENDUM.md, mirroring Flow's own single shared script. + +Assertion map (Flow -> BPMN): + F check_managed_http_fallback.py:110 "connector_key in full_text" (native -> whole-document blob search for + connector present, whole-flow blob search) connector_key (main(), same leniency) + F check_managed_http_fallback.py:57-66 _check_path_params required -> FALLBACK_CHECKS["path_params"]["required"] + evidence ["engce-00000","method","get"] checked against each candidate node's blob + F check_managed_http_fallback.py:69-81 _check_query_params required -> FALLBACK_CHECKS["query_params"]["required"] + evidence [...] + F check_managed_http_fallback.py:84-95 _check_enum required evidence -> FALLBACK_CHECKS["enum"]["required"] + ["gmail.googleapis.com","method","post","importance","medium"] (Flow's (kept verbatim, including "medium" -- + own literal list, kept as-is even though the scenario asks for "high": see GUESS below) + faithfulness contract forbids "fixing" a Flow assertion during a port) + F check_managed_http_fallback.py:32-40 http-fallback node selection -> is_managed_http_node(): Intsvc.HttpExecution + (flow node type == core.action.http.v2) or Intsvc.UnifiedHttpRequest (construct- + translation table's "Managed HTTP" row) + I locate/parse .bpmn -> parse_bpmn(NAME_HINT) + T curated OR generic connectorKey match for the native -> is_native_connector_node(): matches on + branch (BATCH1-ADDENDUM: classify by connectorKey, not connectorKey alone, regardless of + by objectName) curated/generic objectName spelling + T generic HTTP connector (uipath-uipath-http) accepted as -> is_managed_http_node() also accepts + an additional managed-HTTP wrapper form: query_params.yaml Intsvc.ActivityExecution whose + prompt explicitly asks for "a managed HTTP fallback through connectorKey == uipath-uipath-http, + the generic uipath-uipath-http connector" -- the connector- the same shape check_non_catalog_http_ + mode-HTTP-via-generic-connector pattern documented in fallback.py (batch 1 pilot) accepts for + BATCH1-ADDENDUM and modeled by check_non_catalog_http_ an unrelated non-catalog service. This + fallback.py, which is Intsvc.ActivityExecution, not widens acceptance (never narrows a Flow + Intsvc.HttpExecution assertion), so it is safe under the + Normalization pass's "dropping/widening + only" rule. + T collect uipath:input elements at any depth under the node -> context_inputs()/all_node_values() via + node_blob() + +GUESS (flag for reviewer): Flow's `_check_enum` requires the literal substring +"medium" in the HTTP fallback node's blob even though the enum.yaml scenario's +fixed value is "importance": "high". This reads like a latent bug or an +intentional check that the fallback node's schema/enum documentation (which +may enumerate "low, medium, high") is present, not that the request body VALUE +is "medium". The faithfulness contract's "Never" column forbids weakening a +criterion to make it pass, and there is no directive here to fix a suspected +Flow defect during a port, so this checker keeps "medium" as a required +substring verbatim. Report this upstream if the intent was "importance" + +"high". + +No Flow assertions dropped: the native-connector short-circuit, all three +required-evidence lists, and the http-fallback-node-must-exist precondition +all have a BPMN counterpart above. + +Usage (from a task's run_command, cwd = sandbox root): + python3 $REFERENCE_DIR/_shared/check_managed_http_fallback.py +""" + +from __future__ import annotations + +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import context_value, elements, fail, has_type, parse_bpmn # noqa: E402 + +# Task-like leaf elements only -- NOT root.iter(), which also yields ancestor +# containers (bpmn:process, bpmn:definitions). Those ancestors' serialized +# subtree recursively contains every descendant's inputs, so has_type()/ +# context_value() would misclassify them as matching "nodes" too and collapse +# the per-node evidence check into a whole-document one (see +# check_slack_http_fallback.py's is_slack_connector_node docstring for the +# same root.iter() ancestor-matching pitfall). +CANDIDATE_TAGS = ( + "sendTask", + "serviceTask", + "task", + "receiveTask", + "userTask", + "businessRuleTask", + "scriptTask", +) + +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +HTTP_TYPES = ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") +# The generic HTTP connector -- query_params.yaml's prompt explicitly names it +# as the fallback path (BATCH1-ADDENDUM's connector-mode-HTTP-via-generic- +# connector pattern, same shape as check_non_catalog_http_fallback.py). +GENERIC_HTTP_CONNECTOR_KEY = "uipath-uipath-http" + +# Required evidence per check, kept verbatim from Flow's own literal lists +# (see GUESS above re: "enum"'s "medium"). +FALLBACK_CHECKS: dict[str, dict[str, object]] = { + "path_params": { + "label": "Jira Get Issue path-params HTTP fallback", + "required": ["engce-00000", "method", "get"], + }, + "query_params": { + "label": "Google Tasks query-params HTTP fallback", + "required": [ + "tasks.googleapis.com/tasks/v1/lists", + "/tasks", + "method", + "get", + "showhidden", + "true", + ], + }, + "enum": { + "label": "Gmail send-mail enum HTTP fallback", + "required": ["gmail.googleapis.com", "method", "post", "importance", "medium"], + }, +} + + +def is_native_connector_node(node: ET.Element, connector_key: str) -> bool: + if not has_type(node, ACTIVITY_TYPE): + return False + return context_value(node, "connectorKey").strip().lower() == connector_key.lower() + + +def is_managed_http_node(node: ET.Element) -> bool: + if any(has_type(node, token) for token in HTTP_TYPES): + return True + if has_type(node, ACTIVITY_TYPE): + return context_value(node, "connectorKey").strip().lower() == GENERIC_HTTP_CONNECTOR_KEY + return False + + +def node_blob(node: ET.Element) -> str: + """Whole serialized node, lowercased -- mirrors Flow's + ``json.dumps(http_nodes, sort_keys=True).lower()`` whole-node blob + search.""" + return ET.tostring(node, encoding="unicode").lower() + + +def require_all(haystack: str, needles: list[str], label: str) -> list[str]: + return [needle for needle in needles if needle.lower() not in haystack] + + +def main() -> None: + if len(sys.argv) != 4: + fail( + "usage: check_managed_http_fallback.py " + f"<{'|'.join(FALLBACK_CHECKS)}>" + ) + + name_hint, connector_key, check_name = sys.argv[1], sys.argv[2], sys.argv[3] + check = FALLBACK_CHECKS.get(check_name) + if check is None: + fail(f"unknown check {check_name!r}; expected one of {sorted(FALLBACK_CHECKS)}") + + path, root = parse_bpmn(name_hint) + + # Native connector branch -- mirrors Flow's whole-document blob search + # (main():110) exactly, including its leniency (a connectorKey mention + # anywhere in the document is accepted, not only on a matched node). + full_blob = ET.tostring(root, encoding="unicode").lower() + if connector_key.lower() in full_blob: + print(f"OK: native connector present ({connector_key}) in {path}") + return + + candidates = [node for tag in CANDIDATE_TAGS for node in elements(root, tag)] + fallback_nodes = [node for node in candidates if is_managed_http_node(node)] + if not fallback_nodes: + fail( + "Neither native connector nor managed HTTP fallback found " + f"(connector_key={connector_key!r}, check={check_name!r}) in {path}" + ) + + label = str(check["label"]) + required = list(check["required"]) # type: ignore[arg-type] + best_missing: list[str] | None = None + for node in fallback_nodes: + missing = require_all(node_blob(node), required, label) + if not missing: + print(f"OK: managed HTTP fallback has {check_name} evidence in {path}") + return + if best_missing is None or len(missing) < len(best_missing): + best_missing = missing + + fail(f"{label} missing expected evidence: {best_missing}") + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py new file mode 100644 index 0000000000..3ffe0b5012 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py @@ -0,0 +1,161 @@ +#!/usr/bin/env python3 +"""Verify a deterministic path-parameter value is wired into the BPMN (path_params.yaml). + +Ported from Flow `connector_features/check_path_param_value.py`: same +scenario (the Jira issue key ``ENGCE-00000`` must be wired into a real +path-parameter, not just mentioned in the prompt echo), translated from a +JSON `inputs.detail` walk to an XML walk over the registry-driven +`Intsvc.*` connector shell (see skills/uipath-maestro-bpmn/references/ +registry-workflow.md §3 "Connector enrichment"). + +Flow looked in exactly two places: a node's `pathParameters` dict values +(exact match) or its `url`/`endpoint` string (substring match). BPMN has no +fixed home for a path parameter -- the registry does not pin whether it lands +as a `target="path"` input, a `target="query"` input, inside the single +`target="body"` JSON payload, or embedded in a managed-HTTP node's own +url/path field -- so this checker widens the search to all of those homes +(BATCH1-ADDENDUM "Where connector node values live in BPMN" + +_porting/PORTING-BRIEF.md's `T` translation-tolerance for "entity anywhere in +inputs/objectName/path"). Widening only ever makes an assertion easier to +satisfy, never harder, so this stays within the Normalization pass's +"dropping/widening only" rule. + +Assertion map (Flow -> BPMN): + F check_path_param_value.py:87-91 pathParameters dict value == needle -> path_or_query_input_match(): any + (exact match) target="path"/"query" input whose + value/text contains needle + F check_path_param_value.py:92-95 url/endpoint substring match -> context_field_match(): any "url"/ + "path"/"endpoint" context field + containing needle, checked on every + connector/HTTP node + I locate/parse .bpmn -> parse_bpmn(name_hint) + T needle searched across path/query/body, not only -> body_json_match(): the needle also + pathParameters (widened breadth, kept in scope by the matched inside the single target="body" + calling task's instructions) JSON payload, at any nesting depth + T collect uipath:input elements at any depth under the node -> context_inputs() (bpmn_check) used by + every match function above + +No Flow assertions dropped: both of Flow's two search locations (path-param +values, url/endpoint) have a widened BPMN counterpart above; nothing is +required that Flow did not also accept. + +Usage (from a task's run_command, cwd = sandbox root): + python3 $REFERENCE_DIR/_shared/check_path_param_value.py +""" + +from __future__ import annotations + +import json +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import context_inputs, elements, fail, parse_bpmn # noqa: E402 + +CONTEXT_FIELDS = ("url", "path", "endpoint") + +# Task-like leaf elements only -- NOT root.iter(), which also yields ancestor +# containers (bpmn:process, bpmn:definitions). An ancestor's serialized +# subtree recursively contains every descendant's inputs, so context_inputs() +# would "find" a match on the ancestor first and report an unhelpful +# "" node id instead of the real host element. +CANDIDATE_TAGS = ( + "sendTask", + "serviceTask", + "task", + "receiveTask", + "userTask", + "businessRuleTask", + "scriptTask", +) + + +def _flatten_strings(value: object) -> list[str]: + if isinstance(value, str): + return [value] + if isinstance(value, dict): + out: list[str] = [] + for v in value.values(): + out.extend(_flatten_strings(v)) + return out + if isinstance(value, list): + out = [] + for v in value: + out.extend(_flatten_strings(v)) + return out + return [] + + +def path_or_query_input_match(node: ET.Element, needle: str) -> str | None: + node_id = node.attrib.get("id", "") + for inp in context_inputs(node): + if inp.attrib.get("target") not in ("path", "query"): + continue + value = (inp.attrib.get("value") or inp.text or "") + if needle.lower() in value.lower(): + target = inp.attrib.get("target") + name = inp.attrib.get("name") or "?" + return f"{target} input {name!r} of node {node_id!r}" + return None + + +def context_field_match(node: ET.Element, needle: str) -> str | None: + node_id = node.attrib.get("id", "") + for inp in context_inputs(node): + name = inp.attrib.get("name") or "" + if name.lower() not in CONTEXT_FIELDS: + continue + value = (inp.attrib.get("value") or inp.text or "") + if needle.lower() in value.lower(): + return f"{name} field of node {node_id!r}" + return None + + +def body_json_match(node: ET.Element, needle: str) -> str | None: + node_id = node.attrib.get("id", "") + for inp in context_inputs(node): + if inp.attrib.get("target") != "body": + continue + raw = (inp.text or inp.attrib.get("value") or "").strip() + if not raw: + continue + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + continue + for leaf in _flatten_strings(parsed): + if needle.lower() in leaf.lower(): + return f"body JSON payload of node {node_id!r}" + return None + + +def find_needle(root: ET.Element, needle: str) -> str | None: + candidates = [node for tag in CANDIDATE_TAGS for node in elements(root, tag)] + for node in candidates: + for matcher in (path_or_query_input_match, context_field_match, body_json_match): + location = matcher(node, needle) + if location is not None: + return location + return None + + +def main() -> None: + if len(sys.argv) != 3: + fail("usage: check_path_param_value.py ") + + name_hint, needle = sys.argv[1], sys.argv[2] + path, root = parse_bpmn(name_hint) + + location = find_needle(root, needle) + if location is None: + fail( + f"{needle!r} not found in any node's path/query input, url/path/endpoint " + f"context field, or body JSON payload in {path}" + ) + print(f"OK: {needle!r} found in {location}") + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml new file mode 100644 index 0000000000..c46c76cdaa --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml @@ -0,0 +1,109 @@ +task_id: skill-bpmn-enum +description: > + Tests the enum IS feature — configures a connector node with an enum + importance field on the Gmail "Send Mail" activity. Recipient, subject, and + body are fixed so the test can verify the enum value wiring. + Ported from Flow `connector_features/enum.yaml`; the Gmail Send Mail + connector node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the + request body is graded in either of two accepted shapes — one JSON + target="body" input, or one typed target="body" input per field — since + BPMN's canonical single-blob shape coexists with the CLI manifest's stale + separate-inputs guidance (registry-workflow.md §3 "Body shape"). + Validate-only, matching Flow's own scope — the Flow source never called + `flow debug` either, only `flow validate`. +tags: + - uipath-maestro-bpmn + - integration + - "lifecycle:generate" + - "shape:single-node" + - connector + - uipath-google-gmail + - "mode:build" + +run_limits: + expected_turns: 46 + task_timeout: 3600 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "EnumTest" with a manual start. + Build a process that sends an email via the Gmail "Send Mail" connector + activity. Use these exact values: + - to: "is-test@uipath.com" + - subject: "weather today" + - body: "Weather update for today." + - importance: "high" + Validate the final BPMN file. + Use the connection present in Shared/uipath-maestro-flow. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: command_executed + description: "Advisory: live-v1 agent listed connections before choosing one" + tool_name: "Bash" + command_pattern: "(uip|\\$UIP)\\s+is\\s+connections\\s+list" + min_count: 1 + weight: 1.0 + pass_threshold: 0.0 + + - type: run_command + description: "Tenant has at least one Gmail connection (precondition for connector authoring)" + command: "python3 $REFERENCE_DIR/_setup/preflight_connections.py uipath-google-gmail" + timeout: 35 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN file exists and is well-formed XML" + command: "python3 $REFERENCE_DIR/_shared/check_enum_flow.py EnumTest structure" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN uses Gmail connector or task-specific managed HTTP fallback" + command: "python3 $REFERENCE_DIR/_shared/check_managed_http_fallback.py EnumTest uipath-google-gmail enum" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "Connector node's body carries to and importance" + command: "python3 $REFERENCE_DIR/_shared/check_enum_flow.py EnumTest body_params" + timeout: 10 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml new file mode 100644 index 0000000000..f96e2783ff --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml @@ -0,0 +1,109 @@ +task_id: skill-bpmn-path-params +description: > + Tests the path parameters IS feature — configures a connector node with a + path parameter on the Jira "Get Issue" activity. Project + issue type are + freely chosen by the agent; the issue key is deterministic so the test can + verify the path-parameter wiring. + Ported from Flow `connector_features/path_params.yaml`; the Jira Get Issue + connector node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the + path-parameter value is graded across path/query/body inputs since BPMN has + no fixed home for it (registry-workflow.md does not pin where a path + parameter lands). Validate-only, matching Flow's own scope — the Flow + source never called `flow debug` either, only `flow validate`. +tags: + - uipath-maestro-bpmn + - integration + - "lifecycle:generate" + - "shape:multi-node" + - connector + - uipath-atlassian-jira + - "mode:build" + +run_limits: + expected_turns: 52 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "PathParamsTest" with a manual start. + Build a process that retrieves a single Jira issue using the Jira + connector. Pick any Jira project and any issue type from what is + available on the connection — those values are your choice. The issue key + MUST be "ENGCE-00000". + Validate the final BPMN file. + If the Jira connector is not present in the registry, use a HTTP fallback for the + equivalent REST call and finish after validation. Do not keep searching for a connector + node that the registry does not expose. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: command_executed + description: "Advisory: live-v1 agent listed connections before choosing one" + tool_name: "Bash" + command_pattern: "(uip|\\$UIP)\\s+is\\s+connections\\s+list" + min_count: 1 + weight: 1.0 + pass_threshold: 0.0 + + - type: run_command + description: "Tenant has at least one Jira connection (precondition for connector authoring)" + command: "python3 $REFERENCE_DIR/_setup/preflight_connections.py uipath-atlassian-jira" + timeout: 35 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN file exists and is well-formed XML" + command: "python3 -c \"import sys; sys.path.insert(0, '$REFERENCE_DIR'); from _shared.bpmn_check import parse_bpmn; parse_bpmn('PathParamsTest')\"" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN uses Jira connector or task-specific managed HTTP fallback" + command: "python3 $REFERENCE_DIR/_shared/check_managed_http_fallback.py PathParamsTest uipath-atlassian-jira path_params" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN path parameters carry the issue key ENGCE-00000" + command: "python3 $REFERENCE_DIR/_shared/check_path_param_value.py PathParamsTest ENGCE-00000" + timeout: 10 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml new file mode 100644 index 0000000000..add29abcc0 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml @@ -0,0 +1,85 @@ +task_id: skill-bpmn-query-params +description: > + Tests the query parameters IS feature — configures a connector node with a + query parameter on the Google Tasks connector. + Ported from Flow `connector_features/query_params.yaml`; the Google Tasks + list-tasks connector node is modeled as a bpmn:sendTask carrying the + registry Intsvc.ActivityExecution wrapper instead of a Flow connector node, + and the generic-HTTP-connector fallback the prompt asks for is modeled as + the same wrapper with connectorKey uipath-uipath-http (BATCH1-ADDENDUM's + connector-mode-HTTP-via-generic-connector pattern) or as a connectionless + Intsvc.HttpExecution/UnifiedHttpRequest node. Validate-only, matching + Flow's own scope — the Flow source never called `flow debug` either, only + `flow validate`. +tags: + - uipath-maestro-bpmn + - integration + - "lifecycle:generate" + - "shape:multi-node" + - connector + - uipath-google-tasks + - "mode:build" + +run_limits: + expected_turns: 43 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "QueryParamsTest" with a manual start. + You need a process that lists all Google Tasks including hidden ones. Discover + the right Google Tasks list operation and set its "show hidden" query + parameter so hidden tasks are returned. + Validate the final BPMN file. + If the Google Tasks connector operation is not present in the registry, use + a managed HTTP fallback through the generic `uipath-uipath-http` connector + for the same REST call and finish after validation. + Do not keep searching for a connector node that the registry does not expose. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN file exists and is well-formed XML" + command: "python3 -c \"import sys; sys.path.insert(0, '$REFERENCE_DIR'); from _shared.bpmn_check import parse_bpmn; parse_bpmn('QueryParamsTest')\"" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN uses Google Tasks connector or task-specific managed HTTP fallback" + command: "python3 $REFERENCE_DIR/_shared/check_managed_http_fallback.py QueryParamsTest uipath-google-tasks query_params" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 From 5dc0e3bf63cc7b72cc9ca89b275b21055c882598 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 14:52:45 -0700 Subject: [PATCH 14/35] test(bpmn): port the Flow complex_array and multiselect evals Same Slack group-DM scenario and criteria as Flow. The shared users-multiselect grader ports Flow's parse_users and is_users_key tolerance verbatim and reads the request body in both registry forms. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_complex_array.py | 62 +++++ .../_shared/check_multiselect.py | 34 +++ .../_shared/check_slack_multiselect.py | 211 ++++++++++++++++++ .../complex_array/complex_array.yaml | 95 ++++++++ .../multiselect/multiselect.yaml | 87 ++++++++ 5 files changed, 489 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py new file mode 100644 index 0000000000..16babd6350 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py @@ -0,0 +1,62 @@ +#!/usr/bin/env python3 +"""ComplexArray (BPMN): locate/parse the process file, and the advisory +resolved-user-id check. + +Ported from Flow `connector_features/complex_array.yaml`. The core "Slack +group-DM node has its users complex array populated" assertion (Flow +criterion 2) lives in the shared `_shared/check_slack_multiselect.py`, used +identically by `multiselect.yaml`; this script covers this task's other two +criteria, which are project-name-specific (Flow's task fixes the project name +"ComplexArrayTest", unlike multiselect's name-agnostic port). + +Assertion map (Flow -> BPMN): + F criterion 1 flow_contains.py --flow-name ComplexArrayTest '"nodes"' + '"edges"' (file exists and is valid JSON) -> parse_bpmn("ComplexArrayTest") locates and parses the .bpmn + I locate/parse .bpmn with the ComplexArrayTest name hint -> parse_bpmn("ComplexArrayTest") + F criterion 3 flow_contains.py --flow-name ComplexArrayTest + 'U0B7Y855WGG' 'U05Q882RHFZ' (advisory, threshold 0) -> check_ids(): same two literals searched in the located .bpmn's raw text (same weight/threshold) + +Usage: + python3 check_complex_array.py # criterion 1: locate + parse + python3 check_complex_array.py --ids # criterion 3: advisory id search +""" + +from __future__ import annotations + +import os +import sys +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import fail, parse_bpmn # noqa: E402 + +PROJECT_HINT = "ComplexArrayTest" +RECIPIENT_IDS = ("U0B7Y855WGG", "U05Q882RHFZ") + + +def check_parse() -> None: + path, _root = parse_bpmn(PROJECT_HINT) + print(f"OK: {path} exists and parses") + + +def check_ids() -> None: + path, _root = parse_bpmn(PROJECT_HINT) + text = Path(path).read_text(encoding="utf-8", errors="replace") + missing = [uid for uid in RECIPIENT_IDS if uid not in text] + for uid in RECIPIENT_IDS: + print(f"{'OK ' if uid not in missing else 'MISSING'} {uid}") + if missing: + fail(f"resolved Slack user id(s) not found in {path}: {missing}") + print(f"OK: {path} references both resolved Slack user ids") + + +def main() -> None: + if "--ids" in sys.argv[1:]: + check_ids() + else: + check_parse() + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py new file mode 100644 index 0000000000..3ea5e6eadd --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py @@ -0,0 +1,34 @@ +#!/usr/bin/env python3 +"""Multiselect (BPMN): locate/parse the process file. + +Ported from Flow `connector_features/multiselect.yaml` criterion 1 +(`flow_contains.py '"nodes"' '"edges"'`, no `--flow-name` -- Flow's prompt +fixes no project name). The core "Slack group-DM node has exactly 3 entries +in its users multiselect" assertion (Flow criterion 2) lives in the shared +`_shared/check_slack_multiselect.py`, used identically by +`complex_array.yaml`. + +Assertion map (Flow -> BPMN): + F criterion 1 flow_contains.py '"nodes"' '"edges"' (file exists and is + valid JSON, name-agnostic) -> parse_bpmn() locates and parses the .bpmn, no name hint + I locate/parse .bpmn, no name hint (Flow's prompt + names no project) -> parse_bpmn() +""" + +from __future__ import annotations + +import os +import sys + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import parse_bpmn # noqa: E402 + + +def main() -> None: + path, _root = parse_bpmn() + print(f"OK: {path} exists and parses") + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py new file mode 100644 index 0000000000..0d99787097 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py @@ -0,0 +1,211 @@ +#!/usr/bin/env python3 +"""Slack group-DM multiselect (BPMN): shared grader for `complex_array.yaml` +and `multiselect.yaml`. + +Ported from Flow `connector_features/check_multiselect_flow.py`. Same +scenario (a Slack connector node creating a group direct message, whose +`users` field is a multiselect/complex-array of user IDs), re-homed from a +JSON node's `inputs` dict (matched by `"slack" in node.type.lower()`) to a +BPMN `bpmn:sendTask` carrying the registry `Intsvc.ActivityExecution` wrapper +with `connectorKey == uipath-salesforce-slack` (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4). Both Flow +tasks call this same script with the same argv convention (`populated` or an +integer count, default 3), so this one grader is used identically by both +BPMN ports; each task's own `check_.py` handles its remaining, +task-specific criteria (locate/parse, advisory id search). + +Assertion map (Flow -> BPMN): + F check_multiselect_flow.py:86-91 "slack" in node.type.lower() -> slack_tasks(): sendTask carrying Intsvc.ActivityExecution with connectorKey == uipath-salesforce-slack + F check_multiselect_flow.py:93-104 find_users(node.inputs), count/populated -> find_users(body_object(task)), same count/populated check + F check_multiselect_flow.py:26-48 parse_users(): native list, or a string + wrapping an array literal (JSON array or + a `=js:(['U1','U2'])`-style expression) -> parse_users(): identical regex + ast.literal_eval tolerance + F check_multiselect_flow.py:51-57 is_users_key(): 'users' with an optional + array-notation suffix -> is_users_key(): identical regex + F check_multiselect_flow.py:60-76 find_users(): recursive dict/list search -> find_users(): identical recursive search over the merged body object + I locate/parse .bpmn, no name hint (this script is shared by both tasks; + complex_array's own project-name hint is handled by its own + check_complex_array.py, not here) -> parse_bpmn() + I parse a connector node's request body (Flow read a native `inputs` dict; + BPMN puts the whole request in `uipath:input` elements) -> body_object() + T the registry's two observed body forms both count: one whole-body + `target="body"` JSON blob (name="body"), or one typed `target="body"` + input per field (name=) -- registry-workflow.md documents the + first as canonical; the second is what `bpmn_check.body_object()` on + main (not yet on this branch) tolerates, so it is reimplemented locally + here per the porting brief -> body_object() + DROPPED require_no_private_connector_values, require_sequence_integrity, + require_di_for_visible_elements (not in Flow; `bpmn validate` + criterion covers structure) + +Exits non-zero with a `FAIL:` message on the first failure; prints `OK: ...` +on success. +""" + +from __future__ import annotations + +import ast +import json +import os +import re +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import ( # noqa: E402 + NS, + context_value, + elements, + fail, + has_typed_uipath_extension, + parse_bpmn, +) + +SLACK_KEY = "uipath-salesforce-slack" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" + +_USERS_KEY_RE = re.compile(r"users(\[.*\])?") + + +def node_inputs(task: ET.Element) -> list[ET.Element]: + return task.findall(".//uipath:input", NS) + + +def slack_tasks(root: ET.Element) -> list[ET.Element]: + return [ + task + for task in elements(root, "sendTask") + if has_typed_uipath_extension(task, "activity", ACTIVITY_TYPE) + and context_value(task, "connectorKey") == SLACK_KEY + ] + + +def body_object(task: ET.Element) -> dict: + """Merge every `target="body"` input on `task` into one dict, accepting + either registry-observed form: a single `name="body"` input holding the + whole request as a JSON object, or one typed input per field (its own + `name`, value/text is that field's value).""" + obj: dict = {} + for inp in node_inputs(task): + if inp.attrib.get("target") != "body": + continue + name = inp.attrib.get("name") + raw = inp.attrib.get("value") + if raw is None: + raw = inp.text + raw = (raw or "").strip() + if not raw: + continue + if name == "body": + try: + parsed = json.loads(raw) + except json.JSONDecodeError: + continue + if isinstance(parsed, dict): + obj.update(parsed) + continue + if not name: + continue + try: + obj[name] = json.loads(raw) + except json.JSONDecodeError: + obj[name] = raw + return obj + + +def is_users_key(key) -> bool: + """True for the 'users' multiselect key in any of its encodings. + + Accepts the bare name and array-notation variants: 'users', 'users[*]', + 'users[]', 'users[0]' -- same tolerance Flow's grader had for how a .flow + might spell an array-valued key. + """ + return isinstance(key, str) and _USERS_KEY_RE.fullmatch(key) is not None + + +def parse_users(value): + """Normalize a 'users' field value to a list of entries, else None. + + Accepts a native list, or a string holding an array literal such as a + "=js:(['U1','U2','U3'])" expression -- identical tolerance to Flow's + `check_multiselect_flow.parse_users`. + """ + if isinstance(value, list): + return value + if isinstance(value, str): + match = re.search(r"\[.*\]", value, re.DOTALL) + if not match: + return None + literal = match.group(0) + try: + parsed = ast.literal_eval(literal) + except (ValueError, SyntaxError): + parsed = None + if isinstance(parsed, (list, tuple)): + return list(parsed) + entries = re.findall(r"""['"]([^'"]+)['"]""", literal) + return entries or None + return None + + +def find_users(obj): + """Recursively find a 'users' key holding a parseable multiselect value.""" + if isinstance(obj, dict): + for key, value in obj.items(): + if is_users_key(key): + users = parse_users(value) + if users is not None: + return users + found = find_users(value) + if found is not None: + return found + elif isinstance(obj, list): + for item in obj: + found = find_users(item) + if found is not None: + return found + return None + + +def main() -> None: + expected_count = ( + "populated" + if len(sys.argv) > 1 and sys.argv[1] == "populated" + else int(sys.argv[1]) if len(sys.argv) > 1 else 3 + ) + + path, root = parse_bpmn() + tasks = slack_tasks(root) + if not tasks: + fail( + f"no bpmn:sendTask carrying {ACTIVITY_TYPE} for connector key " + f"{SLACK_KEY!r} in {path}" + ) + + reasons = [] + for task in tasks: + node_id = task.attrib.get("id", "") + users = find_users(body_object(task)) + if users is None: + reasons.append(f"node '{node_id}': no 'users' multiselect field found") + continue + if expected_count == "populated": + if users: + print(f"OK: {path} — node '{node_id}' users={users}") + sys.exit(0) + reasons.append(f"node '{node_id}': users field is empty") + continue + if len(users) == expected_count: + print(f"OK: {path} — node '{node_id}' users={users}") + sys.exit(0) + reasons.append( + f"node '{node_id}': users field has {len(users)} entries, " + f"expected {expected_count}: {users}" + ) + + fail("; ".join(reasons)) + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml new file mode 100644 index 0000000000..13f2dff3da --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml @@ -0,0 +1,95 @@ +# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral +# v1-base prompt, outcome-graded criteria at preserved weights, and +# arm-agnostic artifact discovery. +task_id: skill-bpmn-complex-array +description: > + Tests the complex array IS feature — configures the Slack + (uipath-salesforce-slack) "Create Group Direct Message" operation, whose + users[*] field is a complex array of user IDs. Validate-only — no + `bpmn debug` (debug would open a real group DM in the workspace). + Ported from Flow `connector_features/complex_array.yaml`; the connector + activity is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the + users array is read from its target="body" JSON (whole blob or one typed + input per field) instead of Flow's inputs.detail.bodyParameters. +tags: + - uipath-maestro-bpmn + - integration + - "lifecycle:generate" + - "shape:multi-node" + - connector + - uipath-salesforce-slack + - "mode:build" + +run_limits: + expected_turns: 38 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +# Each recipient must match exactly one Slack user, or the populated check below +# cannot pass. "Coder Eval" is deliberately NOT used: it is a prefix of "Coder +# Eval Test", so a substring lookup returns two rows and the agent is then +# required to stop and ask. Re-verify uniqueness before changing a name. +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "ComplexArrayTest" + with a manual start that creates a Slack group direct message between the + users "Coder Eval Test" and "E2E Nightly Summary". Use the connection + present in Shared/uipath-maestro-flow. Do NOT use any activity or node + marked 'beta'. Use the non-beta Slack activity. Validate the final .bpmn + file. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN file exists and parses" + command: "python3 $REFERENCE_DIR/_shared/check_complex_array.py" + timeout: 10 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN has the Slack Create Group Direct Message node with its users complex array populated" + command: "python3 $REFERENCE_DIR/_shared/check_slack_multiselect.py populated" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "Advisory: both recipients resolved to their Slack user IDs (Coder Eval Test, E2E Nightly Summary) rather than written as display names" + command: "python3 $REFERENCE_DIR/_shared/check_complex_array.py --ids" + timeout: 10 + expected_exit_code: 0 + weight: 1.0 + pass_threshold: 0 + + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml new file mode 100644 index 0000000000..2f605352ce --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml @@ -0,0 +1,87 @@ +# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral +# v1-base prompt, outcome-graded criteria at preserved weights, and +# arm-agnostic artifact discovery. +task_id: skill-bpmn-multiselect +description: > + Tests the multiselect IS feature — configures a Slack group direct message + node whose members field takes multiple values, using an existing tenant + connection. + Ported from Flow `connector_features/multiselect.yaml`; the connector + activity is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the + users multiselect is read from its target="body" JSON (whole blob or one + typed input per field) instead of Flow's inputs.detail.bodyParameters. +tags: + - uipath-maestro-bpmn + - integration + - "lifecycle:generate" + - "shape:single-node" + - connector + - "mode:build" + - uipath-salesforce-slack + +run_limits: + expected_turns: 37 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +# Each recipient must match exactly one Slack user, or the 3-entry check below +# cannot pass. "Coder Eval" is deliberately NOT used: it is a prefix of "Coder +# Eval Test", so a substring lookup returns two rows and the agent is then +# required to stop and ask. Re-verify uniqueness before changing a name. +initial_prompt: | + Create a UiPath Maestro BPMN process to create a Slack group direct message + between Coder Eval Test, IS-sandboxes@uipath.com and E2E Nightly Summary. + Do NOT use any activity or node marked 'beta'. Use the non-beta Slack activity. + Use the connection present in Shared/uipath-maestro-flow. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + # Name-agnostic: the prompt doesn't fix a project name. + - type: run_command + description: "A .bpmn file exists and parses" + command: "python3 $REFERENCE_DIR/_shared/check_multiselect.py" + timeout: 10 + expected_exit_code: 0 + weight: 1.5 + pass_threshold: 1.0 + + # Core check: Slack connector node present and the users multiselect + # field has exactly 3 entries (the three group-DM recipients). + - type: run_command + description: "Slack group-DM node present with 3 entries in the users multiselect" + command: "python3 $REFERENCE_DIR/_shared/check_slack_multiselect.py" + timeout: 15 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "The completed process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 From 12ecca2498db651a7a0c06b9f107cb28f6f41943 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 14:52:45 -0700 Subject: [PATCH 15/35] test(bpmn): port the Flow paginated_reference_lookup, enhanced_enum and searchable_joins evals Criteria one-for-one with Flow, including the two advisory CLI-discovery checks on the paged Slack channel listing. Connector presence is graded on the Intsvc.ActivityExecution connectorKey; the resolved channel id is searched across the send-message node's inputs. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_enhanced_enum.py | 100 ++++++++++++++ .../check_paginated_reference_lookup.py | 126 ++++++++++++++++++ .../_shared/check_searchable_joins.py | 105 +++++++++++++++ .../enhanced_enum/enhanced_enum.yaml | 71 ++++++++++ .../paginated_reference_lookup.yaml | 119 +++++++++++++++++ .../searchable_joins/searchable_joins.yaml | 74 ++++++++++ 6 files changed, 595 insertions(+) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_searchable_joins.py create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml create mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py new file mode 100644 index 0000000000..345395306c --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py @@ -0,0 +1,100 @@ +#!/usr/bin/env python3 +"""EnhancedEnumTest (BPMN): WooCommerce connector node presence. + +Ported from Flow `connector_features/enhanced_enum.yaml`'s ``flow_contains.py`` +criteria: same scenario (a WooCommerce "get product reviews" node exposes an +enhanced-enum sort-order field with friendly display labels rather than raw +codes), translated from a JSON node/edge substring search to an XML walk over +the registry-driven ``Intsvc.ActivityExecution`` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3). This is an +offline authoring task -- Flow's own grader never asserted a live connection, +a specific sort value, or a validate pass, and neither does this port. + +Two subcommands (subcommand-dispatched, matching the sibling connector +graders in this suite): + + check_exists Flow's ``flow_contains.py --flow-name EnhancedEnumTest + '"nodes"' '"edges"'`` -- the .bpmn exists and is + well-formed XML. + check_connector Flow's ``flow_contains.py --flow-name EnhancedEnumTest + 'uipath-automattic-woocommerce'`` -- a connector node + references the WooCommerce connector key. + +Assertion map (Flow → BPMN): + F criterion 1 flow_contains --flow-name EnhancedEnumTest '"nodes"' '"edges"' + (Flow file exists and is valid JSON) + → check_exists(): parse_bpmn("EnhancedEnumTest") (well-formed + XML is the BPMN analog of valid JSON) + F criterion 2 flow_contains --flow-name EnhancedEnumTest + 'uipath-automattic-woocommerce' + → check_connector(): an Intsvc.ActivityExecution + bpmn:sendTask with connectorKey uipath-automattic-woocommerce + I locate/parse .bpmn (file exists, well-formed XML) + → parse_bpmn() + DROPPED require_no_private_connector_values / require_sequence_integrity + / require_di_for_visible_elements / connection-binding checks + (not in Flow; Flow has no `validate` criterion on this task + either, so none is added here) + +Note: the task's tag list keeps Flow's misspelled connector tag +`uipath-automaticc-woocommerce` verbatim (per the porting brief, tags are +carried unchanged); this grader checks the REAL connector key +`uipath-automattic-woocommerce`, matching what Flow's own grader checked. +""" + +from __future__ import annotations + +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import context_value, elements, fail, has_type, parse_bpmn # noqa: E402 + +NAME_HINT = "EnhancedEnumTest" +CONNECTOR_KEY = "uipath-automattic-woocommerce" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" + + +def find_connector_node(root: ET.Element) -> ET.Element | None: + for task in elements(root, "sendTask"): + if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == CONNECTOR_KEY: + return task + return None + + +def check_exists() -> None: + path, _root = parse_bpmn(NAME_HINT) + print(f"OK: {path} exists and is well-formed XML") + + +def check_connector() -> None: + _path, root = parse_bpmn(NAME_HINT) + + node = find_connector_node(root) + if node is None: + fail( + f"no bpmn:sendTask carries {ACTIVITY_TYPE} with connectorKey " + f"{CONNECTOR_KEY!r}" + ) + print( + f"OK: {CONNECTOR_KEY} sendTask present " + f"(objectName={context_value(node, 'objectName')!r})" + ) + + +DISPATCH = { + "check_exists": check_exists, + "check_connector": check_connector, +} + + +def main() -> None: + if len(sys.argv) < 2 or sys.argv[1] not in DISPATCH: + sys.exit(f"usage: {sys.argv[0]} {{{'|'.join(DISPATCH)}}}") + DISPATCH[sys.argv[1]]() + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py new file mode 100644 index 0000000000..9c89d36862 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py @@ -0,0 +1,126 @@ +#!/usr/bin/env python3 +"""SlackPaginationTest (BPMN): connector node wired with a paginated-lookup id. + +Ported from Flow `connector_features/paginated_reference_lookup.yaml`'s +``flow_contains.py`` criteria: same scenario (a Slack "Send Message to +Channel" node targets a channel whose id can only be resolved by paging past +page 1 of the Slack channel resources, or by the SDK's lookup resolver), +translated from a JSON node/edge substring search to an XML walk over the +registry-driven ``Intsvc.ActivityExecution`` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3). The two +``command_executed`` criteria that grade the CLI discovery/pagination +transcript (paged resource-list loop, `flow validate`→`bpmn validate`) stay in +the task YAML unchanged/verb-swapped; they are not this script's concern. + +Two subcommands (subcommand-dispatched, matching the sibling connector +graders in this suite): + + check_exists Flow's ``flow_contains.py --flow-name SlackPaginationTest`` + (no extra assertions) -- the .bpmn exists and parses. + check_wired Flow's ``flow_contains.py --flow-name SlackPaginationTest + 'uipath.connector.uipath-salesforce-slack.send-message-to-channel' + '"C083AN4E61E"'`` -- a Slack send-message-to-channel node + carries the resolved channel id C083AN4E61E. + +Assertion map (Flow → BPMN): + F criterion 4 flow_contains --flow-name SlackPaginationTest (existence only) + → check_exists(): parse_bpmn("SlackPaginationTest") + F criterion 5 flow_contains --flow-name SlackPaginationTest + 'uipath.connector.uipath-salesforce-slack.send-message-to-channel' + '"C083AN4E61E"' + → check_wired(): an Intsvc.ActivityExecution bpmn:sendTask + with connectorKey uipath-salesforce-slack whose objectName + names the Send Message to Channel operation, and whose + serialised XML contains the channel id C083AN4E61E + (path/query/body input, any depth, either the curated + separate-inputs form or the single JSON `target="body"` + form -- registry-workflow.md §3 "Body shape") + I locate/parse .bpmn (file exists, well-formed XML) + → parse_bpmn() + T curated objectName spelling tolerance (send_message_to_channel + / send_message_to_channel_v2, any separator/case) + → SEND_MESSAGE_RE + T channel id present anywhere in the node's inputs, at any + depth, in either body form → has_type() substring match over + the node's full serialised XML (both a raw JSON CDATA string + and a typed separate-input value contain "C083AN4E61E" as a + literal substring) + DROPPED require_no_private_connector_values / require_sequence_integrity + / require_di_for_visible_elements / connection-binding checks + (not in Flow; the `bpmn validate` criterion covers structure) +""" + +from __future__ import annotations + +import os +import re +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import ( # noqa: E402 + context_value, + elements, + fail, + has_type, + parse_bpmn, +) + +NAME_HINT = "SlackPaginationTest" +CONNECTOR_KEY = "uipath-salesforce-slack" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" +CHANNEL_ID = "C083AN4E61E" + +# GUESS: registry-workflow.md §3's example table names `send_message_to_channel_v2` +# for this connector/operation; match loosely on the concept (any separator, +# optional "_v2"/"v2" suffix, either "message" spelling) rather than pin one +# exact objectName spelling. +SEND_MESSAGE_RE = re.compile(r"send[\s_-]*messages?[\s_-]*to[\s_-]*channel", re.IGNORECASE) + + +def find_slack_node(root: ET.Element) -> ET.Element | None: + for task in elements(root, "sendTask"): + if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == CONNECTOR_KEY: + object_name = context_value(task, "objectName") + if SEND_MESSAGE_RE.search(object_name): + return task + return None + + +def check_exists() -> None: + path, _root = parse_bpmn(NAME_HINT) + print(f"OK: {path} exists and is well-formed XML") + + +def check_wired() -> None: + _path, root = parse_bpmn(NAME_HINT) + + node = find_slack_node(root) + if node is None: + fail( + f"no bpmn:sendTask carries {ACTIVITY_TYPE} with connectorKey " + f"{CONNECTOR_KEY!r} and an objectName naming Send Message to Channel" + ) + print(f"OK: {CONNECTOR_KEY} Send Message to Channel node present " + f"(objectName={context_value(node, 'objectName')!r})") + + if not has_type(node, CHANNEL_ID): + fail(f"Slack send-message node does not carry the resolved channel id {CHANNEL_ID!r}") + print(f"OK: Slack send-message node carries channel id {CHANNEL_ID!r}") + + +DISPATCH = { + "check_exists": check_exists, + "check_wired": check_wired, +} + + +def main() -> None: + if len(sys.argv) < 2 or sys.argv[1] not in DISPATCH: + sys.exit(f"usage: {sys.argv[0]} {{{'|'.join(DISPATCH)}}}") + DISPATCH[sys.argv[1]]() + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_searchable_joins.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_searchable_joins.py new file mode 100644 index 0000000000..f14b0c23a5 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_searchable_joins.py @@ -0,0 +1,105 @@ +#!/usr/bin/env python3 +"""SearchableJoinsTest (BPMN): Salesforce connector node presence. + +Ported from Flow `connector_features/searchable_joins.yaml`'s +``flow_contains.py`` criteria: same scenario (a Salesforce query node joins +in related Opportunities for each Account in the same query, then branches on +whether any opportunities came back), translated from a JSON node/edge +substring search to an XML walk over the registry-driven +``Intsvc.ActivityExecution`` connector shell (see +skills/uipath-maestro-bpmn/references/registry-workflow.md §3). Flow's own +grader never asserted the join/branch shape itself (only that the file +exists, that a Salesforce connector node is present, and that the flow +validates), so this port keeps that same breadth rather than narrowing or +widening it. + +Two subcommands (subcommand-dispatched, matching the sibling connector +graders in this suite): + + check_exists Flow's ``flow_contains.py --flow-name SearchableJoinsTest + '"nodes"' '"edges"'`` -- the .bpmn exists and is + well-formed XML. + check_connector Flow's ``flow_contains.py --flow-name SearchableJoinsTest + 'uipath-salesforce-sfdc'`` -- a connector node references + the Salesforce connector key. + +The task's third criterion (the completed process validates) runs +``_shared/validate_bpmn.py`` directly from the YAML and needs no code here. + +Assertion map (Flow → BPMN): + F criterion 1 flow_contains --flow-name SearchableJoinsTest '"nodes"' '"edges"' + (Flow file exists and is valid JSON) + → check_exists(): parse_bpmn("SearchableJoinsTest") + (well-formed XML is the BPMN analog of valid JSON) + F criterion 2 flow_contains --flow-name SearchableJoinsTest + 'uipath-salesforce-sfdc' + → check_connector(): an Intsvc.ActivityExecution + bpmn:sendTask with connectorKey uipath-salesforce-sfdc + F criterion 3 _shared/validate_flow.py (the completed flow validates) + → _shared/validate_bpmn.py, wired directly in the task YAML + I locate/parse .bpmn (file exists, well-formed XML) + → parse_bpmn() + DROPPED require_no_private_connector_values / require_sequence_integrity + / require_di_for_visible_elements / connection-binding checks + (not in Flow; the `bpmn validate` criterion covers structure) + DROPPED join/branch shape check (Flow never asserted the related-object + join or the has-opportunities branch structurally -- only the + connector key and that the artifact validates) +""" + +from __future__ import annotations + +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) + +from _shared.bpmn_check import context_value, elements, fail, has_type, parse_bpmn # noqa: E402 + +NAME_HINT = "SearchableJoinsTest" +CONNECTOR_KEY = "uipath-salesforce-sfdc" +ACTIVITY_TYPE = "Intsvc.ActivityExecution" + + +def find_connector_node(root: ET.Element) -> ET.Element | None: + for task in elements(root, "sendTask"): + if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == CONNECTOR_KEY: + return task + return None + + +def check_exists() -> None: + path, _root = parse_bpmn(NAME_HINT) + print(f"OK: {path} exists and is well-formed XML") + + +def check_connector() -> None: + _path, root = parse_bpmn(NAME_HINT) + + node = find_connector_node(root) + if node is None: + fail( + f"no bpmn:sendTask carries {ACTIVITY_TYPE} with connectorKey " + f"{CONNECTOR_KEY!r}" + ) + print( + f"OK: {CONNECTOR_KEY} sendTask present " + f"(objectName={context_value(node, 'objectName')!r})" + ) + + +DISPATCH = { + "check_exists": check_exists, + "check_connector": check_connector, +} + + +def main() -> None: + if len(sys.argv) < 2 or sys.argv[1] not in DISPATCH: + sys.exit(f"usage: {sys.argv[0]} {{{'|'.join(DISPATCH)}}}") + DISPATCH[sys.argv[1]]() + + +if __name__ == "__main__": + main() diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml new file mode 100644 index 0000000000..d938015c65 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml @@ -0,0 +1,71 @@ +# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral +# v1-base prompt, outcome-graded criteria at preserved weights, and +# arm-agnostic artifact discovery. +task_id: skill-bpmn-enhanced-enum +description: > + Tests the enhanced enum IS feature — configures a connector node with an + enhanced enum field with display labels on the WooCommerce connector. + + Ported from Flow `connector_features/enhanced_enum.yaml`; the WooCommerce + operation is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper instead of a Flow connector node. This is + an offline authoring task — Flow's own criteria never asserted a live + connection or a validate pass, and neither does this port. +tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", connector, uipath-automaticc-woocommerce, "mode:build"] + +run_limits: + expected_turns: 35 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "EnhancedEnumTest" + with a manual start. + Build a process that retrieves WooCommerce product reviews and lets the caller + choose whether they come back oldest-first or newest-first, picking from friendly + labels rather than raw codes. Discover the get-product-reviews operation and set + its sort-order field. Produce the final .bpmn file. + If the WooCommerce operation exists in the registry but the tenant has no + live connection, keep the real WooCommerce operation node with its inputs left unset + and stop there. Do not fabricate a tenant connection merely to make product + validation pass; the connector shape is the deliverable in this offline task. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN file exists and is well-formed XML" + command: "python3 $REFERENCE_DIR/_shared/check_enhanced_enum.py check_exists" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN process has a connector node referencing uipath-automattic-woocommerce" + command: "python3 $REFERENCE_DIR/_shared/check_enhanced_enum.py check_connector" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml new file mode 100644 index 0000000000..68f01444dd --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml @@ -0,0 +1,119 @@ +# Same-ground campaign contract (Tao, 2026-08-10): loop-neutral v1-base +# prompt; v1-weighted intersection criteria plus report-only offline advisories; +# paired limit envelope. Cross-arm parity is enforced by the sync gate. +task_id: skill-bpmn-paginated-reference-lookup +description: > + Build a UiPath Maestro BPMN process with a Slack `Send Message to Channel` + node targeting the `simple` channel. The channel lives on a later page of + the `is-sandboxes` Slack channel resources, so resolving the Slack channel + id requires either explicit pagination or the SDK's lookup resolver. + Triggered by the `resources.md` read-before-call rule added in #1059 + (ENGCE-58198) — old behavior was to abandon pagination after page 1 and + bail to AskUserQuestion. + + Ported from Flow `connector_features/paginated_reference_lookup.yaml`; the + Slack send-message node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper instead of a Flow connector node, and + validation runs via `bpmn validate` instead of `flow validate`. This + structural port stays validate-only, matching Flow's own e2e task, which + also carries no `flow debug`/`bpmn debug` criterion — only the paged + discovery transcript and the resolved-id artifact are graded. +tags: + - uipath-maestro-bpmn + - integration + - e2e + - "mode:build" + - "lifecycle:generate" + - "shape:multi-node" + - connector + - uipath-salesforce-slack + - "feature:records" + +run_limits: + expected_turns: 30 + task_timeout: 1200 + max_turns: 80 + turn_timeout: 900 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +initial_prompt: | + Create a new UiPath Maestro BPMN process project called "SlackPaginationTest" + with a manual start. It should send a Slack message saying "hello from + coder-eval" to the channel named `simple` using Slack's "Send Message to + Channel" activity. + Do NOT use any activity or node marked 'beta'. Use the non-beta Slack activity. + For Slack, use the connection name `is-sandboxes`. If more than one + connection matches that name, use any enabled one. + Validate the BPMN process when you're done. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: +# TWO ROUTES ARE CORRECT: the live-v1 loop pages `uip is resources run list` +# by hand (the collection may be exposed as `conversations` or +# `curated_channels`); the SDK route resolves the name inside +# `npx flow-sdk registry prepare … --resolve channel:`. The first criterion is +# v1-route telemetry is advisory (the SDK arm never runs it); the second records +# "went past page 1" for either route and only counts a command that succeeded — +# typing the flag on a failing command does not page. It is also advisory because +# a command cut off by a turn timeout has unknown status. The final artifact +# criterion (resolved id C083AN4E61E) stays the authority. + - type: command_executed + description: "Advisory (live-v1 route): paged Slack channel-resource loop ran more than once" + tool_name: "Bash" + command_pattern: 'uip\s+is\s+resources\s+run\s+list\s+\\?"?uipath-salesforce-slack\\?"?\s+\\?"?(curated_channels|conversations)' + min_count: 2 + weight: 3.0 + pass_threshold: 0.0 + + - type: command_executed + description: "Agent went past page 1 — a channel-resource list call carrying `nextPage=`, or `registry prepare … --resolve channel:` which pages inside `prepare`" + tool_name: "Bash" + command_pattern: '(uip\s+is\s+resources\s+run\s+list\s+\\?"?uipath-salesforce-slack\\?"?\s+\\?"?(curated_channels|conversations)[^\n]*nextPage=|registry\s+prepare\s+\\?"?uipath-salesforce-slack\\?"?[^\n]*--resolve\s+\\?"?channel:)' + min_count: 1 + require_success: true + weight: 3.0 + pass_threshold: 0.0 + + - type: command_executed + description: "Agent ran `uip maestro bpmn validate` on the generated BPMN process" + tool_name: "Bash" + command_pattern: '(uip|\$UIP)\s+maestro\s+bpmn\s+validate' + min_count: 1 + weight: 2.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN file exists (path-agnostic discovery)" + command: "python3 $REFERENCE_DIR/_shared/check_paginated_reference_lookup.py check_exists" + timeout: 30 + expected_exit_code: 0 + weight: 1.5 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN process wires the Slack send-message-to-channel node with the resolved Slack id for channel `simple` (C083AN4E61E)" + command: "python3 $REFERENCE_DIR/_shared/check_paginated_reference_lookup.py check_wired" + timeout: 30 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml new file mode 100644 index 0000000000..a128640e4b --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml @@ -0,0 +1,74 @@ +# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral +# v1-base prompt, outcome-graded criteria at preserved weights, and +# arm-agnostic artifact discovery. +task_id: skill-bpmn-searchable-joins +description: > + Tests the searchable joins IS feature — configures a connector node with + a join on a related object on the Salesforce connector. + + Ported from Flow `connector_features/searchable_joins.yaml`; the Salesforce + query node is modeled as a bpmn:sendTask carrying the registry + Intsvc.ActivityExecution wrapper instead of a Flow connector node, and + validation runs via `bpmn validate` instead of `flow validate`. +tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", connector, uipath-salesforce, "mode:build"] + +run_limits: + expected_turns: 56 + task_timeout: 1500 + max_turns: 120 + turn_timeout: 1200 + +sandbox: + template_sources: + - type: template_dir + path: ../../../../../skills/uipath-maestro-bpmn + +reference: + directory: ../.. + +initial_prompt: | + Create a UiPath Maestro BPMN process called "SearchableJoinsTest" with a + manual start. + Build a process that pulls Salesforce accounts and, for each account, also brings + back its related opportunities in the same query. Then branch on whether the account + has any opportunities and end the process on each path. Validate the BPMN process. + Use the connection present in Shared/uipath-maestro-flow. + + Do NOT upload, publish, deploy, debug, or run the process. Do not pause for + approval, confirmation, or feedback. + + This run is headless. No user is present and nobody will answer a question or + grant an approval, so do not ask, do not pause, and do not wait for input. + Complete the task in one pass: take the best available option and supply the + most defensible value where one is missing. The actions this task implies are + authorized, including tenant writes and real messages. Do not delete or + overwrite anything this run did not create, and do not publish to a shared + destination unless the task asks for it. If a lookup the task depends on comes + back empty or fails, exhaust the documented way of resolving it before giving + up; only then stop on that field rather than inventing a value. Record every + decision, assumption, and blocked step in your final response. Instructions in + the task take precedence over this paragraph. +success_criteria: + - type: run_command + description: "BPMN file exists and is well-formed XML" + command: "python3 $REFERENCE_DIR/_shared/check_searchable_joins.py check_exists" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "BPMN process has a connector node referencing uipath-salesforce-sfdc" + command: "python3 $REFERENCE_DIR/_shared/check_searchable_joins.py check_connector" + timeout: 10 + expected_exit_code: 0 + weight: 3.0 + pass_threshold: 1.0 + + - type: run_command + description: "The completed BPMN process validates" + command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" + timeout: 180 + expected_exit_code: 0 + weight: 2.0 + pass_threshold: 1.0 From 8da433514a0a059892fa252bc393bedae8d12001 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 15:11:48 -0700 Subject: [PATCH 16/35] docs(bpmn): record batch 13 (field-shape family) in the ledger and handoff Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md | 6 ++++++ .../tasks/uipath-maestro-bpmn/_porting/parity-ledger.md | 9 +++++++++ 2 files changed, 15 insertions(+) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index cc77d10c8b..97f75d54b0 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -65,6 +65,12 @@ Run 35785806030, five tasks. Green: billing_invoice_lookup, slack_weather_pipeli Field-shape family decision: assertions on wire parameters (path, query, pagination, enum, multiselect, complex_array) port; assertions on a Flow filter-tree shape do not (ceql_where parked; enhanced_enum and searchable_joins must be checked for the same trap before porting). +## Batch 13 (field-shape family) + +All eight remaining field-shape evals were ported after reading their Flow graders: none asserts a filter tree, they grade wire parameters and validate. Run 35789221753: green on first run: enum, query_params, multiselect, searchable_joins, complex_array (advisory miss only). Agent failures at iteration 1: path_params (issue key left as an unbound variable), enhanced_enum (no connector node), paginated_reference_lookup (channel by name, no pagination); all three retried in batch 14 (run 35790934047). slack_channel_description_simulated parked after iteration 3 (page-1-only channel listing; Flow 4/12). + +Skill findings added: Slack channel-id resolution/pagination is not taught (three tasks now hit it); WooCommerce connector node not produced; fixed literal values get parametrised into unbound variables. + ## Probe bucket (16): pilot ported, 11 decided, 4 blocked `connector_features/ceql_where` is ported (commit fd6312fde), not yet run. The probe confirmed the filter carrier exists: `Intsvc.ActivityExecution` enrichment for the Entra `groups` List operation exposes a `where` parameter (type `query`, `FilterBuilder`, `hasCEQL: true`), and the CI-passing Data Fabric artifact carries the same tree as a `target="query" name="queryExpression" type="json"` input. As in Flow, the sandbox has no live tenant for enrichment, so the port grades the same standalone `where_detail.json` planning artifact plus the connector node and terminate end. No `bpmn validate` gate, matching Flow. Its one review flag: the groups-operation tolerance (objectName contains `group`, or `groups` + GET/list) has no CI-passed fixture yet. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index bbc2602e4d..9e30a9142d 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -72,6 +72,15 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | `interactive/bellevue_weather_simulated/…` (it.2) | `interactive/bellevue_weather_simulated/` | PARKED (skill gap) after it.2 (run 35785806030) | runtime "Cannot read property 'temperature_2m' of undefined" in the summarize script: same HttpExecution response-shape gap that parked multi_node/bellevue_weather. | | `connector_features/ceql_where.yaml` (it.2) | `connector_features/ceql_where/` | PARKED (surface gap) after it.2 (run 35785806030) | both runs: agent writes the CEQL string `displayName='active'` (the connector's sanctioned `where` parameter) plus a flat object, never the numeric-groupOperator tree Flow's grader requires. The tree is a Flow-skill construct with no BPMN carrier. Verdict for the family: tree-shape assertions do not port; wire-parameter assertions (path, query, pagination, enum, multiselect, complex_array) do. | | `interactive/slack_channel_description_simulated/…` (it.2) | `interactive/slack_channel_description_simulated/` | GRADER DEFECT it.2 (run 35785806030); fixed, it.3 in batch 13 | agent left an untyped draft under `Solution/` beside the real solution; `find_bpmn_file` without a hint saw two projects. It now drops candidates carrying no registry-typed node; replays to the real file. | +| `connector_features/enum.yaml` | `connector_features/enum/` | PASS 1.0 first run (run 35789221753) | field-shape family | +| `connector_features/query_params.yaml` | `connector_features/query_params/` | PASS 1.0 first run (run 35789221753) | field-shape family | +| `connector_features/multiselect.yaml` | `connector_features/multiselect/` | PASS 1.0 first run (run 35789221753) | field-shape family | +| `connector_features/searchable_joins.yaml` | `connector_features/searchable_joins/` | PASS 1.0 first run (run 35789221753) | field-shape family | +| `connector_features/complex_array.yaml` | `connector_features/complex_array/` | PASS 0.875 first run (run 35789221753) | only the advisory resolved-user-id check (threshold 0) missed; Flow's own version fully passes 3/12 | +| `connector_features/path_params.yaml` (it.1) | `connector_features/path_params/` | FAIL 0.83 it.1 (run 35789221753); it.2 in batch 14 | agent built the Jira GET as UnifiedHttpRequest with the issue key as an unbound variable, never the fixed ENGCE-00000 the prompt gives. Flow 12/12. | +| `connector_features/enhanced_enum.yaml` (it.1) | `connector_features/enhanced_enum/` | FAIL 0.5 it.1 (run 35789221753); it.2 in batch 14 | agent produced no connector node at all (only an enum-typed variable). Flow 12/12. | +| `connector_features/paginated_reference_lookup.yaml` (it.1) | `connector_features/paginated_reference_lookup/` | FAIL 0.28 it.1 (run 35789221753); it.2 in batch 14 | agent sent to channel "simple" by name, never resolved/paginated to C083AN4E61E; both discovery advisories 0. Flow 11/12: BPMN skill does not teach Slack channel-id resolution. | +| `interactive/slack_channel_description_simulated/…` (it.3) | `interactive/slack_channel_description_simulated/` | PARKED (skill gap) after it.3 (run 35789221753) | debug completed; the agent listed page 1 of conversations and scripted the description out of it, so office-bellevue never appeared. Same pagination gap as multi_node/slack_channel_description it.1. Flow's version fully passes 4/12. | | `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | ## Ported 1:1 (21) From 25a39c163950a50be18ed1de66408a1f37cce3f4 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 15:22:55 -0700 Subject: [PATCH 17/35] docs(bpmn): record batch 14 and the batch 15 dispatch Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md | 4 ++++ tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md | 3 +++ 2 files changed, 7 insertions(+) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index 97f75d54b0..d7c303fd5d 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -71,6 +71,10 @@ All eight remaining field-shape evals were ported after reading their Flow grade Skill findings added: Slack channel-id resolution/pagination is not taught (three tasks now hit it); WooCommerce connector node not produced; fixed literal values get parametrised into unbound variables. +## Batch 14 and 15 + +Run 35790934047: path_params green on iteration 2; enhanced_enum parked (no connector node in either run: the WooCommerce connector never materialises as a BPMN node); paginated_reference_lookup got past discovery but wrote the channel id with its last character dropped. Its third and final iteration is run 35791969905 (batch 15); record its result here. + ## Probe bucket (16): pilot ported, 11 decided, 4 blocked `connector_features/ceql_where` is ported (commit fd6312fde), not yet run. The probe confirmed the filter carrier exists: `Intsvc.ActivityExecution` enrichment for the Entra `groups` List operation exposes a `where` parameter (type `query`, `FilterBuilder`, `hasCEQL: true`), and the CI-passing Data Fabric artifact carries the same tree as a `target="query" name="queryExpression" type="json"` input. As in Flow, the sandbox has no live tenant for enrichment, so the port grades the same standalone `where_detail.json` planning artifact plus the connector node and terminate end. No `bpmn validate` gate, matching Flow. Its one review flag: the groups-operation tolerance (objectName contains `group`, or `groups` + GET/list) has no CI-passed fixture yet. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index 9e30a9142d..3d3dedb7b7 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -81,6 +81,9 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | `connector_features/enhanced_enum.yaml` (it.1) | `connector_features/enhanced_enum/` | FAIL 0.5 it.1 (run 35789221753); it.2 in batch 14 | agent produced no connector node at all (only an enum-typed variable). Flow 12/12. | | `connector_features/paginated_reference_lookup.yaml` (it.1) | `connector_features/paginated_reference_lookup/` | FAIL 0.28 it.1 (run 35789221753); it.2 in batch 14 | agent sent to channel "simple" by name, never resolved/paginated to C083AN4E61E; both discovery advisories 0. Flow 11/12: BPMN skill does not teach Slack channel-id resolution. | | `interactive/slack_channel_description_simulated/…` (it.3) | `interactive/slack_channel_description_simulated/` | PARKED (skill gap) after it.3 (run 35789221753) | debug completed; the agent listed page 1 of conversations and scripted the description out of it, so office-bellevue never appeared. Same pagination gap as multi_node/slack_channel_description it.1. Flow's version fully passes 4/12. | +| `connector_features/path_params.yaml` (it.2) | `connector_features/path_params/` | PASS 1.0 it.2 (run 35790934047) | agent used the curated Jira get-issue node with issueId=ENGCE-00000 as a path input | +| `connector_features/enhanced_enum.yaml` (it.2) | `connector_features/enhanced_enum/` | PARKED (skill gap) after it.2 (run 35790934047) | both runs: no connector node in the artifact at all (types empty; only an enum-typed variable). The WooCommerce connector never gets a node in BPMN; Flow 12/12. | +| `connector_features/paginated_reference_lookup.yaml` (it.2) | `connector_features/paginated_reference_lookup/` | FAIL 0.76 it.2 (run 35790934047); it.3 in batch 15 (run 35791969905) | discovery advisories now pass; the send node carries channel "C083AN4E61", the resolved id with its last character dropped. Agent transcription error, not grader. | | `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | ## Ported 1:1 (21) From 4c68886005142a878b058f541e2694228089fbeb Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Tue, 22 Sep 2026 15:33:23 -0700 Subject: [PATCH 18/35] docs(bpmn): final live-tier status after batch 15 (23 green, 8 parked) Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/LIVE-HANDOFF.md | 85 +++++++------------ .../_porting/parity-ledger.md | 1 + 2 files changed, 30 insertions(+), 56 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index d7c303fd5d..0870b64be9 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -8,7 +8,9 @@ The Flow suite (`tests/tasks/uipath-maestro-flow/`) has 131 tasks; the BPMN suit Reading the Flow graders during the loop reclassified four "structural" tasks as live (their graders debug): `slack_http_fallback`, `bellevue_weather_simulated`, `cli_dice_roller_simulated`, `slack_channel_description_simulated`; and four "live" tasks as structural (their graders never debug): `smoke_error`, `jdbc_databricks_query`, `webhook_waitfor_parallel`, `testmanager_crud_grounded` (self-reported result file, as Flow). The live bucket is therefore 21 tasks. -## Live bucket status +## Status (as of 2026-09-22, after batch 15) + +Live bucket (21) and field-shape probes (9) on this branch. Every row is a CI result on the alpha tenant, codex driver. | Task | State | Evidence | |---|---|---| @@ -16,64 +18,35 @@ Reading the Flow graders during the loop reclassified four "structural" tasks as | e2e/jira_get_issue | green | run 35501830119 | | e2e/jira_create_issue | green | run 35503094182 | | e2e/escalation_jira_ticket | green | run 35503094182 | -| e2e/escalation_orchestrator_paths | green (7 debug runs) | run 35503094182 | -| e2e/escalation_slack_alert | green on iteration 2 | run 35524004307; iteration 1 the agent omitted the Slack `folderKey` binding (runtime 102010) | -| multi_node/slack_channel_description | green on iteration 2 | run 35525387843; iteration 1 the agent omitted the channel parameter | -| multi_node/bellevue_weather | parked, skill gap | runs 35523787101 + 35525387843: identical runtime fault, the script task reads `temperature_2m` off an undefined HTTP response. The skill does not teach the `Intsvc.HttpExecution` response shape well enough for downstream scripts | -| e2e/jira_search_triage | parked, skill/platform gap | run 35525387843: runtime 400008 "Failed to evaluate the input collection variable for the marker element" — `multiInstanceLoopCharacteristics` over a connector response (`=vars.Var_SearchResponse.issues`) does not evaluate | -| e2e/jira_lifecycle | parked, needs live investigation | three different runtime failures in three runs (our CLI poll cap bug; instance never terminal in 720 s; `bpmn debug` exit 1 before creating an instance). Flow's own version is flaky (0.82 typical, 2/12 zero in the week's nightlies) | -| multi_node/slack_weather_pipeline | FAIL 0.375, iteration 1 of 3 | run 35538279757: runtime 300501 "Slack channel office-bellevue was not found" in the agent's channel-select script; agent defect (channel exists, Flow finds it) | -| multi_node/billing_invoice_lookup | green on the graded criteria (0.91); bindings advisory fixed, not re-run | run 35538279757; grader read its own ephemeral live solution as a second project | -| multi_node/billing_discrepancy_detector | FAIL 0.30, iteration 1 of 3 | run 35538279757: Integration Services 400 "Expected a field name expression but got 'StringValue'" on the ERP query (malformed Data Service filter, agent authoring); accountTier not derived from CRM | +| e2e/escalation_orchestrator_paths | green | run 35503094182 | +| e2e/escalation_slack_alert | green it.2 | run 35524004307 | +| multi_node/slack_channel_description | green it.2 | run 35525387843 | | connector_features/generic_dynamic_node | green | run 35538279757 | -| connector_features/slack_http_fallback | 0.76; grader fixed, not re-run | run 35538279757: debug completed; grader wanted `emoji.list`, connector generic resource is `emoji_list_GET` | | connector_features/jdbc_databricks_query (structural) | green | run 35538279757 | -| connector_trigger/webhook_waitfor_parallel (structural) | 0.47; grader fixed, not re-run | run 35538279757: agent used intermediateCatchEvent + WaitForEvent and Intsvc.UnifiedHttpRequest, both valid | | connector_features/datafabric_connector/smoke_error (structural) | green | run 35538279757 | -| connector_features/testmanager_crud_grounded (self-report, Flow `skip:true` dropped) | 0.89; grader fixed, not re-run | run 35538279757: two byte-identical `.bpmn` (scaffold + solution copy) | -| interactive/bellevue_weather_simulated | written, reviewed, not run | commit 80033663f; live criterion timeout 1050, task_timeout 2550 (sanctioned) | -| interactive/cli_dice_roller_simulated | written, reviewed, not run | commit 8fc7a7692; task_timeout 2800 (sanctioned) | -| interactive/slack_channel_description_simulated | written, reviewed, not run | commit f000d026d; all five Flow criteria kept, live timeout 1050 | - -Batch 10 landed; see "Batch 10 and after". The three interactive ports have never been dispatched. - -## Batch 10 and after - -Run 35538279757 (nine ports, one dispatch). Green: smoke_error, generic_dynamic_node, jdbc_databricks_query. Four more failed only on grader defects, all fixed on this branch and replayed green against the downloaded CI artifacts (`gh run download 35538279757`, `**/00/artifacts/`): - -1. `find_bpmn_file` with no hint now treats byte-identical `.bpmn` copies as one artifact (testmanager_crud_grounded: the agent copied its scaffold into the solution wrapper). -2. `resolve_project(exclude_under=…)`: a live grader's own `uip solution projects import` leaves an identical project under its run directory; later criteria in the same task must exclude it (both billing graders pass `LIVE_RUN_DIR`). Any future multi-criterion live grader needs the same. -3. Classify wait-for-event by the `Intsvc.WaitForEvent` wrapper, not the BPMN tag: the agent emits `bpmn:intermediateCatchEvent` + messageEventDefinition as well as `bpmn:receiveTask`, and both validate (webhook_waitfor_parallel; same lesson as trigger_lifecycle for EventTrigger). -4. Accept `Intsvc.UnifiedHttpRequest` wherever a grader accepts `Intsvc.HttpExecution`; registry-workflow.md lists both for the managed HTTP sendTask. -5. Slack's generic resource for the `emoji.list` endpoint is `emoji_list_GET`; the fallback grader matches `emoji[._]list`. - -Two real failures, one iteration spent each: billing_discrepancy_detector (Integration Services 400 on the ERP query filter, "Expected a field name expression but got 'StringValue'": the agent wrote a malformed Data Service filter; add to the skill findings as "Data Service query filter grammar") and slack_weather_pipeline (script task could not find channel `office-bellevue`, which exists and Flow's agent finds; likely channel-list pagination). - -Next dispatch, one batch: the four grader-fixed tasks for confirmation, the two real failures (iteration 2), the three interactive simulated ports and `ceql_where` (first run). Ten tasks. - -## Batch 11 (after the structural PR merged) - -Branch rebased onto main (PR #3426 squashed as 345c1da8a). Run 35783045540, ten tasks. Green: slack_http_fallback, webhook_waitfor_parallel, testmanager_crud_grounded (the three batch-10 grader fixes confirmed) and cli_dice_roller_simulated (first run). billing_invoice_lookup hit a platform 504 during polling (infra, rerun). bellevue_weather_simulated's simulation stopped on turn 1 with empty agent output (harness, rerun). Real failures: billing_discrepancy_detector parked after two different malformed Data Service where clauses (skill gap: where clause from a process variable); slack_channel_description_simulated (channel_not_found), slack_weather_pipeline (wrong Slack connection bound, 401 + MISSING_BINDING) and ceql_where (CEQL string instead of the canonical tree) get one more iteration. Flow's own nightlies for these three pass 4/12, 6/12 and 11/12, so the BPMN flakiness is at parity except ceql_where. - -Grader tolerance added this round: every live grader that classified managed HTTP by `Intsvc.HttpExecution` now also accepts `Intsvc.UnifiedHttpRequest` (both listed in registry-workflow.md). Also landed on main via #3476: `bpmn_check.body_object()` reads sendTask bodies in both registry forms (one JSON blob, or one typed input per field); any new grader that reads a body must use it. - -Batch 12 = billing_invoice_lookup (rerun), bellevue_weather_simulated (rerun), slack_channel_description_simulated (it.2), slack_weather_pipeline (it.3), ceql_where (it.2). - -## Batch 12 - -Run 35785806030, five tasks. Green: billing_invoice_lookup, slack_weather_pipeline (iteration 3). Parked: bellevue_weather_simulated (same `temperature_2m` response-shape fault as its non-simulated twin) and ceql_where (the agent writes the connector's sanctioned CEQL `where` string, never Flow's numeric-groupOperator tree; the tree is a Flow-skill construct). slack_channel_description_simulated failed on a grader defect: an abandoned untyped draft under `Solution/` beside the real solution; `find_bpmn_file` now drops candidates with no registry-typed node. Iteration 3 goes in batch 13 with the field-shape ports. - -Field-shape family decision: assertions on wire parameters (path, query, pagination, enum, multiselect, complex_array) port; assertions on a Flow filter-tree shape do not (ceql_where parked; enhanced_enum and searchable_joins must be checked for the same trap before porting). - -## Batch 13 (field-shape family) - -All eight remaining field-shape evals were ported after reading their Flow graders: none asserts a filter tree, they grade wire parameters and validate. Run 35789221753: green on first run: enum, query_params, multiselect, searchable_joins, complex_array (advisory miss only). Agent failures at iteration 1: path_params (issue key left as an unbound variable), enhanced_enum (no connector node), paginated_reference_lookup (channel by name, no pagination); all three retried in batch 14 (run 35790934047). slack_channel_description_simulated parked after iteration 3 (page-1-only channel listing; Flow 4/12). - -Skill findings added: Slack channel-id resolution/pagination is not taught (three tasks now hit it); WooCommerce connector node not produced; fixed literal values get parametrised into unbound variables. - -## Batch 14 and 15 - -Run 35790934047: path_params green on iteration 2; enhanced_enum parked (no connector node in either run: the WooCommerce connector never materialises as a BPMN node); paginated_reference_lookup got past discovery but wrote the channel id with its last character dropped. Its third and final iteration is run 35791969905 (batch 15); record its result here. +| connector_features/slack_http_fallback | green | run 35783045540 | +| connector_trigger/webhook_waitfor_parallel (structural) | green | run 35783045540 | +| connector_features/testmanager_crud_grounded | green | run 35783045540 | +| interactive/cli_dice_roller_simulated | green | run 35783045540 | +| multi_node/billing_invoice_lookup | green | run 35785806030 | +| multi_node/slack_weather_pipeline | green it.3 | run 35785806030 | +| connector_features/enum | green | run 35789221753 | +| connector_features/query_params | green | run 35789221753 | +| connector_features/multiselect | green | run 35789221753 | +| connector_features/searchable_joins | green | run 35789221753 | +| connector_features/complex_array | green 0.875 (advisory miss only) | run 35789221753 | +| connector_features/path_params | green it.2 | run 35790934047 | +| connector_features/paginated_reference_lookup | green it.3 | run 35791969905 | +| multi_node/bellevue_weather | parked, skill gap | HttpExecution response shape (`temperature_2m` of undefined), runs 35523787101 + 35525387843 | +| interactive/bellevue_weather_simulated | parked, same gap | run 35785806030 | +| e2e/jira_search_triage | parked, platform/skill gap | multi-instance over a connector response, 400008 (run 35525387843) | +| e2e/jira_lifecycle | parked, needs live investigation | three different runtime failures; Flow flaky | +| multi_node/billing_discrepancy_detector | parked, skill gap | Data Service where clause from a process variable, two different 400s (runs 35538279757, 35783045540) | +| interactive/slack_channel_description_simulated | parked, skill gap | Slack channel pagination: page 1 only (run 35789221753); Flow 4/12 | +| connector_features/ceql_where | parked, surface gap | agent writes the CEQL `where` string, never Flow's filter tree (runs 35783045540, 35785806030) | +| connector_features/enhanced_enum | parked, skill gap | no WooCommerce connector node in either run (runs 35789221753, 35790934047) | + +Total: 23 green, 8 parked. Not started: the 4 probes that need tenant fixtures (billing_dispute_analyst / _resolution / _writer need a published agent substitute for Flow inline agents; single_node/file_attachment needs a file-typed process variable). ## Probe bucket (16): pilot ported, 11 decided, 4 blocked diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index 3d3dedb7b7..b4159c942e 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -84,6 +84,7 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | `connector_features/path_params.yaml` (it.2) | `connector_features/path_params/` | PASS 1.0 it.2 (run 35790934047) | agent used the curated Jira get-issue node with issueId=ENGCE-00000 as a path input | | `connector_features/enhanced_enum.yaml` (it.2) | `connector_features/enhanced_enum/` | PARKED (skill gap) after it.2 (run 35790934047) | both runs: no connector node in the artifact at all (types empty; only an enum-typed variable). The WooCommerce connector never gets a node in BPMN; Flow 12/12. | | `connector_features/paginated_reference_lookup.yaml` (it.2) | `connector_features/paginated_reference_lookup/` | FAIL 0.76 it.2 (run 35790934047); it.3 in batch 15 (run 35791969905) | discovery advisories now pass; the send node carries channel "C083AN4E61", the resolved id with its last character dropped. Agent transcription error, not grader. | +| `connector_features/paginated_reference_lookup.yaml` (it.3) | `connector_features/paginated_reference_lookup/` | PASS 1.0 it.3 (run 35791969905) | discovery advisories and the resolved channel id all green | | `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | ## Ported 1:1 (21) From 00f654964da08c9fa8ba3cdcc38d07ff4ffbc65d Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Wed, 23 Sep 2026 12:18:54 -0700 Subject: [PATCH 19/35] test(bpmn): skip the eight parked live ports until their skill or platform gap closes Each carries its CI evidence in a comment. Criteria stay one-for-one with Flow so the tasks resume unchanged once the gap is fixed; skipping keeps known failures out of the nightly and smoke gates, as the structural PR's review asked for its own parked tasks. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../connector_features/ceql_where/ceql_where.yaml | 4 ++++ .../connector_features/enhanced_enum/enhanced_enum.yaml | 4 ++++ .../e2e/jira_lifecycle/jira_lifecycle.yaml | 4 ++++ .../e2e/jira_search_triage/jira_search_triage.yaml | 4 ++++ .../bellevue_weather_simulated.yaml | 4 ++++ .../slack_channel_description_simulated.yaml | 4 ++++ .../multi_node/bellevue_weather/bellevue_weather.yaml | 4 ++++ .../billing_discrepancy_detector.yaml | 4 ++++ 8 files changed, 32 insertions(+) diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml index 856c926890..0ab76bbfb6 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml @@ -19,6 +19,10 @@ description: > either authoring loop) applies identically to Flow and to Maestro BPMN. tags: [uipath-maestro-bpmn, integration, connector, ceql, filter, uipath-microsoft-azureactivedirectory, "mode:build"] +# Parked (surface gap). Criteria stay one-for-one with Flow; unskip when the gap closes. +# surface gap: the agent writes the connector's sanctioned CEQL `where` string, never Flow's numeric-groupOperator filter tree, which has no BPMN carrier (CI runs 35783045540, 35785806030). +skip: true + run_limits: expected_turns: 34 task_timeout: 1500 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml index d938015c65..8e520bb159 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml @@ -13,6 +13,10 @@ description: > connection or a validate pass, and neither does this port. tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", connector, uipath-automaticc-woocommerce, "mode:build"] +# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. +# skill gap: no WooCommerce connector node is produced in either run (CI runs 35789221753, 35790934047). +skip: true + run_limits: expected_turns: 35 task_timeout: 1500 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml index ff23e971f4..c552f31fd4 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml @@ -34,6 +34,10 @@ description: > payload. tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", "node:loop", "node:switch", "node:decision", connector, e2e, uipath-atlassian-jira, "mode:build"] +# Parked (three different runtime failures in three runs (CI runs 35503094182, 35524004307, 35525387843); Flow's own version is flaky. Needs a live investigation before it can gate anything.). Criteria stay one-for-one with Flow; unskip when the gap closes. +# three different runtime failures in three runs (CI runs 35503094182, 35524004307, 35525387843); Flow's own version is flaky. Needs a live investigation before it can gate anything. +skip: true + run_limits: expected_turns: 55 task_timeout: 2700 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml index d4082cee26..1da8807818 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml @@ -26,6 +26,10 @@ description: > tenant re-read Flow itself performs (arithmetic in the grader's docstring). tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", "node:loop", connector, e2e, uipath-atlassian-jira, "mode:build"] +# Parked (platform/skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. +# platform/skill gap: multiInstanceLoopCharacteristics over a connector response fails at runtime with 400008 (CI run 35525387843). +skip: true + run_limits: expected_turns: 45 task_timeout: 2400 diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml index 566a4eba2e..ca621a8c9d 100644 --- a/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml +++ b/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml @@ -26,6 +26,10 @@ description: > as elicited) are unchanged from Flow's own grader. tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", "node:decision", "feature:http", simulation] +# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. +# skill gap: same `temperature_2m`-of-undefined runtime fault as multi_node/bellevue_weather (CI run 35785806030). +skip: true + sandbox: template_sources: - type: template_dir diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml index bb3a49776a..ccaf3ad26f 100644 --- a/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml +++ b/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml @@ -30,6 +30,10 @@ description: > withheld-project-name advisory. tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", connector, simulation] +# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. +# skill gap: the agent lists only page 1 of Slack conversations, so the target channel never appears (CI run 35789221753); Flow's version fully passes 4/12 nightlies. +skip: true + run_limits: # Ceiling on agent turns across the whole dialog, with generous headroom so a # long build is never truncated mid-flow. The dialog length itself is bounded diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml index cc225770ad..1cf3f8fde4 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml @@ -21,6 +21,10 @@ description: > the two verdict strings) are unchanged from Flow's own grader. tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", "node:decision", "feature:http"] +# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. +# skill gap: the summarize script reads `temperature_2m` off an undefined Intsvc.HttpExecution response in every run (CI runs 35523787101, 35525387843); the skill does not teach the managed-HTTP response shape. +skip: true + sandbox: template_sources: - type: template_dir diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml index b3f54670f1..e980f3ad9d 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml @@ -28,6 +28,10 @@ tags: - connector - path-to-ga +# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. +# skill gap: the Data Service where clause built from a process variable is malformed in both runs (400 'Expected a field name expression', then an empty interpolated value; CI runs 35538279757, 35783045540). +skip: true + run_limits: max_turns: 120 turn_timeout: 2400 From fa00806c22dbb158565d31d00efd34808c998dd0 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Wed, 23 Sep 2026 12:25:20 -0700 Subject: [PATCH 20/35] test(bpmn): move the eight parked ports to test/bpmn-port-parked They never went green in CI (skill or platform gaps, evidence in each YAML) and were carried as skip: true. A PR of passing tasks should not ship eight skipped ones; they live on a stacked branch until their gap closes. The ledger is collapsed to one final row per task and points at that branch. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/LIVE-HANDOFF.md | 20 +- .../_porting/parity-ledger.md | 139 ++-- .../check_billing_discrepancy_detector.py | 780 ------------------ .../_shared/check_ceql_where.py | 269 ------ .../check_channel_description_simulated.py | 287 ------- .../_shared/check_enhanced_enum.py | 100 --- .../_shared/check_jira_lifecycle.py | 354 -------- .../_shared/check_jira_search_triage.py | 283 ------- .../_shared/check_weather_bpmn.py | 275 ------ .../_shared/check_weather_bpmn_simulated.py | 299 ------- .../ceql_where/ceql_where.yaml | 76 -- .../enhanced_enum/enhanced_enum.yaml | 75 -- .../e2e/jira_lifecycle/_setup/jira_is.py | 67 -- .../e2e/jira_lifecycle/_setup/seed_jira.py | 39 - .../jira_lifecycle/_setup/teardown_jira.py | 22 - .../e2e/jira_lifecycle/jira_lifecycle.yaml | 135 --- .../e2e/jira_search_triage/_setup/jira_is.py | 65 -- .../jira_search_triage/_setup/seed_jira.py | 26 - .../_setup/teardown_jira.py | 22 - .../jira_search_triage.yaml | 111 --- .../bellevue_weather_simulated.yaml | 142 ---- .../slack_channel_description_simulated.yaml | 192 ----- .../bellevue_weather/bellevue_weather.yaml | 95 --- .../billing_discrepancy_detector.yaml | 131 --- 24 files changed, 76 insertions(+), 3928 deletions(-) delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml delete mode 100644 tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py delete mode 100644 tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml delete mode 100644 tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml delete mode 100644 tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml delete mode 100644 tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml delete mode 100644 tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index 0870b64be9..532e2b2ad5 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -1,6 +1,6 @@ # Flow → BPMN eval porting: live-tier handoff -Branch: `test/bpmn-port-live`, stacked on `test/bpmn-port-connectors` (PR #3426, the structural bucket). This document is the state of the live-tier port work as of 2026-09-20 and how to continue it. Methodology files sit beside it in this directory. +Branch `test/bpmn-port-live` (PR #3502) holds the live-tier and field-shape ports that pass; branch `test/bpmn-port-parked` (stacked on it) holds the eight that do not, each `skip: true` with its evidence. The structural bucket is PR #3426. Methodology files sit beside this document; per-task history is `parity-ledger.md`. ## Context @@ -10,7 +10,7 @@ Reading the Flow graders during the loop reclassified four "structural" tasks as ## Status (as of 2026-09-22, after batch 15) -Live bucket (21) and field-shape probes (9) on this branch. Every row is a CI result on the alpha tenant, codex driver. +Live bucket (21) and field-shape probes (9). Every row is a CI result on the alpha tenant, codex driver; green rows are in PR #3502, parked rows on `test/bpmn-port-parked`. | Task | State | Evidence | |---|---|---| @@ -37,14 +37,14 @@ Live bucket (21) and field-shape probes (9) on this branch. Every row is a CI re | connector_features/complex_array | green 0.875 (advisory miss only) | run 35789221753 | | connector_features/path_params | green it.2 | run 35790934047 | | connector_features/paginated_reference_lookup | green it.3 | run 35791969905 | -| multi_node/bellevue_weather | parked, skill gap | HttpExecution response shape (`temperature_2m` of undefined), runs 35523787101 + 35525387843 | -| interactive/bellevue_weather_simulated | parked, same gap | run 35785806030 | -| e2e/jira_search_triage | parked, platform/skill gap | multi-instance over a connector response, 400008 (run 35525387843) | -| e2e/jira_lifecycle | parked, needs live investigation | three different runtime failures; Flow flaky | -| multi_node/billing_discrepancy_detector | parked, skill gap | Data Service where clause from a process variable, two different 400s (runs 35538279757, 35783045540) | -| interactive/slack_channel_description_simulated | parked, skill gap | Slack channel pagination: page 1 only (run 35789221753); Flow 4/12 | -| connector_features/ceql_where | parked, surface gap | agent writes the CEQL `where` string, never Flow's filter tree (runs 35783045540, 35785806030) | -| connector_features/enhanced_enum | parked, skill gap | no WooCommerce connector node in either run (runs 35789221753, 35790934047) | +| multi_node/bellevue_weather | parked on `test/bpmn-port-parked`, skill gap | HttpExecution response shape (`temperature_2m` of undefined), runs 35523787101 + 35525387843 | +| interactive/bellevue_weather_simulated | parked on `test/bpmn-port-parked`, same gap | run 35785806030 | +| e2e/jira_search_triage | parked on `test/bpmn-port-parked`, platform/skill gap | multi-instance over a connector response, 400008 (run 35525387843) | +| e2e/jira_lifecycle | parked on `test/bpmn-port-parked`, needs live investigation | three different runtime failures; Flow flaky | +| multi_node/billing_discrepancy_detector | parked on `test/bpmn-port-parked`, skill gap | Data Service where clause from a process variable, two different 400s (runs 35538279757, 35783045540) | +| interactive/slack_channel_description_simulated | parked on `test/bpmn-port-parked`, skill gap | Slack channel pagination: page 1 only (run 35789221753); Flow 4/12 | +| connector_features/ceql_where | parked on `test/bpmn-port-parked`, surface gap | agent writes the CEQL `where` string, never Flow's filter tree (runs 35783045540, 35785806030) | +| connector_features/enhanced_enum | parked on `test/bpmn-port-parked`, skill gap | no WooCommerce connector node in either run (runs 35789221753, 35790934047) | Total: 23 green, 8 parked. Not started: the 4 probes that need tenant fixtures (billing_dispute_analyst / _resolution / _writer need a published agent substitute for Flow inline agents; single_node/file_attachment needs a file-typed process variable). diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index b4159c942e..e537cbaf47 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -11,81 +11,74 @@ Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 | Portable pending a feasibility probe | 16 | | Not portable (Flow-only surface) | 30 | -## Porting ledger (branch `test/bpmn-port-connectors`) +## Porting ledger -| Flow task | BPMN port | CI result | Notes | +One row per task, final state. Iterations are summarised in the notes; runs are GitHub Actions `run-coder-eval.yml` ids on the alpha tenant, codex driver. Structural rows landed in PR #3426; live and field-shape rows are PR #3502; parked rows live on branch `test/bpmn-port-parked` (stacked on #3502), each with `skip: true` and its evidence in the YAML. + +### Structural (PR #3426) + +| Flow task | BPMN port | Final | Notes | +|---|---|---|---| +| `connector_features/drive_to_slack.yaml` | `connector_features/drive_to_slack/` | PASS (run 35484984200) | pilot | +| `…/datafabric_connector/smoke_create_all_types.yaml` | `…/smoke_create_all_types/` | PASS it.2 (run 35489744689) | grader widened to the generic entity-CRUD form | +| `…/datafabric_connector/integration_create_get.yaml` | `…/integration_create_get/` | PASS it.2 (run 35489744689) | same | +| `…/datafabric_connector/contractregistry_crud_filters.yaml` | `…/contractregistry_crud_filters/` | PASS it.2 (run 35489744689) | same, plus transitive output mapping | +| `…/datafabric_connector/smoke_query.yaml` | `…/smoke_query/` | SKIPPED, surface gap | passed it.3 (run 35490499651) only via an invented `sortField` input; the curated Query Entity Records template has no sort-field parameter (only `isAscending`), so the "sorted by score" criterion has no carrier. Two smoke-gate runs (35674400362) confirmed. | +| `…/datafabric_connector/smoke_update.yaml` | `…/smoke_update/` | PASS (run 35499789502) | | +| `…/datafabric_connector/smoke_file_activities.yaml` | `…/smoke_file_activities/` | PASS (run 35499789502) | | +| `…/datafabric_connector/e2e_contract_intake_pipeline.yaml` | `…/e2e_contract_intake_pipeline/` | PASS (run 35499789502) | | +| `…/datafabric_connector/trigger_lifecycle.yaml` | `…/trigger_lifecycle/` | PASS it.3 (run 35501830119) | it.1 agent omitted the entity parameter; it.2 grader required messageEventDefinition as a direct child | +| `…/datafabric_connector/smoke_update_existing_flow.yaml` | `…/smoke_update_existing_flow/` | PASS (run 35500726138) | brownfield scaffold via `bpmn init` | +| `connector_features/testmanager_{testcase,testset,requirement}_lifecycle`, `testmanager_{attachments,execution_results,generic_records}` | same names | PASS (runs 35488848026, 35499789502) | Flow's `skip: true` not carried over | +| `connector_features/non-catalog-http-fallback/…` | `connector_features/non_catalog_http_fallback/` | PASS (run 35500726138) | grader now ActivityExecution-only (#3476) | +| `single_node/outlook_waitfor_email/…` | same | PASS (run 35500726138) | | +| `single_node/outlook_trigger_inbox/…` | same | PASS it.2 (run 35501830119) | it.1 agent omitted `parentFolderId`; `uip is triggers` advisory stays 0 | +| `e2e/devcon_expense_approval.yaml` | `e2e/devcon_expense_approval/` | SKIPPED, platform gap (0.885 it.3, run 35503094182) | every HITL assertion passes; `validate` needs a deployed Action App for Actions.HITL | +| `interactive/customer_escalation_simulated/…` | same | PASS 0.94 (run 35501830119) | advisory name check missed, as in Flow | +| `interactive/expense_approval_simulated/…` | same | SKIPPED, platform gap (0.70) | same Action App gap | +| `interactive/hitl_schema_design_simulated/…` | same | SKIPPED, platform gap (0.68) | same Action App gap | +| `interactive/solution_select.yaml` | `interactive/solution_select/` | SKIPPED, skill gap (0.46, run 35538478362) | no "existing solutions → ask" greenfield rule; agent auto-scaffolds | +| `connector_trigger/trigger_with_filter.yaml` | — | not ported | no structured trigger-filter carrier on `Intsvc.EventTrigger` | + +### Live and field-shape (PR #3502) + +| Flow task | BPMN port | Final | Notes | |---|---|---|---| -| `connector_features/drive_to_slack.yaml` | `connector_features/drive_to_slack/` | PASS 1/1 (run 35484984200) | pilot | -| `connector_features/datafabric_connector/smoke_create_all_types.yaml` | `…/datafabric_connector/smoke_create_all_types/` | PASS (run 35489744689, iteration 2) | grader widened to generic entity-CRUD form | -| `connector_features/datafabric_connector/integration_create_get.yaml` | `…/datafabric_connector/integration_create_get/` | PASS (run 35489744689, iteration 2) | same | -| `connector_features/datafabric_connector/contractregistry_crud_filters.yaml` | `…/datafabric_connector/contractregistry_crud_filters/` | PASS (run 35489744689, iteration 2) | same + transitive output mapping | -| `connector_features/datafabric_connector/smoke_query.yaml` | `…/datafabric_connector/smoke_query/` | SKIPPED (surface gap) after PASS it.3 (run 35490499651) and two smoke-gate fails (run 35674400362 attempts 1+2) | the curated Query Entity Records template has no sort-field parameter (only `isAscending`); Flow carried `_sortFieldName`. The passing runs used an invented `sortField` input or CEQL `ORDER BY`; without a sanctioned carrier the task is `skip: true`, criteria unchanged. | -| `connector_features/datafabric_connector/smoke_update.yaml` | `…/datafabric_connector/smoke_update/` | PASS (run 35499789502) | batch 2 | -| `connector_features/datafabric_connector/smoke_file_activities.yaml` | `…/datafabric_connector/smoke_file_activities/` | PASS (run 35499789502) | batch 2 | -| `connector_features/datafabric_connector/e2e_contract_intake_pipeline.yaml` | `…/datafabric_connector/e2e_contract_intake_pipeline/` | PASS (run 35499789502) | batch 2 | -| `connector_features/datafabric_connector/trigger_lifecycle.yaml` | `…/datafabric_connector/trigger_lifecycle/` | PASS it.3 (run 35501830119) after two grader fixes (it.1 was a real agent omission of the entity param; it.2 a grader over-strictness) | | -| `connector_features/testmanager_attachments/…` | `connector_features/testmanager_attachments/` | PASS (run 35499789502) | batch 2 | -| `connector_features/testmanager_execution_results/…` | `connector_features/testmanager_execution_results/` | PASS (run 35499789502) | batch 2 | -| `connector_features/testmanager_generic_records/…` | `connector_features/testmanager_generic_records/` | PASS (run 35499789502) | batch 2 | -| `connector_features/testmanager_requirement_lifecycle/…` | `connector_features/testmanager_requirement_lifecycle/` | PASS (run 35499789502) | batch 2 | -| `connector_features/testmanager_testset_lifecycle/…` | `connector_features/testmanager_testset_lifecycle/` | PASS (run 35499789502) | batch 2 | -| `connector_features/datafabric_connector/smoke_update_existing_flow.yaml` | `…/datafabric_connector/smoke_update_existing_flow/` | PASS (run 35500726138) | batch 3, brownfield scaffold via bpmn init | -| `connector_features/non-catalog-http-fallback/…` | `connector_features/non_catalog_http_fallback/` | PASS (run 35500726138) | batch 3 | -| `single_node/outlook_waitfor_email/…` | `single_node/outlook_waitfor_email/` | PASS (run 35500726138) | batch 3 | -| `single_node/outlook_trigger_inbox/…` | `single_node/outlook_trigger_inbox/` | PASS it.2 (run 35501830119); it.1 the agent omitted parentFolderId | advisory `uip is triggers` telemetry stays 0 (skill does not teach trigger discovery) | -| `e2e/devcon_expense_approval.yaml` | `e2e/devcon_expense_approval/` | PARTIAL 0.885 it.3 (run 35503094182) | every HITL/schema/wiring assertion passes; only `validate` fails (Actions.HITL MISSING_BINDING: needs a deployed Action App; Flow's inline quick-form has no tenant dependency). Iterations exhausted; parity-minus-validate, same platform gap as the two simulated HITL ports. | -| `e2e/jira_get_issue/…` | `e2e/jira_get_issue/` | PASS (run 35501830119) | LIVE pilot: ephemeral solution + bpmn debug + variables-all recipe works | -| `interactive/customer_escalation_simulated/…` | `interactive/customer_escalation_simulated/` | PASS 0.94 (run 35501830119) | only the advisory name check (threshold 0) missed, as in Flow | -| `interactive/expense_approval_simulated/…` | `interactive/expense_approval_simulated/` | PARTIAL 0.70 (run 35501830119) | everything passes except `validate`: Actions.HITL needs a deployed Action App binding (MISSING_BINDING with placeholder appId). Platform gap vs Flow's inline quick-form. | -| `interactive/hitl_schema_design_simulated/…` | `interactive/hitl_schema_design_simulated/` | PARTIAL 0.68 (run 35501830119) | same validate/Action App gap | -| `interactive/solution_select.yaml` | `interactive/solution_select/` | PARKED (skill gap, run 35501830119) | BPMN skill has no existing-solution selection rule; agent auto-scaffolded `WeatherAlertSolution/` without asking, exactly as predicted. Port kept in tree as the documented gap. | -| `e2e/jira_create_issue/…` | `e2e/jira_create_issue/` | PASS (run 35503094182) | live | -| `e2e/escalation_jira_ticket/…` | `e2e/escalation_jira_ticket/` | PASS (run 35503094182) | live | -| `e2e/escalation_orchestrator_paths/…` | `e2e/escalation_orchestrator_paths/` | PASS (run 35503094182) | live, 7 debug runs | -| `e2e/escalation_slack_alert/…` | `e2e/escalation_slack_alert/` | PASS it.2 (run 35524004307); it.1 the agent omitted the Slack folderKey binding (runtime 102010) | live | -| `e2e/jira_lifecycle/…` | `e2e/jira_lifecycle/` | PARKED after 3 iterations | it.1 our poll-cap bug; it.2 instance never terminal in 720s; it.3 (run 35525387843) `bpmn debug` exited 1 before creating an instance. Three different runtime failures of a multi-instance Jira loop; Flow's own version is flaky (0.82 typical, 2/12 zero). Needs a live investigation, not more retries. | -| `e2e/jira_search_triage/…` | `e2e/jira_search_triage/` | PARKED (skill gap) after 3 iterations | it.3 (run 35525387843) runtime 400008 "Failed to evaluate the input collection variable for the marker element": the multi-instance `inputCollection="=vars.Var_SearchResponse.issues"` over a connector response does not evaluate — multi-instance over connector output not taught/supported. | -| `multi_node/bellevue_weather/…` | `multi_node/bellevue_weather/` | PARKED (skill gap) after it.1+it.2 (runs 35523787101, 35525387843) | identical runtime fault both times: the script task reads `temperature_2m` off an undefined HTTP response — the skill does not teach the Intsvc.HttpExecution response shape well enough for downstream scripts. | -| `multi_node/slack_channel_description/…` | `multi_node/slack_channel_description/` | PASS it.2 (run 35525387843); it.1 the agent omitted the Slack channel parameter | live | -| `connector_trigger/trigger_with_filter.yaml` | — | PARKED (skill gap) | Flow asserts a structured `filter` tree (groupOperator + filters[], MST-8802 guard); BPMN `Intsvc.EventTrigger` declares `filter` only as an untyped object with no template placeholder and the skill says trigger properties are CLI-owned enrichment. Re-port once a persisted filter shape is documented. | -| `connector_features/datafabric_connector/smoke_error.yaml` | `…/datafabric_connector/smoke_error/` | PASS 1.0 (run 35538279757) | batch 10, structural | -| `connector_features/generic_dynamic_node/…` | `connector_features/generic_dynamic_node/` | PASS 1.0 (run 35538279757) | batch 10, live: ServiceNow acr_user list ran, empty array as expected | -| `connector_features/jdbc_databricks_query/…` | `connector_features/jdbc_databricks_query/` | PASS 1.0 (run 35538279757) | batch 10, structural | -| `multi_node/billing_invoice_lookup/…` | `multi_node/billing_invoice_lookup/` | PASS 0.91 it.1 (run 35538279757); grader fixed, not re-run | live: all three malformed inputs normalized and queried. Only the advisory `bindings` step failed, on the grader's own ephemeral live solution being read as a second project (fixed: `resolve_project(exclude_under=[LIVE_RUN_DIR])`). | -| `connector_features/testmanager_crud_grounded/…` | `connector_features/testmanager_crud_grounded/` | 0.89 it.1 (run 35538279757); grader fixed, not re-run | self-report round-trip verified live. The node-types criterion died on two byte-identical `.bpmn` (scaffold + solution copy); `find_bpmn_file` now treats identical copies as one artifact. Replays green on the CI artifact. | -| `connector_features/slack-http-fallback/…` | `connector_features/slack_http_fallback/` | 0.76 it.1 (run 35538279757); grader fixed, not re-run | live debug completed with no incidents. The fallback check looked for `emoji.list`; the Slack connector's generic resource for that endpoint is `emoji_list_GET`, which the agent used. Tolerance `emoji[._]list` added; replays green. | -| `connector_trigger/webhook_waitfor_parallel.yaml` | `connector_trigger/webhook_waitfor_parallel/` | 0.47 it.1 (run 35538279757); grader fixed, not re-run | agent emitted the wait as `bpmn:intermediateCatchEvent` + `Intsvc.WaitForEvent` (validates) and the GET as `Intsvc.UnifiedHttpRequest` (the skill lists it beside HttpExecution). Grader now classifies by wrapper type and accepts both HTTP types; replays green. Advisory `uip is webhooks config` telemetry 0, as feared. | -| `multi_node/billing_discrepancy_detector/…` | `multi_node/billing_discrepancy_detector/` | FAIL 0.30 it.1 (run 35538279757) | validate passed; live debug raised an Integration Services 400 on `Task_QueryERP`: "Expected a field name expression but got 'StringValue'" (malformed Data Service query filter, agent authoring). Advisory: `accountTier` did not derive from the CRM query. Also hit the same `bindings` grader defect (fixed). One iteration left to spend when live work resumes. | -| `multi_node/slack_weather_pipeline/…` | `multi_node/slack_weather_pipeline/` | FAIL 0.375 it.1 (run 35538279757) | validate passed; live debug incident 300501 in script task `Task_SelectChannel`: "Slack channel office-bellevue was not found" — the agent's channel lookup did not find a channel Flow's port finds (likely list pagination/limit). Agent defect, not grader; retry when live work resumes. | -| `connector_features/slack-http-fallback/…` (batch 11) | `connector_features/slack_http_fallback/` | PASS 1.0 (run 35783045540) | confirmed after the emoji_list tolerance | -| `connector_trigger/webhook_waitfor_parallel.yaml` (batch 11) | `connector_trigger/webhook_waitfor_parallel/` | PASS 1.0 (run 35783045540) | confirmed after wrapper-type classification | -| `connector_features/testmanager_crud_grounded/…` (batch 11) | `connector_features/testmanager_crud_grounded/` | PASS 1.0 (run 35783045540) | confirmed after identical-copy tolerance | -| `interactive/cli_dice_roller_simulated/…` | `interactive/cli_dice_roller_simulated/` | PASS 1.0 first run (run 35783045540) | live, simulated user | -| `multi_node/billing_invoice_lookup/…` (batch 11) | `multi_node/billing_invoice_lookup/` | INFRA (run 35783045540): platform 504 on poll-instance-status; not counted | rerun in batch 12 | -| `multi_node/billing_discrepancy_detector/…` (it.2) | `multi_node/billing_discrepancy_detector/` | PARKED (skill gap) after it.2 (run 35783045540) | it.1 400 "Expected a field name expression but got 'StringValue'"; it.2 400 "Error parsing query: SELECT * FROM DUMMY WHERE `accountNumber`=" (variable interpolated as empty). Both: the skill does not teach how to build a Data Service where clause from a process variable. Flow passes 11/12 nightlies. | -| `connector_features/ceql_where.yaml` (it.1) | `connector_features/ceql_where/` | FAIL 0 it.1 (run 35783045540) | agent wrote the CEQL string `displayName='active'` plus a flat {field, operator, value} object, not the canonical tree (numeric groupOperator + filters[]) the prompt asks for. Iteration 2 in batch 12; if repeated, park: BPMN's sanctioned `where` carrier is a CEQL string. | -| `interactive/bellevue_weather_simulated/…` (it.1) | `interactive/bellevue_weather_simulated/` | HARNESS (run 35783045540): simulation stopped on turn 1 with an empty agent output (stop_token), no HTTP node built | rerun in batch 12; Flow's version passes 8/12 | -| `interactive/slack_channel_description_simulated/…` (it.1) | `interactive/slack_channel_description_simulated/` | FAIL 0.52 it.1 (run 35783045540) | runtime 400 channel_not_found on Get_Channel_Info (agent passed a channel the bot is not in). Flow's version fully passes 4/12 nightlies. Iteration 2 in batch 12. | -| `multi_node/slack_weather_pipeline/…` (it.2) | `multi_node/slack_weather_pipeline/` | FAIL 0 it.2 (run 35783045540) | validate MISSING_BINDING on the Slack node + runtime 401 "Invalid Organization or User secret" (wrong connection bound). it.1 was channel-not-found. Flow's version passes 6/12 nightlies. Iteration 3 in batch 12, then park. | -| `multi_node/billing_invoice_lookup/…` (batch 12) | `multi_node/billing_invoice_lookup/` | PASS 1.0 (run 35785806030) | live; the batch-11 504 was infra | -| `multi_node/slack_weather_pipeline/…` (it.3) | `multi_node/slack_weather_pipeline/` | PASS 1.0 it.3 (run 35785806030) | live; it.1 channel lookup, it.2 wrong connection — agent flakiness at Flow parity (Flow 6/12) | -| `interactive/bellevue_weather_simulated/…` (it.2) | `interactive/bellevue_weather_simulated/` | PARKED (skill gap) after it.2 (run 35785806030) | runtime "Cannot read property 'temperature_2m' of undefined" in the summarize script: same HttpExecution response-shape gap that parked multi_node/bellevue_weather. | -| `connector_features/ceql_where.yaml` (it.2) | `connector_features/ceql_where/` | PARKED (surface gap) after it.2 (run 35785806030) | both runs: agent writes the CEQL string `displayName='active'` (the connector's sanctioned `where` parameter) plus a flat object, never the numeric-groupOperator tree Flow's grader requires. The tree is a Flow-skill construct with no BPMN carrier. Verdict for the family: tree-shape assertions do not port; wire-parameter assertions (path, query, pagination, enum, multiselect, complex_array) do. | -| `interactive/slack_channel_description_simulated/…` (it.2) | `interactive/slack_channel_description_simulated/` | GRADER DEFECT it.2 (run 35785806030); fixed, it.3 in batch 13 | agent left an untyped draft under `Solution/` beside the real solution; `find_bpmn_file` without a hint saw two projects. It now drops candidates carrying no registry-typed node; replays to the real file. | -| `connector_features/enum.yaml` | `connector_features/enum/` | PASS 1.0 first run (run 35789221753) | field-shape family | -| `connector_features/query_params.yaml` | `connector_features/query_params/` | PASS 1.0 first run (run 35789221753) | field-shape family | -| `connector_features/multiselect.yaml` | `connector_features/multiselect/` | PASS 1.0 first run (run 35789221753) | field-shape family | -| `connector_features/searchable_joins.yaml` | `connector_features/searchable_joins/` | PASS 1.0 first run (run 35789221753) | field-shape family | -| `connector_features/complex_array.yaml` | `connector_features/complex_array/` | PASS 0.875 first run (run 35789221753) | only the advisory resolved-user-id check (threshold 0) missed; Flow's own version fully passes 3/12 | -| `connector_features/path_params.yaml` (it.1) | `connector_features/path_params/` | FAIL 0.83 it.1 (run 35789221753); it.2 in batch 14 | agent built the Jira GET as UnifiedHttpRequest with the issue key as an unbound variable, never the fixed ENGCE-00000 the prompt gives. Flow 12/12. | -| `connector_features/enhanced_enum.yaml` (it.1) | `connector_features/enhanced_enum/` | FAIL 0.5 it.1 (run 35789221753); it.2 in batch 14 | agent produced no connector node at all (only an enum-typed variable). Flow 12/12. | -| `connector_features/paginated_reference_lookup.yaml` (it.1) | `connector_features/paginated_reference_lookup/` | FAIL 0.28 it.1 (run 35789221753); it.2 in batch 14 | agent sent to channel "simple" by name, never resolved/paginated to C083AN4E61E; both discovery advisories 0. Flow 11/12: BPMN skill does not teach Slack channel-id resolution. | -| `interactive/slack_channel_description_simulated/…` (it.3) | `interactive/slack_channel_description_simulated/` | PARKED (skill gap) after it.3 (run 35789221753) | debug completed; the agent listed page 1 of conversations and scripted the description out of it, so office-bellevue never appeared. Same pagination gap as multi_node/slack_channel_description it.1. Flow's version fully passes 4/12. | -| `connector_features/path_params.yaml` (it.2) | `connector_features/path_params/` | PASS 1.0 it.2 (run 35790934047) | agent used the curated Jira get-issue node with issueId=ENGCE-00000 as a path input | -| `connector_features/enhanced_enum.yaml` (it.2) | `connector_features/enhanced_enum/` | PARKED (skill gap) after it.2 (run 35790934047) | both runs: no connector node in the artifact at all (types empty; only an enum-typed variable). The WooCommerce connector never gets a node in BPMN; Flow 12/12. | -| `connector_features/paginated_reference_lookup.yaml` (it.2) | `connector_features/paginated_reference_lookup/` | FAIL 0.76 it.2 (run 35790934047); it.3 in batch 15 (run 35791969905) | discovery advisories now pass; the send node carries channel "C083AN4E61", the resolved id with its last character dropped. Agent transcription error, not grader. | -| `connector_features/paginated_reference_lookup.yaml` (it.3) | `connector_features/paginated_reference_lookup/` | PASS 1.0 it.3 (run 35791969905) | discovery advisories and the resolved channel id all green | -| `connector_features/testmanager_testcase_lifecycle/…` | `connector_features/testmanager_testcase_lifecycle/` | PASS (run 35488848026) | Flow's skip:true not carried over | +| `e2e/jira_get_issue/…` | same | PASS (run 35501830119) | live pilot: ephemeral solution + `bpmn debug` + `variables-all` | +| `e2e/jira_create_issue/…` | same | PASS (run 35503094182) | | +| `e2e/escalation_jira_ticket/…` | same | PASS (run 35503094182) | | +| `e2e/escalation_orchestrator_paths/…` | same | PASS (run 35503094182) | 7 debug runs | +| `e2e/escalation_slack_alert/…` | same | PASS it.2 (run 35524004307) | it.1 agent omitted the Slack `folderKey` binding (102010) | +| `multi_node/slack_channel_description/…` | same | PASS it.2 (run 35525387843) | it.1 agent omitted the channel parameter | +| `connector_features/datafabric_connector/smoke_error.yaml` | `…/smoke_error/` | PASS (run 35538279757) | structural | +| `connector_features/generic_dynamic_node/…` | same | PASS (run 35538279757) | | +| `connector_features/jdbc_databricks_query/…` | same | PASS (run 35538279757) | structural | +| `connector_features/slack-http-fallback/…` | `connector_features/slack_http_fallback/` | PASS (run 35783045540) | it.1 grader wanted `emoji.list`; the connector's generic resource is `emoji_list_GET` | +| `connector_trigger/webhook_waitfor_parallel.yaml` | same | PASS (run 35783045540) | it.1 grader classified the wait by BPMN tag; agent emits `intermediateCatchEvent` + `Intsvc.WaitForEvent` and `Intsvc.UnifiedHttpRequest` | +| `connector_features/testmanager_crud_grounded/…` | same | PASS (run 35783045540) | it.1 two byte-identical `.bpmn` copies read as ambiguity | +| `interactive/cli_dice_roller_simulated/…` | same | PASS (run 35783045540) | | +| `multi_node/billing_invoice_lookup/…` | same | PASS (run 35785806030) | it.1 grader read its own ephemeral live solution as a second project; run 35783045540 was a platform 504 | +| `multi_node/slack_weather_pipeline/…` | same | PASS it.3 (run 35785806030) | it.1 channel not found, it.2 wrong Slack connection bound (401); Flow passes 6/12 nightlies | +| `connector_features/enum.yaml` | `connector_features/enum/` | PASS (run 35789221753) | | +| `connector_features/query_params.yaml` | `connector_features/query_params/` | PASS (run 35789221753) | | +| `connector_features/multiselect.yaml` | `connector_features/multiselect/` | PASS (run 35789221753) | | +| `connector_features/searchable_joins.yaml` | `connector_features/searchable_joins/` | PASS (run 35789221753) | | +| `connector_features/complex_array.yaml` | `connector_features/complex_array/` | PASS 0.875 (run 35789221753) | only the advisory resolved-user-id check missed; Flow fully passes 3/12 | +| `connector_features/path_params.yaml` | `connector_features/path_params/` | PASS it.2 (run 35790934047) | it.1 agent left the issue key as an unbound variable | +| `connector_features/paginated_reference_lookup.yaml` | `connector_features/paginated_reference_lookup/` | PASS it.3 (run 35791969905) | it.1 channel by name, no pagination; it.2 channel id with its last character dropped | + +### Parked (branch `test/bpmn-port-parked`, all `skip: true`) + +| Flow task | BPMN port | Evidence | +|---|---|---| +| `multi_node/bellevue_weather/…` | same | runs 35523787101, 35525387843: script reads `temperature_2m` off an undefined `Intsvc.HttpExecution` response; response shape not taught | +| `interactive/bellevue_weather_simulated/…` | same | run 35785806030: same fault; run 35783045540 was a harness stop on turn 1 | +| `e2e/jira_search_triage/…` | same | run 35525387843: 400008, multi-instance over a connector response does not evaluate | +| `e2e/jira_lifecycle/…` | same | three different runtime failures (runs 35503094182, 35524004307, 35525387843); Flow flaky (0.82 typical) | +| `multi_node/billing_discrepancy_detector/…` | same | runs 35538279757, 35783045540: two different malformed Data Service where clauses built from a process variable | +| `interactive/slack_channel_description_simulated/…` | same | run 35789221753: page 1 of conversations only, target channel never appears; it.2 (run 35785806030) was a grader defect since fixed; Flow fully passes 4/12 | +| `connector_features/ceql_where.yaml` | `connector_features/ceql_where/` | runs 35783045540, 35785806030: agent writes the connector's CEQL `where` string, never Flow's numeric-groupOperator tree; no BPMN carrier | +| `connector_features/enhanced_enum.yaml` | `connector_features/enhanced_enum/` | runs 35789221753, 35790934047: no WooCommerce connector node produced | ## Ported 1:1 (21) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py deleted file mode 100644 index 37c2f5fa85..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_discrepancy_detector.py +++ /dev/null @@ -1,780 +0,0 @@ -#!/usr/bin/env python3 -"""BillingDiscrepancyDetector (BPMN): detector / bindings / advisory checks. - -Ported from Flow `multi_node/billing_discrepancy_detector/`: same scenario -(two independent Data Service reads -- BillingDisputeERP and -BillingDisputeCRM -- fanned out from the trigger/start and rejoined at a -merge/join before an overcharge computation), same three graded scripts -(`check_billing_discrepancy_detector.py`, `_shared/check_bindings_no_stubs.py`, -`_shared/advisory_billing_discrepancy_detector.py`), collapsed here into ONE -module with a `detector` / `bindings` / `advisory` dispatch (per -PORTING-BRIEF: "no new shared utility module" for this batch) so the three -YAML criteria share one script, invoked as: - - check_billing_discrepancy_detector.py detector (live; weight 5.0) - check_billing_discrepancy_detector.py bindings (advisory; weight 1.0) - check_billing_discrepancy_detector.py advisory (advisory; weight 1.0) - -Translated from a JSON node/edge walk to an XML walk over the registry-driven -`Intsvc.ActivityExecution` connector shell (registry-workflow.md §3-4) and the -BPMN live-debug surface (`_shared/bpmn_live.py`, LIVE-ADDENDUM's canonical -pattern: ephemeral solution import, `bpmn debug`, `debug-instance -variables-all`/`incidents`). - -Assertion map (Flow -> BPMN): - F check_billing_discrepancy_detector.py:42 assert_flow_has_any_node_type(ENTITY_QUERY_HINTS) - -> detector(): query_entity_nodes() non-empty - F check_billing_discrepancy_detector.py:43 assert_flow_has_node_type(["core.logic.merge"]) - -> detector(): find_join_gateways() non-empty - F check_billing_discrepancy_detector.py:46 run_debug(inputs=INPUTS, timeout=240) - -> detector(): LIVE-ADDENDUM canonical pattern (ephemeral - solution init + import + sha256 pin + bpmn_live.run_debug + - debug-instance variables-all/incidents) - F check_billing_discrepancy_detector.py:48-51 assert_output_value(payload, 1610/1/"MCS-2026-04872"/"Enterprise") - -> detector(): assert_output_value() over collect_output_leaves() - (root scope Globals + every element's Outputs, per - LIVE-ADDENDUM: a root public output has read back null) - F advisory_billing_discrepancy_detector.py:71-89 two entity-read nodes, one per entity (ERP/CRM) - -> advisory(): query_entity_nodes() + entity_of() - F advisory_billing_discrepancy_detector.py:91-101 exactly one merge, bpmn:ParallelGateway join - -> advisory(): find_join_gateways() (exactly one gateway with - >=2 incoming flows) - F advisory_billing_discrepancy_detector.py:102-107 merge fed by >=2 distinct sources, continues downstream - -> advisory(): distinct incoming sourceRefs + outgoing flow check - F advisory_billing_discrepancy_detector.py:108-110 fork exists (some node has >=2 outgoing edges) - -> advisory(): has_fork() (generic -- not pinned to a gateway - type, matching Flow's own generic node-degree check) - F advisory_billing_discrepancy_detector.py:112-144 the two queries are MUTUALLY UNREACHABLE, both reachable - from the (single) trigger - -> advisory(): graph.reachable()/reaches_blocked() -- an F use - of `_shared/graph.py` `reaches`, per PORTING-BRIEF, because - Flow asserts this unreachability itself - F advisory_billing_discrepancy_detector.py:146-151 both filters computed, each from its own input - (ERP<-invoiceNumber, CRM<-accountNumber) - -> advisory(): reads_field() over the variable-derivation graph - F advisory_billing_discrepancy_detector.py:153-160 no answer literal (1610/2590/4200/Enterprise) - -> advisory(): carries_literal() (scoped to bpmn:process, - excluding bpmndi diagram coordinates) - F advisory_billing_discrepancy_detector.py:162-173 declared contract: in-globals present, out-globals present - with the contract's type - -> advisory(): declared_id() presence + declared type checks - F advisory_billing_discrepancy_detector.py:175-195 each output sourced from its own side (tier<-CRM, - matchedInvoiceNumber<-ERP, both numbers<-ERP) - -> advisory(): sourced_from() over the variable-derivation graph - F advisory_billing_discrepancy_detector.py:197-198 each read resolves to the tenant it is pointed at - (connection/folder PAIR of distinct real uuids) - -> advisory(): assert_connection_resolves() over the declared - block - F check_bindings_no_stubs.py:91-142 bindings*.json Connection resources are non-stub, and a - connector node without one fails - -> bindings(): packed bindings_v2.json Connection resources - (mirrors e2e/customer_escalation_triage/ - check_customer_escalation_package.py) - I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) - -> bpmn_check.find_bpmn_file()/parse_bpmn()/resolve_project() - I ephemeral solution init + `solution projects import` + sha256 pin of the imported bytes -- `bpmn debug` - runs against an imported project, unlike `flow debug`, which runs directly against the discovered project - -> LIVE-ADDENDUM canonical pattern (mirrors - e2e/jira_get_issue/_shared/check_jira_get_issue.py) - I `uip maestro bpmn pack` + zip read of bindings_v2.json (BPMN has no bare generated bindings.json outside a - pack/refresh -- registry-workflow.md §4) - -> bindings(): mirrors check_customer_escalation_package.py - T curated|generic entity-CRUD objectName classification (BATCH1-ADDENDUM) - -> query_entity_nodes() - T transitive variable derivation through BPMN.Variables copy tasks (Grading contract's allowed T list) - -> build_variable_graph()/trace() (var-id graph, name-matched, - not node-id/hop-bounded like Flow's $vars..output) - DROPPED require_no_private_connector_values / require_sequence_integrity / require_di_for_visible_elements - (not in Flow; the `bpmn validate` criterion covers structure) - -No native Data Fabric shape exists on the BPMN side of this port -(BATCH1-ADDENDUM: "Build this with the UiPath Data Service Integration Service -connector... for every entity operation" -- BPMN has no `core.datafabric.*` -analogue), so `entity_reads`'s NATIVE_READ branch from Flow's -advisory_flow_utils has no BPMN carrier and is not ported. -""" - -from __future__ import annotations - -import json -import os -import re -import subprocess -import sys -import tempfile -import xml.etree.ElementTree as ET -import zipfile -from collections import Counter, defaultdict -from pathlib import Path - -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 - -from _shared import bpmn_live # noqa: E402 -from _shared import graph # noqa: E402 -from _shared.bpmn_check import ( # noqa: E402 - NS, - attr, - elements, - fail, - find_bpmn_file, - parse_bpmn, - resolve_project, -) -from _shared.bpmn_live import ( # noqa: E402 - CheckFailure, - connector_context, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, -) - -NAME_HINT = "BillingDiscrepancyDetector" -CONNECTOR_KEY = "uipath-uipath-dataservice" -ACTIVITY_TYPE = "Intsvc.ActivityExecution" -# Query Entity Records curated/preview spellings (BATCH1-ADDENDUM). -QUERY_OBJECT_NAMES = {"queryentityrecordscurated", "queryentityrecords_v3"} -ERP = "BillingDisputeERP" -CRM = "BillingDisputeCRM" -FORBIDDEN_LITERALS = ["1610", "2590", "4200", "Enterprise"] -OUT_CONTRACT = { - "totalOvercharge": "number", - "discrepancyCount": "number", - "matchedInvoiceNumber": "string", - "accountTier": "string", -} -IN_CONTRACT = [ - "invoiceNumber", - "accountNumber", - "disputedLineNumber", - "disputedUnitPrice", - "disputedQuantity", -] - -UUID_RE = re.compile( - r"^[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}$" -) -STUB_UUID_RE = re.compile(r"^0{8}-0{4}-0{4}-0{4}-") -VAR_REF = re.compile(r"vars\.([A-Za-z_][\w]*)") - - -def is_real_uuid(value) -> bool: - rendered = str(value or "").strip() - return bool(UUID_RE.fullmatch(rendered)) and not STUB_UUID_RE.match(rendered) - - -# ── connector-activity XML helpers (BATCH1-ADDENDUM shapes) ────────────────── - - -def has_type(el: ET.Element, token: str) -> bool: - return token in ET.tostring(el, encoding="unicode") - - -def activity_root(task: ET.Element) -> ET.Element | None: - return task.find(".//uipath:activity", NS) - - -def all_inputs(task: ET.Element) -> list[ET.Element]: - root = activity_root(task) - if root is None: - return [] - return root.findall(".//uipath:input", NS) - - -def input_val(inp: ET.Element) -> str: - return inp.attrib.get("value") or (inp.text or "") - - -def context_value(task: ET.Element, name: str) -> str: - for inp in all_inputs(task): - if inp.attrib.get("name") == name: - return input_val(inp) - return "" - - -def query_entity_nodes(root: ET.Element) -> list[ET.Element]: - """Every sendTask carrying a Query Entity Records op, curated OR generic.""" - nodes = [] - for task in elements(root, "sendTask"): - if not has_type(task, ACTIVITY_TYPE): - continue - if connector_context(task).get("connectorKey") != CONNECTOR_KEY: - continue - object_name_l = context_value(task, "objectName").strip().lower() - operation_l = context_value(task, "operation").strip().lower() - method_u = context_value(task, "method").strip().upper() - is_curated = object_name_l in QUERY_OBJECT_NAMES - is_generic = object_name_l in {ERP.lower(), CRM.lower()} and ( - operation_l in ("list", "retrieve") or method_u == "GET" - ) - if is_curated or is_generic: - nodes.append(task) - return nodes - - -def entity_of(task: ET.Element) -> str | None: - """The entity a query node addresses (BATCH1-ADDENDUM: any input value or - the context path -- the skill does not pin where entityName lands).""" - path_value = context_value(task, "path").lower() - input_values = {input_val(inp).strip().lower() for inp in all_inputs(task)} - matches = [ - entity - for entity in (ERP, CRM) - if entity.lower() in input_values or entity.lower() in path_value - ] - return matches[0] if len(matches) == 1 else None - - -def find_join_gateways(root: ET.Element) -> list[tuple[ET.Element, list[ET.Element]]]: - """(gateway, incoming flows) for every bpmn:parallelGateway with >=2 incoming.""" - flows = elements(root, "sequenceFlow") - - def in_flows(node_id: str) -> list[ET.Element]: - return [f for f in flows if attr(f, "targetRef") == node_id] - - return [ - (gw, in_flows(attr(gw, "id"))) - for gw in elements(root, "parallelGateway") - if len(in_flows(attr(gw, "id"))) >= 2 - ] - - -def has_fork(root: ET.Element) -> bool: - """Any node with >=2 outgoing sequence flows (Flow's own generic check -- - not pinned to a gateway type).""" - counts = Counter(attr(f, "sourceRef") for f in elements(root, "sequenceFlow")) - return any(count >= 2 for count in counts.values()) - - -def reaches_blocked(root: ET.Element, source: str, target: str, blocked: str) -> bool: - return target in graph.reachable(root, source, blocked={blocked}) - - -# ── variable-derivation graph (var id -> name, var id -> derives-from ids) ─── - - -def build_variable_graph(root: ET.Element): - """(name_by_id, derives_from, var_written_by). - - `name_by_id` comes from every declared `` entry - (input/inputOutput/output). `derives_from`/`var_written_by` come from - every `` anywhere in the document - (a connector activity's own output, or a BPMN.Variables mapping's copy) -- - the T translation of Flow's node-id `$vars..output` dependency graph - onto BPMN's var-id addressing. - """ - name_by_id: dict[str, str] = {} - for var in root.findall(".//uipath:variables/*", NS): - var_id = var.attrib.get("id") - name = var.attrib.get("name") - if var_id and name: - name_by_id[var_id] = name.strip().lower() - - owner_tags = { - f"{{{graph.BPMN_NS}}}{tag}" - for tag in ( - "startEvent", - "endEvent", - "task", - "sendTask", - "receiveTask", - "serviceTask", - "scriptTask", - "userTask", - "businessRuleTask", - "subProcess", - "callActivity", - ) - } - parents = graph.parent_map(root) - - def owner_of(element: ET.Element) -> str | None: - node = element - while node in parents: - node = parents[node] - if node.tag in owner_tags and node.attrib.get("id"): - return node.attrib["id"] - return None - - derives_from: dict[str, set[str]] = defaultdict(set) - var_written_by: dict[str, set[str]] = defaultdict(set) - for out in root.iter(f"{{{NS['uipath']}}}output"): - var_id = out.attrib.get("var") - if not var_id: - continue - owner = owner_of(out) - if owner: - var_written_by[var_id].add(owner) - derives_from[var_id].update(VAR_REF.findall(out.attrib.get("source") or "")) - return name_by_id, derives_from, var_written_by - - -def trace(start_ids, derives_from, predicate) -> bool: - seen: set[str] = set() - stack = list(start_ids) - while stack: - current = stack.pop() - if current in seen: - continue - seen.add(current) - if predicate(current): - return True - stack.extend(derives_from.get(current, ())) - return False - - -def reads_field(task: ET.Element, field_lower: str, name_by_id, derives_from) -> bool: - refs: set[str] = set() - for inp in all_inputs(task): - refs.update(VAR_REF.findall(input_val(inp))) - return trace(refs, derives_from, lambda v: name_by_id.get(v) == field_lower) - - -def declared_id(root: ET.Element, tag: str, name_lower: str) -> str | None: - for var in root.findall(f".//uipath:variables/uipath:{tag}", NS): - if (var.attrib.get("name") or "").strip().lower() == name_lower: - return var.attrib.get("id") - return None - - -def declared_type(root: ET.Element, tag: str, var_id: str) -> str | None: - for var in root.findall(f".//uipath:variables/uipath:{tag}", NS): - if var.attrib.get("id") == var_id: - return var.attrib.get("type") - return None - - -def sourced_from(root, output_name, task_id, derives_from, var_written_by): - out_id = declared_id(root, "output", output_name.lower()) - if not out_id: - return False, None - return trace({out_id}, derives_from, lambda v: task_id in var_written_by.get(v, set())), out_id - - -def carries_literal(process_el: ET.Element, forbidden: str) -> bool: - """Search only the process subtree (excludes bpmndi diagram coordinates, - which can coincidentally collide with a small forbidden number).""" - token = re.compile(rf"(? dict[str, ET.Element]: - return { - b.attrib.get("id"): b - for b in root.findall(".//uipath:bindings/uipath:binding", NS) - if b.attrib.get("id") - } - - -def resolve_binding_ref(bindings_by_id: dict[str, ET.Element], ref: str) -> ET.Element | None: - match = re.fullmatch(r"=bindings\.([\w-]+)", (ref or "").strip()) - return bindings_by_id.get(match.group(1)) if match else None - - -def assert_connection_resolves(task: ET.Element, label: str, bindings_by_id) -> str: - """F advisory_billing_discrepancy_detector.py:197-198: connection/folder - PAIR of distinct real uuids -- the connector-shape half of - `assert_read_resolves` (the native half has no BPMN carrier, see module - docstring).""" - connection_ref = context_value(task, "connection") - conn_binding = resolve_binding_ref(bindings_by_id, connection_ref) - if conn_binding is None: - fail( - f"the {label} query's context 'connection' is {connection_ref!r}, which does not " - "resolve to a declared " - ) - conn_value = conn_binding.attrib.get("default") or conn_binding.attrib.get("resourceKey") - if not is_real_uuid(conn_value): - fail( - f"the {label} query's connection binding {conn_binding.attrib.get('id')!r} " - f"default/resourceKey is {conn_value!r}, not a real connection id" - ) - - folder_ref = context_value(task, "folderKey") - folder_binding = resolve_binding_ref(bindings_by_id, folder_ref) - if folder_binding is None: - fail( - f"the {label} query's context 'folderKey' is {folder_ref!r}, which does not resolve " - "to a declared (a folder-scoped connector activity needs a paired " - "folder binding -- registry-workflow.md §4)" - ) - folder_value = folder_binding.attrib.get("default") or folder_binding.attrib.get("resourceKey") - if not is_real_uuid(folder_value): - fail( - f"the {label} query's folder binding {folder_binding.attrib.get('id')!r} " - f"default/resourceKey is {folder_value!r}, not a real folder id" - ) - if conn_value == folder_value: - fail( - f"the {label} query's connection and folder bindings both resolve to the SAME uuid " - f"({conn_value}) -- the folder binding needs the connection's FOLDER key, not its id" - ) - return f"connection={conn_value[:8]}... folder={folder_value[:8]}..." - - -# ── advisory: STRUCTURAL shape (weight 1.0, pass_threshold 0.0) ───────────── - - -def advisory() -> None: - path, root = parse_bpmn(NAME_HINT) - process = root.find("bpmn:process", NS) - if process is None: - fail(f"{path}: no bpmn:process element found") - - # 1. two entity-read nodes, one per entity. - nodes = query_entity_nodes(root) - if len(nodes) != 2: - fail( - f"expected exactly TWO Data Service query-entity-records connector nodes " - f"(ERP + CRM), found {len(nodes)}" - ) - by_entity: dict[str, ET.Element] = {} - for task in nodes: - entity = entity_of(task) - if entity is None: - fail( - f"query node {attr(task, 'id')!r} does not clearly address exactly one of " - f"{ERP!r}/{CRM!r} in any input value or context path" - ) - if entity in by_entity: - fail(f"both query nodes address {entity!r}; the scenario needs one {ERP} and one {CRM}") - by_entity[entity] = task - missing = [e for e in (ERP, CRM) if e not in by_entity] - if missing: - fail(f"no query node addresses {missing}; entities queried: {sorted(by_entity)}") - erp, crm = by_entity[ERP], by_entity[CRM] - erp_id, crm_id = attr(erp, "id"), attr(crm, "id") - - # 2. exactly one join, a real join, continuing downstream; a fork exists. - joins = find_join_gateways(root) - if len(joins) != 1: - fail( - f"expected exactly one bpmn:parallelGateway acting as a join (>=2 incoming flows), " - f"found {len(joins)}" - ) - join_gw, incoming = joins[0] - join_id = attr(join_gw, "id") - sources = sorted({attr(f, "sourceRef") for f in incoming}) - if len(sources) < 2: - fail(f"join gateway {join_id!r} is fed by {len(sources)} distinct source(s) ({sources})") - if not [f for f in elements(root, "sequenceFlow") if attr(f, "sourceRef") == join_id]: - fail(f"join gateway {join_id!r} has no outgoing sequence flow; the joined path must continue") - if not has_fork(root): - fail("no fork found: no node has >=2 outgoing sequence flows, so nothing fans out before the join") - - # 3. the two queries are on MUTUALLY UNREACHABLE branches, both reachable - # from the (single) start event. - if reaches_blocked(root, erp_id, crm_id, join_id) or reaches_blocked(root, crm_id, erp_id, join_id): - first, second = ( - (erp_id, crm_id) if reaches_blocked(root, erp_id, crm_id, join_id) else (crm_id, erp_id) - ) - fail( - f"{second!r} is downstream of {first!r} -- the two lookups are CHAINED with a join " - "bolted on, not a fan-out from the start event" - ) - starts = elements(root, "startEvent") - if len(starts) != 1: - fail(f"a process has exactly one root start event, found {len(starts)}") - start_id = attr(starts[0], "id") - reachable_from_start = graph.reachable(root, start_id) - for node_id in (erp_id, crm_id): - if node_id not in reachable_from_start: - fail(f"query node {node_id!r} is not reachable from the start event {start_id!r}") - - # 4. both filters computed, each from its own input. - name_by_id, derives_from, var_written_by = build_variable_graph(root) - for label, task, field in ((ERP, erp, "invoicenumber"), (CRM, crm, "accountnumber")): - if not reads_field(task, field, name_by_id, derives_from): - fail( - f"the {label} query node does not reference the {field} input (directly, or " - "transitively through a BPMN.Variables copy task) in any of its inputs" - ) - - # 5. none of the answers is written in. - for bad in FORBIDDEN_LITERALS: - if carries_literal(process, bad): - fail( - f"the process carries the literal {bad!r}. Every one of the answers " - f"({', '.join(FORBIDDEN_LITERALS)}) has to come from the tenant" - ) - - # 6. the declared contract, and where each output comes from. - for name in IN_CONTRACT: - if declared_id(root, "input", name.lower()) is None: - fail(f"the process declares no public input named {name!r}") - for name, want in OUT_CONTRACT.items(): - out_id = declared_id(root, "output", name.lower()) - if out_id is None: - fail(f"the process declares no public output named {name!r}") - got = declared_type(root, "output", out_id) - if got != want: - fail(f"output {name!r} is declared type {got!r}; the contract asks for {want!r}") - - ok, out_id = sourced_from(root, "accountTier", crm_id, derives_from, var_written_by) - if not ok: - fail(f"output 'accountTier' (var {out_id!r}) does not derive from {crm_id!r} -- the tier comes from {CRM}") - ok, out_id = sourced_from(root, "matchedInvoiceNumber", erp_id, derives_from, var_written_by) - if not ok: - fail( - f"output 'matchedInvoiceNumber' (var {out_id!r}) does not derive from {erp_id!r} -- " - f"the matched invoice comes from {ERP}" - ) - for name in ("totalOvercharge", "discrepancyCount"): - ok, out_id = sourced_from(root, name, erp_id, derives_from, var_written_by) - if not ok: - fail( - f"output {name!r} (var {out_id!r}) does not derive from {erp_id!r} -- the contracted " - f"amount it is computed from lives in the {ERP} rows" - ) - - # 7. each read resolves to the tenant it is pointed at. - bindings_by_id = declared_bindings(root) - resolutions = [ - assert_connection_resolves(task, label, bindings_by_id) - for label, task in ((ERP, erp), (CRM, crm)) - ] - - print( - f"OK: {path} -- {ERP}={erp_id!r} and {CRM}={crm_id!r} on mutually unreachable branches from " - f"{start_id!r}, converging on parallelGateway join {join_id!r} ({len(sources)} sources, " - "continues downstream); both filters computed from their own inputs; no answer literals; " - f"outputs {sorted(OUT_CONTRACT)} each sourced from its own side; {'; '.join(resolutions)}" - ) - - -# ── bindings: no-stub advisory (weight 1.0, pass_threshold 0.0) ───────────── - - -def bindings() -> None: - bpmn_path = find_bpmn_file(NAME_HINT) - root = ET.parse(bpmn_path).getroot() - needs_connection = any(connector_context(node).get("connectorKey") for node in root.iter()) - - project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) - with tempfile.TemporaryDirectory(prefix="billing-pack-") as out_dir: - packed = subprocess.run( - ["uip", "maestro", "bpmn", "pack", str(project_dir), out_dir, "--output", "json"], - capture_output=True, - text=True, - timeout=90, - ) - if packed.returncode != 0: - fail(f"uip maestro bpmn pack failed (exit {packed.returncode}): {packed.stdout}\n{packed.stderr}") - try: - payload = bpmn_live.parse_json_output(packed.stdout, "pack") - except CheckFailure as error: - fail(str(error)) - if not isinstance(payload, dict) or str(payload.get("Result", "")).casefold() != "success": - fail(f"pack JSON did not report Success: {payload}") - - packages = list(Path(out_dir).glob("*.nupkg")) - if len(packages) != 1: - fail(f"expected exactly one .nupkg, found: {[p.name for p in packages]}") - with zipfile.ZipFile(packages[0]) as archive: - by_basename = {Path(name).name: name for name in archive.namelist()} - if "bindings_v2.json" not in by_basename: - if needs_connection: - fail( - "packed archive has no bindings_v2.json, but the process carries a " - "connector node that needs a connection binding" - ) - print("no bindings_v2.json, and no connector node that needs one") - return - bindings_doc = json.loads(archive.read(by_basename["bindings_v2.json"])) - - resources = bindings_doc.get("resources") - if not isinstance(resources, list): - fail(f"bindings_v2.json has no resources array: {bindings_doc}") - connections = [r for r in resources if isinstance(r, dict) and r.get("resource") == "Connection"] - if not connections: - if needs_connection: - fail("bindings_v2.json declares no Connection resources, but the process carries a connector node") - print("no Connection resources, and no connector node that needs one") - return - stubbed = [r.get("key") for r in connections if not is_real_uuid(r.get("key"))] - if stubbed: - fail(f"bindings_v2.json Connection keys must be real connection ids, not unresolved stubs: {stubbed}") - print(f"{len(connections)} connection binding(s) in bindings_v2.json, all populated with non-stub values") - - -# ── detector: live run (weight 5.0) ───────────────────────────────────────── - -INPUTS = { - "invoiceNumber": "MCS-2026-04872", - "accountNumber": "ACCT-98201-NE", - "disputedLineNumber": 5, - "disputedUnitPrice": 300, - "disputedQuantity": 14, -} -EXPECTED_OUTPUTS = (1610, 1, "MCS-2026-04872", "Enterprise") - -LIVE_RUN_DIR = Path("billing-discrepancy-detector-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} - -# Worst-case wall clock this checker's `detector` criterion can spend, priced -# the way _shared/test_criterion_budgets.py prices a run_debug(...) call: the -# call below passes no timeout/retries/backoff kwargs, so it prices at -# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). The -# surrounding CLI steps (solution init/import, variables-all, incidents) are -# not priced by that guard, so their sum is added by hand here, mirroring -# e2e/jira_get_issue and multi_node/slack_weather_pipeline: -# 90 (solution init) + 180 (solution import) + 480 (debug) -# + 120 (variables-all) + 120 (incidents) = 990 -# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 -# Flow's own criterion timeout (600) does not cover this; raised to 1050 in -# billing_discrepancy_detector.yaml (documented deviation, sanctioned by -# LIVE-ADDENDUM: the budget is a property of the CLI surface, not of what is -# graded). - - -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - -def collect_output_leaves(variables_data) -> list: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- mirrors Flow's own assert_output_value(), - which flattens the whole debug payload's declared outputs. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return leaves - - -def assert_output_value(leaves: list, expected) -> None: - """F check_billing_discrepancy_detector.py:48-51 (assert_output_value): - numeric exact match; string case-insensitive substring.""" - for value in leaves: - if value == expected: - return - if isinstance(expected, str) and isinstance(value, str) and expected.lower() in value.lower(): - return - raise CheckFailure(f"no output equals expected {expected!r}; outputs: {str(leaves)[:1000]}") - - -def detector() -> None: - bpmn_path = find_bpmn_file(NAME_HINT) - root = ET.parse(bpmn_path).getroot() - - if len(query_entity_nodes(root)) < 2: - fail( - "expected at least two Data Service query-entity-records connector nodes " - f"(found {len(query_entity_nodes(root))})" - ) - if not find_join_gateways(root): - fail("no bpmn:parallelGateway acts as a join (>=2 incoming flows)") - - project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "BillingDiscrepancyDetectorLiveEval" - initialized = run_cli(["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in {solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") - - print(f"debug inputs: {INPUTS}") - debug_data, instance_id = bpmn_live.run_debug(imported_project, INPUTS, LIVE_RUN_DIR / "debug.log") - print(f"OK: debug completed (instance {instance_id})") - - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure(f"final status was {final_status!r}" + ("; " + "; ".join(detail) if detail else "")) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print(f"OK: bpmn debug completed (FinalStatus={final_status}, no incidents)") - - leaves = collect_output_leaves(variables_data) - for expected in EXPECTED_OUTPUTS: - assert_output_value(leaves, expected) - print("OK: overcharge=1610, count=1, invoice MCS-2026-04872, tier Enterprise") - print("PASS: all BillingDiscrepancyDetector checks passed") - - -_MODES = {"detector": detector, "bindings": bindings, "advisory": advisory} - - -def main() -> None: - mode = sys.argv[1] if len(sys.argv) > 1 else "detector" - handler = _MODES.get(mode) - if handler is None: - sys.exit(f"FAIL: unknown mode {mode!r}; expected one of {sorted(_MODES)}") - handler() - - -if __name__ == "__main__": - try: - main() - except CheckFailure as error: - raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py deleted file mode 100644 index cc1f73bf73..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_ceql_where.py +++ /dev/null @@ -1,269 +0,0 @@ -#!/usr/bin/env python3 -"""CEQL where (BPMN): verify the agent's planned filter JSON in -``where_detail.json`` carries a canonical CEQL filter tree per -``skills/uipath-platform/references/integration-service/activities.md`` -— section "Filter Trees (CEQL)" — and that the .bpmn file references the -registered Microsoft Entra (Azure AD) connector with the List Groups -operation, plus a Terminate end event for routing. - -Ported from Flow `connector_features/ceql_where.yaml`'s -``check_ceql_where_flow.py``: same scenario (plan a structured CEQL filter -tree for Entra's List Groups operation; build a connector node + Terminate -routing), translated from a JSON node/edge walk to an XML walk over the -registry-driven ``Intsvc.ActivityExecution`` connector shell (see -skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) and a -Terminate end event (see skills/uipath-maestro-bpmn/references/ -structural-bpmn.md, "Terminate (end events only)"). - -Why we still grade ``where_detail.json`` and not the node's live inputs: - The prompt forbids live node/connection configuration (no tenant in the - sandbox), which is what would populate the enriched `where`/`queryExpression` - input on the `Intsvc.ActivityExecution` node (registry-workflow.md §3 — - enrichment requires a live `--connection-id`). `uip maestro bpmn validate` - accepts a connector node with an empty/draft body, so requiring a fully - enriched body here would test something the prompt forbids and the CLI - doesn't enforce. This is the same rationale the Flow grader documents, and - it holds identically for BPMN: `where_detail.json` is the artifact the - prompt asks the agent to plan, so that is the artifact we grade. - -Assertion map (Flow → BPMN): - F check_ceql_where_flow.py:116-133 where_detail.json filter-tree shape → _check_where_detail() (verbatim: format-agnostic JSON check) - F check_ceql_where_flow.py:167-173 CONNECTOR_KEY referenced in flow → connector_task(root, CONNECTOR_KEY) present - F check_ceql_where_flow.py:140-154 _is_groups_operation node match → _is_groups_operation(task) - F check_ceql_where_flow.py:182 assert_flow_has_node_type(["terminate"]) → _has_terminate_end_event(root) - I locate/parse .bpmn → parse_bpmn() - T curated vs generic connector form → _is_groups_operation() accepts objectName containing "group" (any - curated spelling) OR objectName=="groups" with method GET / operation - List/list-groups (the generic form; confirmed live against the - uipath-microsoft-azureactivedirectory connector's `groups` object, - whose only describable object name is "groups" — the curated name - "ListGroups" is never a valid --object-name on its own) - DROPPED require_no_private_connector_values (not in Flow) - DROPPED require_sequence_integrity (not in Flow; `bpmn validate` is not graded here either, matching Flow) - DROPPED require_di_for_visible_elements (not in Flow) - DROPPED connection-binding check (Flow never checked connections; node is expected to stay draft, no live tenant) -""" - -from __future__ import annotations - -import glob -import json -import os -import sys -import xml.etree.ElementTree as ET - -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) - -from _shared.bpmn_check import NS, elements, fail, parse_bpmn # noqa: E402 - -CONNECTOR_KEY = "uipath-microsoft-azureactivedirectory" -ACTIVITY_TYPE = "Intsvc.ActivityExecution" -WHERE_DETAIL_GLOB = "**/where_detail.json" -EXPECTED_FIELD = "displayname" -EXPECTED_VALUE = "active" - - -# --- where_detail.json filter-tree checks (verbatim from Flow's -# check_ceql_where_flow.py — this artifact is a standalone JSON planning file -# unrelated to the .flow/.bpmn format, so the check does not change at all) --- - - -def _walk(node): - """Yield every dict in a nested filter tree (groups + leaves).""" - if isinstance(node, dict): - yield node - for v in node.values(): - yield from _walk(v) - elif isinstance(node, list): - for item in node: - yield from _walk(item) - - -def _leaf_field(n: dict): - return n.get("id") or n.get("fieldName") or n.get("field") or n.get("name") - - -def _leaf_value(n: dict): - v = n.get("value") - if isinstance(v, dict): - return v.get("value") - return v - - -def _looks_like_filter_tree(node) -> bool: - """A canonical filter-tree dict carries a numeric ``groupOperator`` and a - list of ``filters``. Used to locate the tree regardless of the key the - agent stored it under (e.g. top-level ``filter``, ``filterTree``, or - nested under ``plannedDetail.filter``).""" - return ( - isinstance(node, dict) - and isinstance(node.get("groupOperator"), (int, float)) - and isinstance(node.get("filters"), list) - ) - - -def _find_filter_tree(plan): - """Return the first filter-tree-shaped dict found anywhere in ``plan``. - The prompt asks the agent to capture a filter for review but does not pin - the JSON key, so accept the tree under any key.""" - for node in _walk(plan): - if _looks_like_filter_tree(node): - return node - return None - - -def _assert_filter_tree_shape(tree, *, source: str) -> None: - """Per Filter Trees (CEQL) doc: structured tree with numeric - groupOperator (0 = And, 1 = Or), at least one leaf with PascalCase - operator referencing displayName='active'. Leaves use ``id`` (canonical) - or fall back to ``fieldName``/``field``/``name`` for older shapes.""" - if not isinstance(tree, dict): - sys.exit(f"FAIL: {source} must be a filter-tree object") - - if not isinstance(tree.get("groupOperator"), (int, float)): - sys.exit( - f"FAIL: {source}.groupOperator must be a number " - "(0 = And, 1 = Or) — see Filter Trees (CEQL) doc" - ) - - filters = tree.get("filters") - if not isinstance(filters, list) or not filters: - sys.exit(f"FAIL: {source}.filters must be a non-empty list") - - leaves = [n for n in _walk(tree) if isinstance(n.get("operator"), str)] - if not leaves: - sys.exit(f"FAIL: {source} has no leaf filter with `operator`") - - fields = [_leaf_field(n) for n in leaves] - if not any(isinstance(f, str) and EXPECTED_FIELD in f.lower() for f in fields): - sys.exit( - f"FAIL: {source} leaves do not reference the displayName field " - f"(found fields: {[f for f in fields if f]})" - ) - - values = [_leaf_value(n) for n in leaves] - if not any(isinstance(v, str) and v.strip().lower() == EXPECTED_VALUE for v in values): - sys.exit( - f"FAIL: {source} has no leaf with value '{EXPECTED_VALUE}' " - f"(found values: {[v for v in values if v is not None]})" - ) - - -def _check_where_detail() -> None: - matches = glob.glob(WHERE_DETAIL_GLOB, recursive=True) - if not os.path.exists("where_detail.json") and not matches: - sys.exit("FAIL: where_detail.json not found") - path = "where_detail.json" if os.path.exists("where_detail.json") else matches[0] - try: - plan = json.load(open(path)) - except json.JSONDecodeError as e: - sys.exit(f"FAIL: {path} is not valid JSON: {e}") - - filter_tree = _find_filter_tree(plan) - if filter_tree is None: - sys.exit( - "FAIL: where_detail.json has no filter-tree object (a dict with a " - "numeric `groupOperator` and a `filters` list) under any key — " - "the prompt requires a structured CEQL filter tree" - ) - _assert_filter_tree_shape(filter_tree, source="where_detail.json filter tree") - - -# --- .bpmn structural checks --- - - -def context_inputs(task: ET.Element) -> list[ET.Element]: - return task.findall(".//uipath:input", NS) - - -def context_value(task: ET.Element, name: str) -> str: - for inp in context_inputs(task): - if inp.attrib.get("name") == name: - return inp.attrib.get("value") or (inp.text or "") - return "" - - -def has_type(el: ET.Element, token: str) -> bool: - return token in ET.tostring(el, encoding="unicode") - - -def connector_task(root: ET.Element, connector_key: str) -> ET.Element | None: - for task in elements(root, "sendTask"): - if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == connector_key: - return task - return None - - -def _is_groups_operation(task: ET.Element) -> bool: - """Match either connector form (see BATCH1-ADDENDUM "Lessons from batch 1 - CI"): a curated objectName that names the group (contains "group"), or - the generic object form — objectName=="groups" with method GET / an - operation naming List — confirmed live against the connector's - enrichment (`uip is activities list uipath-microsoft-azureactivedirectory - --output json`: ListGroups -> ObjectName "groups", MethodName "GET"; - `uip maestro bpmn registry get Intsvc.ActivityExecution --object-name - groups --operation List` -> Operation.Name "List", Curated "List Groups"). - """ - if context_value(task, "connectorKey") != CONNECTOR_KEY: - return False - - object_name = (context_value(task, "objectName") or "").lower() - if "group" in object_name: - return True - - method = (context_value(task, "method") or "").upper() - operation = (context_value(task, "operation") or "").lower() - path = (context_value(task, "path") or "").rstrip("/").lower() - return ( - object_name == "groups" - and (method == "GET" or "list" in operation) - and (not path or path.endswith("/groups")) - ) - - -def _has_terminate_end_event(root: ET.Element) -> bool: - for end in elements(root, "endEvent"): - if end.find("bpmn:terminateEventDefinition", NS) is not None: - return True - return False - - -def _check_bpmn_structure() -> None: - path, root = parse_bpmn("CeqlWhereTest") - - task = connector_task(root, CONNECTOR_KEY) - if task is None: - fail( - f"BPMN does not reference the registered Azure AD / Entra connector " - f"key {CONNECTOR_KEY!r} on a bpmn:sendTask carrying {ACTIVITY_TYPE}. " - "Display names like 'Microsoft Entra' or 'Microsoft Entra ID' are " - "NOT registry keys — confirm the registered key with " - "`uip maestro bpmn registry search`." - ) - - if not _is_groups_operation(task): - fail( - f"connector sendTask does not target the List Groups operation " - f"(objectName={context_value(task, 'objectName')!r}, " - f"method={context_value(task, 'method')!r}, " - f"operation={context_value(task, 'operation')!r})" - ) - print(f"OK: {CONNECTOR_KEY} sendTask targets List Groups") - - if not _has_terminate_end_event(root): - fail("no bpmn:endEvent with bpmn:terminateEventDefinition found") - print(f"OK: {path} has a Terminate end event") - - -def main() -> None: - _check_where_detail() - _check_bpmn_structure() - print( - f"OK: where_detail.json carries canonical CEQL filter tree on " - f"displayName='{EXPECTED_VALUE}'; BPMN targets {CONNECTOR_KEY} " - "List Groups; Terminate end event present" - ) - - -if __name__ == "__main__": - main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py deleted file mode 100644 index 04b01514d1..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description_simulated.py +++ /dev/null @@ -1,287 +0,0 @@ -#!/usr/bin/env python3 -"""SlackChannelDescription (simulated, BPMN): structural + live checks. - -Ported from Flow `interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml` -(via `tests/tasks/uipath-maestro-flow/_shared/check_channel_description_simulated.py`, -itself byte-identical in its assertions to the retired non-simulated Flow -original -- see that file's own docstring). Same scenario as the non-simulated -BPMN sibling (`multi_node/slack_channel_description/_shared/check_channel_description.py`): -a manual-start process retrieves the channel description of #office-bellevue via -the Slack Integration Service connector and outputs it, driven here by a -simulated non-technical user who withholds the channel and project name until -asked. Identical live sequence and Slack-node classification to the non- -simulated sibling; the only difference from that sibling is BPMN-file discovery -staying name-agnostic (the simulated persona withholds the project name, so no -hint is available -- mirrors Flow's own `flow_check._find_project`, which never -filters by name either). - -Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per -check. Nothing is created against the tenant by this scenario (it only reads a -channel's description), so there is no side-effect record to tear down -- -matching both Flow graders, neither of which has a teardown. - -Assertion map (Flow -> BPMN): - F check_channel_description_simulated.py:27 assert_flow_uses_connector_target('uipath-salesforce-slack') - -> find_connector_nodes(): any element carrying - Intsvc.ActivityExecution whose connectorKey context field - equals uipath-salesforce-slack. No operation filter -- Flow's - own assertion has none either. - F check_channel_description_simulated.py:28 run_debug(timeout=240) implicitly requires finalStatus == - "Completed" (flow_check.run_debug raises on a non-Completed - status internally; `bpmn debug` returns only an instance id, so - the check is explicit here) - -> FinalStatus in COMPLETED_STATUSES and debug-instance - incidents is empty - F check_channel_description_simulated.py:29 assert_outputs_contain(payload, ADDRESS_FRAGMENTS, require_all=True) - -> every fragment found among the root scope's variable leaves - AND every element's Outputs (incl. nested connector `response`) - in `debug-instance variables-all` (LIVE-ADDENDUM: a root PUBLIC - OUTPUT has read back null even when mapped correctly, so the - search is not scoped to one declared output variable) - I locate/parse .bpmn, name-agnostic (file exists, well-formed XML, project directory - resolved -- no name hint, since the simulated persona withholds the project name - until asked and the grader must not assume the agent used it) - -> bpmn_check.find_bpmn_file()/resolve_project() - I ephemeral solution init + `solution projects import` + sha256 pin of the imported - bytes against the submitted file -- `bpmn debug` runs against an imported project, - unlike `flow debug`, which runs directly against the discovered project directory - -> LIVE-ADDENDUM canonical live pattern (mirrors the non- - simulated sibling and e2e/jira_get_issue's checker) - DROPPED the HTTP-proxy fallback branch of assert_flow_uses_connector_target (a - `core.action.http.v2` node with bodyParameters.targetConnector) -- a legacy Flow - accommodation for connector-backed flows authored before native connector node - types existed. The BPMN skill's registry enrichment always emits - Intsvc.ActivityExecution for a connector activity (registry-workflow.md §3), so no - analogous construct exists to translate. Same drop as the non-simulated sibling. - DROPPED require_no_private_connector_values / require_sequence_integrity / - require_di_for_visible_elements / connection-binding checks -- not in Flow; the - `bpmn validate` criterion covers structure. -""" - -from __future__ import annotations - -import json -import os -import sys -import xml.etree.ElementTree as ET -from pathlib import Path - -# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_channel_description.py -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 - -from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 -from _shared import bpmn_live # noqa: E402 -from _shared.bpmn_live import ( # noqa: E402 - CheckFailure, - connector_context, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, -) - -CONNECTOR_KEY = "uipath-salesforce-slack" -ACTIVITY_TYPE = "Intsvc.ActivityExecution" - -ADDRESS_FRAGMENTS = [ - "700 Bellevue Way NE", - "Suite 2000", - "Bellevue", - "WA 98004", -] - -LIVE_RUN_DIR = Path("slack-channel-description-simulated-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} - -# Worst-case wall clock this checker can spend, priced the way -# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug -# call below passes no timeout/retries/backoff kwargs, so it prices at -# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). -# The surrounding CLI steps (solution init/import, variables-all, incidents) -# are not priced by that guard, so their sum is added by hand here, exactly as -# the non-simulated sibling's checker does for the identical sequence, and the -# criterion `timeout:` in slack_channel_description_simulated.yaml documents -# the arithmetic: -# 90 (solution init) + 180 (solution import) + 480 (debug) -# + 120 (variables-all) + 120 (incidents) = 990 -# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 -# Flow's own criterion timeout (600) does not fit the extra CLI steps a BPMN -# live grade needs (solution init/import, separate variables-all/incidents -# reads), so it is raised to 1050 -- the one sanctioned deviation from -# "criteria identical" (LIVE-ADDENDUM: a property of the CLI surface, not of -# what is graded). - - -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - -def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: - """Every element carrying an Intsvc.ActivityExecution targeting connector_key. - - No operation filter: Flow's own assertion (`assert_flow_uses_connector_target`) - only requires SOME connector node for the key, not a specific op, so this - mirrors that breadth. Scans every descendant, not a fixed tag list (registry - templates may emit a connector activity as sendTask, serviceTask, or a plain - task) -- mirrors bpmn_live.index_runtime_connectors' own scanning discipline. - """ - found = [] - for node in root.iter(): - context = connector_context(node) - if context.get("connectorKey") != connector_key: - continue - if ACTIVITY_TYPE not in ET.tostring(node, encoding="unicode"): - continue - found.append(node) - return found - - -def collect_output_haystack(variables_data: object) -> str: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- it mirrors Flow's own assert_outputs_contain(), - which flattens the whole outputs payload. Element Outputs include a - connector's nested `response` object, which _leaves() flattens along with - everything else. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return "\n".join(str(v) for v in leaves).lower() - - -def main() -> None: - # Name-agnostic: the simulated persona withholds the project name until - # asked, so this grader (like Flow's simulated grader) must not assume the - # agent used "SlackChannelDescription" -- unlike the non-simulated - # sibling, which pins a NAME_HINT. - bpmn_path = find_bpmn_file() - raw = Path(bpmn_path).read_text(encoding="utf-8") - if CONNECTOR_KEY not in raw: - _fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") - print(f"OK: bpmn references {CONNECTOR_KEY}") - - try: - root = ET.parse(bpmn_path).getroot() - except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") - - connector_nodes = find_connector_nodes(root, CONNECTOR_KEY) - if not connector_nodes: - _fail( - f"bpmn does not reference a {CONNECTOR_KEY} connector node " - f"({ACTIVITY_TYPE})" - ) - print(f"OK: bpmn references a {CONNECTOR_KEY} connector node") - - project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "SlackChannelDescriptionSimulatedLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") - - debug_data, instance_id = bpmn_live.run_debug( - imported_project, {}, LIVE_RUN_DIR / "debug.log" - ) - print(f"OK: debug completed (instance {instance_id})") - - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) - - haystack = collect_output_haystack(variables_data) - missing = [f for f in ADDRESS_FRAGMENTS if f.lower() not in haystack] - if missing: - _fail( - f"outputs missing address fragments {missing}; " - f"expected all of {ADDRESS_FRAGMENTS}\noutputs: {haystack[:1000]}" - ) - print("OK: bpmn outputs contain the Bellevue office address") - print("PASS: all SlackChannelDescription (simulated) checks passed") - - -if __name__ == "__main__": - try: - main() - except CheckFailure as error: - raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py deleted file mode 100644 index 345395306c..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_enhanced_enum.py +++ /dev/null @@ -1,100 +0,0 @@ -#!/usr/bin/env python3 -"""EnhancedEnumTest (BPMN): WooCommerce connector node presence. - -Ported from Flow `connector_features/enhanced_enum.yaml`'s ``flow_contains.py`` -criteria: same scenario (a WooCommerce "get product reviews" node exposes an -enhanced-enum sort-order field with friendly display labels rather than raw -codes), translated from a JSON node/edge substring search to an XML walk over -the registry-driven ``Intsvc.ActivityExecution`` connector shell (see -skills/uipath-maestro-bpmn/references/registry-workflow.md §3). This is an -offline authoring task -- Flow's own grader never asserted a live connection, -a specific sort value, or a validate pass, and neither does this port. - -Two subcommands (subcommand-dispatched, matching the sibling connector -graders in this suite): - - check_exists Flow's ``flow_contains.py --flow-name EnhancedEnumTest - '"nodes"' '"edges"'`` -- the .bpmn exists and is - well-formed XML. - check_connector Flow's ``flow_contains.py --flow-name EnhancedEnumTest - 'uipath-automattic-woocommerce'`` -- a connector node - references the WooCommerce connector key. - -Assertion map (Flow → BPMN): - F criterion 1 flow_contains --flow-name EnhancedEnumTest '"nodes"' '"edges"' - (Flow file exists and is valid JSON) - → check_exists(): parse_bpmn("EnhancedEnumTest") (well-formed - XML is the BPMN analog of valid JSON) - F criterion 2 flow_contains --flow-name EnhancedEnumTest - 'uipath-automattic-woocommerce' - → check_connector(): an Intsvc.ActivityExecution - bpmn:sendTask with connectorKey uipath-automattic-woocommerce - I locate/parse .bpmn (file exists, well-formed XML) - → parse_bpmn() - DROPPED require_no_private_connector_values / require_sequence_integrity - / require_di_for_visible_elements / connection-binding checks - (not in Flow; Flow has no `validate` criterion on this task - either, so none is added here) - -Note: the task's tag list keeps Flow's misspelled connector tag -`uipath-automaticc-woocommerce` verbatim (per the porting brief, tags are -carried unchanged); this grader checks the REAL connector key -`uipath-automattic-woocommerce`, matching what Flow's own grader checked. -""" - -from __future__ import annotations - -import os -import sys -import xml.etree.ElementTree as ET - -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) - -from _shared.bpmn_check import context_value, elements, fail, has_type, parse_bpmn # noqa: E402 - -NAME_HINT = "EnhancedEnumTest" -CONNECTOR_KEY = "uipath-automattic-woocommerce" -ACTIVITY_TYPE = "Intsvc.ActivityExecution" - - -def find_connector_node(root: ET.Element) -> ET.Element | None: - for task in elements(root, "sendTask"): - if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == CONNECTOR_KEY: - return task - return None - - -def check_exists() -> None: - path, _root = parse_bpmn(NAME_HINT) - print(f"OK: {path} exists and is well-formed XML") - - -def check_connector() -> None: - _path, root = parse_bpmn(NAME_HINT) - - node = find_connector_node(root) - if node is None: - fail( - f"no bpmn:sendTask carries {ACTIVITY_TYPE} with connectorKey " - f"{CONNECTOR_KEY!r}" - ) - print( - f"OK: {CONNECTOR_KEY} sendTask present " - f"(objectName={context_value(node, 'objectName')!r})" - ) - - -DISPATCH = { - "check_exists": check_exists, - "check_connector": check_connector, -} - - -def main() -> None: - if len(sys.argv) < 2 or sys.argv[1] not in DISPATCH: - sys.exit(f"usage: {sys.argv[0]} {{{'|'.join(DISPATCH)}}}") - DISPATCH[sys.argv[1]]() - - -if __name__ == "__main__": - main() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py deleted file mode 100644 index 365ac7bc92..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_lifecycle.py +++ /dev/null @@ -1,354 +0,0 @@ -#!/usr/bin/env python3 -"""JiraLifecycle (BPMN): structural + live + tenant checks for a -multi-instance-loop-and-gateway process. - -Ported from Flow `e2e/jira_lifecycle/_shared/check_jira_lifecycle.py`: same -scenario (a manual-start process iterates a seeded batch of issues and, per -item, creates a Jira issue then routes on the item's `priority` through a -branch node to a branch-specific Add-Comment), translated from a JSON node -walk + inline `flow debug` payload to an XML walk over the BPMN loop/gateway -constructs plus the BPMN live-debug surface (`_shared/bpmn_live.py`, per -LIVE-ADDENDUM's canonical pattern: ephemeral solution import, `bpmn debug`, -`debug-instance variables-all`/`incidents`). - -Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per -check. Every confirmed key is recorded to ``.created_keys`` as soon as it is -seen, so post_run teardown deletes it even if a later assertion fails. - -Assertion map (Flow -> BPMN): - F check_jira_lifecycle.py:75 JIRA_KEY not in raw or '"nodes"' not in raw - ('"nodes"' marker dropped -- XML has no JSON - "nodes" key, see I below) - -> JIRA_KEY not in raw text of the .bpmn - F check_jira_lifecycle.py:79 assert_flow_has_node_type(["core.logic.loop"]) - -> has_multi_instance_loop(root): a - bpmn:multiInstanceLoopCharacteristics element - is present anywhere in the process - F check_jira_lifecycle.py:80 assert_flow_has_any_node_type(["core.logic.switch", - "core.logic.decision"]) - -> has_conditional_gateway(root): a - bpmn:exclusiveGateway element together with - at least one bpmn:conditionExpression is - present anywhere in the process - F check_jira_lifecycle.py:84 run_debug(timeout=600) implicitly requires - finalStatus == "Completed" (flow_check.run_debug - raises on a non-Completed status internally; - `bpmn debug` returns only an instance id, so the - check is explicit here) - -> FinalStatus in COMPLETED_STATUSES and - debug-instance incidents is empty - F check_jira_lifecycle.py:90-92 cands = regex findall of the project's - issue-key pattern over get_last_debug_raw() - (the whole inline flow debug payload text) - -> same regex over json.dumps(variables_data) - (the whole debug-instance variables-all - payload text) -- the create nodes' - responses land there regardless of how the - process mapped its outputs, mirroring - Flow's own comment - F check_jira_lifecycle.py:108 marker not in comment_blob -> branch routed - the comment correctly - -> same check, same jira_is.get_issue() field - F check_jira_lifecycle.py:115 missing = [s for s in want_marker if s not in - found] -> the loop created every seeded issue - -> same check - I locate/parse .bpmn (file exists, well-formed XML, project - directory resolved) - -> bpmn_check.find_bpmn_file()/resolve_project() - I ephemeral solution init + `solution projects import` + sha256 - pin of the imported bytes against the submitted file -- - `bpmn debug` runs against an imported project, unlike - `flow debug`, which runs directly against the discovered - project directory - -> LIVE-ADDENDUM canonical live pattern - (mirrors - e2e/customer_escalation_triage/check_customer_escalation_behavior.py - and _shared/check_jira_get_issue.py) - I Flow imports the shared `_shared/jira_is.py` helper - (connection_id() / get_issue()); no BPMN equivalent shared - module exists yet and BATCH1-ADDENDUM asks graders not to - add a new one while several tasks port in parallel, so - the same two `uip is resources run` calls are inlined - below (byte-identical operation names/queries to the - task's own `_setup/jira_is.py`) - -> _connection_id() / _get_issue() below - DROPPED require_no_private_connector_values / require_sequence_integrity - / require_di_for_visible_elements / connection-binding checks - / per-node connector-operation classification -- not in - Flow (Flow's own structural check never classifies the - Create-Issue/Add-Comment node ops either, deferring - entirely to the LIVE tenant re-read); the `bpmn validate` - criterion covers structure -""" - -from __future__ import annotations - -import json -import os -import re -import subprocess -import sys -import xml.etree.ElementTree as ET -from pathlib import Path - -# …/uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 - -from _shared import bpmn_live # noqa: E402 -from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 -from _shared.bpmn_live import ( # noqa: E402 - CheckFailure, - get_ci, - incident_records, - payload_data, - run_cli, - sha256, -) - -JIRA_KEY = "uipath-atlassian-jira" -# Same tenant target as the task's own _setup/jira_is.py (and the pilot -# e2e/jira_get_issue/_setup/jira_is.py) -- a shared tenant fixture, not Flow -# vocabulary, kept verbatim per LIVE-ADDENDUM. -FOLDER_PATH = "Shared/uipath-maestro-flow" -CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" -NAME_HINT = "JiraLifecycle" - -LIVE_RUN_DIR = Path("jira-lifecycle-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} -# Worst-case wall clock this checker can spend, priced the way -# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug -# call below passes a literal timeout=720, so it prices at -# bpmn_live.debug_budget(720) == 720 (one attempt, no backoff). -# The surrounding CLI steps (solution init/import, variables-all, incidents) -# are not priced by that guard, so their sum is added by hand here and the -# criterion `timeout:` in jira_lifecycle.yaml documents the arithmetic: -# 90 (solution init) + 180 (solution import) + 720 (debug) -# + 120 (variables-all) + 120 (incidents) = 1230 -# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1290 -# Flow's own criterion timeout (1320) already covers this (30s to spare), so -# it is kept verbatim rather than raised. -# -# The 720 below is passed as a literal (not this comment's named constant) so -# test_criterion_budgets.py's static AST pricer -- which only recognizes -# literal timeout=/retries=/backoff_seconds= arguments on the run_debug(...) -# call itself -- can price it. - - -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - -def _local(tag: str) -> str: - return tag.rsplit("}", 1)[-1] - - -def has_multi_instance_loop(root: ET.Element) -> bool: - return any(_local(el.tag) == "multiInstanceLoopCharacteristics" for el in root.iter()) - - -def has_conditional_gateway(root: ET.Element) -> bool: - """A bpmn:exclusiveGateway with at least one bpmn:conditionExpression - anywhere in the process -- same any-of level of specificity as Flow's own - ``assert_flow_has_any_node_type(["core.logic.switch", "core.logic.decision"])``, - which likewise only checks node-type presence, not branch wiring.""" - has_gateway = any(_local(el.tag) == "exclusiveGateway" for el in root.iter()) - has_condition = any(_local(el.tag) == "conditionExpression" for el in root.iter()) - return has_gateway and has_condition - - -def _run(*args: str) -> dict: - out = subprocess.run( - ["uip", *args, "--output", "json"], - capture_output=True, text=True, timeout=120, - ).stdout - return json.loads(out) - - -def _connection_id() -> str: - folder_key = _run("or", "folders", "get", FOLDER_PATH)["Data"]["Key"] - conns = _run( - "is", "connections", "list", JIRA_KEY, "--folder-key", folder_key, "--refresh" - )["Data"] - return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) - - -def _get_issue(conn_id: str, key: str, project: str, issuetype_id: str) -> dict | None: - """Return the issue's `fields` dict (includes `summary` and `comment`), - or None if it doesn't exist (404).""" - env = _run( - "is", "resources", "run", "get", JIRA_KEY, "curated_get_issue", - "--connection-id", conn_id, - "--query", f"project={project}&issuetype={issuetype_id}&issueId={key}", - ) - if env.get("Result") == "Failure": - return None - return env["Data"].get("fields", {}) - - -def _record_key(key: str) -> None: - """Append a confirmed key to .created_keys (dedup) for teardown.""" - kf = Path(".created_keys") - seen = set(kf.read_text().split()) if kf.is_file() else set() - if key not in seen: - with kf.open("a") as f: - f.write(key + "\n") - - -def main() -> None: - seed_path = Path("seed.json") - if not seed_path.is_file(): - _fail("seed.json is missing from the sandbox (pre_run seed did not run)") - seed = json.loads(seed_path.read_text(encoding="utf-8")) - issues = seed["issues"] - project = seed["project_key"] - issuetype_id = seed["issuetype_id"] - # summary -> expected comment marker for that item's branch - want_marker = { - i["summary"]: (seed["escalated_marker"] if i["priority"] == "High" else seed["routine_marker"]) - for i in issues - } - - # 1. STRUCTURAL ---------------------------------------------------------- - bpmn_path = find_bpmn_file(NAME_HINT) - raw = Path(bpmn_path).read_text(encoding="utf-8") - if JIRA_KEY not in raw: - _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") - print(f"OK: bpmn references {JIRA_KEY}") - - try: - root = ET.parse(bpmn_path).getroot() - except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") - - if not has_multi_instance_loop(root): - _fail("bpmn does not contain a bpmn:multiInstanceLoopCharacteristics loop") - if not has_conditional_gateway(root): - _fail("bpmn does not contain a bpmn:exclusiveGateway with a conditionExpression") - print("OK: bpmn contains a multi-instance loop and a conditional exclusive gateway") - - # 2. LIVE ------------------------------------------------------------------ - project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "JiraLifecycleLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") - - debug_data, instance_id = bpmn_live.run_debug( - imported_project, {}, LIVE_RUN_DIR / "debug.log", timeout=720 - ) - print(f"OK: debug completed (instance {instance_id})") - - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) - - # Every CE- key that appears anywhere in the variables-all payload -- - # the create nodes' responses land in the runtime scopes/element outputs - # regardless of how the process mapped them, mirroring Flow's own - # whole-payload scan. - variables_text = json.dumps(variables_data) - cands = list(dict.fromkeys(re.findall(rf"\b{re.escape(project)}-\d+\b", variables_text))) - if not cands: - _fail(f"no issue key (e.g. {project}-123) in debug-instance variables-all -- the loop created nothing") - print(f"OK: candidate keys from debug: {cands}") - - # 3. TENANT ------------------------------------------------------------ - conn = _connection_id() - found: dict[str, str] = {} # summary -> key, for issues that are ours - for key in cands: - fields = _get_issue(conn, key, project, issuetype_id) - if not fields: - continue - _record_key(key) # real issue this run created -- always clean it up - summary = fields.get("summary") - if summary in want_marker: - found[summary] = key - marker = want_marker[summary] - comment_blob = json.dumps(fields.get("comment")) - if marker not in comment_blob: - _fail( - f"issue {key} ({summary!r}) is missing its expected branch " - f"comment {marker!r} -- the gateway routed it to the wrong " - f"branch (or no comment was posted)" - ) - - missing = [s for s in want_marker if s not in found] - if missing: - _fail( - f"the loop did not create every seeded issue -- missing {missing}; " - f"created and matched: {list(found.values())}" - ) - - print(f"OK: all {len(want_marker)} issues created and each carries its correct branch comment") - print("PASS: all JiraLifecycle checks passed") - - -if __name__ == "__main__": - try: - main() - except CheckFailure as error: - raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py deleted file mode 100644 index 5b5cd67bec..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_search_triage.py +++ /dev/null @@ -1,283 +0,0 @@ -#!/usr/bin/env python3 -"""JiraSearchTriage (BPMN): structural + live + tenant checks for a -JQL-search-driven triage process. - -Ported from Flow `e2e/jira_search_triage/_shared/check_jira_search_triage.py`: -same scenario (a manual-start process searches Jira by a seeded JQL and, for -each match, adds a triage comment via a loop), translated from a JSON node -walk + inline `flow debug` payload to an XML walk over the registry-driven -`Intsvc.ActivityExecution` connector shell (see -skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4) plus the -BPMN live-debug surface (`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical -pattern: ephemeral solution import, `bpmn debug`, `debug-instance incidents`). -The loop construct is `bpmn:multiInstanceLoopCharacteristics` (see -skills/uipath-maestro-bpmn/references/structural-bpmn.md), the BPMN -translation of Flow's `core.logic.loop` node type. - -Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per -check. The two issues are created/cleaned up by seed_jira.py / -teardown_jira.py; this check only reads the tenant through the process it -grades, never directly. - -Assertion map (Flow → BPMN): - F check_jira_search_triage.py:40 JIRA_KEY not in raw ('"nodes"' marker dropped -- XML has no JSON "nodes" key, see I below) - → JIRA_KEY not in raw text of the .bpmn - F check_jira_search_triage.py:43 SEARCH_OP_RE.search(raw) (Search-Issues-by-JQL op referenced) - → find_search_issue_nodes(): a sendTask carrying Intsvc.ActivityExecution - whose connectorKey is uipath-atlassian-jira and whose objectName/method - classify as Search Issues by JQL (curated `issue_search_get` OR a generic - GET node whose serialized XML contains a `jql` token -- catalog checked - via `uip is activities list uipath-atlassian-jira --output json`) - F check_jira_search_triage.py:45 assert_flow_has_node_type(["core.logic.loop"]) - → has_loop(): at least one bpmn:multiInstanceLoopCharacteristics element - (PORTING-BRIEF construct-translation table: Loop → multiInstanceLoopCharacteristics) - F check_jira_search_triage.py:48 run_debug(timeout=600) implicitly requires finalStatus == "Completed" - (flow_check.run_debug raises on a non-Completed status internally; - `bpmn debug` returns only an instance id, so the check is explicit here) - → FinalStatus in COMPLETED_STATUSES and debug-instance incidents is empty - F check_jira_search_triage.py:51-60 conn = jira_is.connection_id(); for key in issue_keys: re-read + assert - marker in fields["comment"] - → identical tenant re-read against the same jira_is.py helper, unchanged - I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) - → bpmn_check.find_bpmn_file()/resolve_project() - I ephemeral solution init + `solution projects import` + sha256 pin of the imported - bytes against the submitted file -- `bpmn debug` runs against an imported project, - unlike `flow debug`, which runs directly against the discovered project directory - → LIVE-ADDENDUM canonical live pattern (mirrors - e2e/jira_get_issue/check_jira_get_issue.py) - T curated OR generic entity-CRUD classification (BATCH1-ADDENDUM); GET/GETBYID - equivalence not needed here (search is GET-only), but the same dual-form tolerance - (curated objectName vs generic object+method) applies - → is_search_issue_node() - DROPPED require_no_private_connector_values / require_sequence_integrity / - require_di_for_visible_elements / connection-binding checks -- not in Flow; the - `bpmn validate` criterion covers structure. No structural check for the Add-Comment - node either -- Flow's own grader never asserts one (it proves the comment landed via - the tenant re-read instead), so none is added here. - -No output-value assertion is made (unlike the jira_get_issue port): Flow's own -grader never reads `flow debug`'s output payload for this task, only the -tenant re-read, so `debug-instance variables-all` is not called here. -""" - -from __future__ import annotations - -import json -import os -import re -import sys -import xml.etree.ElementTree as ET -from pathlib import Path - -HERE = os.path.dirname(os.path.abspath(__file__)) # …/uipath-maestro-bpmn/_shared -# …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_jira_get_issue.py -sys.path.insert(0, os.path.dirname(HERE)) # noqa: E402 -# This task's own _setup/ in the repo (not the sandbox mount -- this checker -# runs from $REFERENCE_DIR, so it reads jira_is.py straight off disk here, -# the same way check_customer_escalation_behavior.py reads its own HERE/_setup). -# No new _shared module is added (BATCH1-ADDENDUM: agents write in parallel, -# do not add new shared modules) -- this imports the verbatim per-task copy. -sys.path.insert(0, os.path.join(HERE, "..", "e2e", "jira_search_triage", "_setup")) # noqa: E402 - -from _shared.bpmn_check import elements, find_bpmn_file, resolve_project # noqa: E402 -from _shared import bpmn_live # noqa: E402 -from _shared.bpmn_live import ( # noqa: E402 - CheckFailure, - connector_context, - get_ci, - incident_records, - payload_data, - run_cli, - sha256, -) -import jira_is # noqa: E402 - -JIRA_KEY = "uipath-atlassian-jira" -SEARCH_OP_RE = re.compile(r"search[\s_-]?issues?|issue_search_get|search-issues-by-jql", re.IGNORECASE) -GENERIC_SEARCH_OBJECTS = {"issue", "issues"} -GENERIC_SEARCH_METHODS = {"GET"} -ACTIVITY_TYPE = "Intsvc.ActivityExecution" -NAME_HINT = "JiraSearchTriage" - -LIVE_RUN_DIR = Path("jira-search-triage-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} - -# Worst-case wall clock this checker can spend, priced the way -# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug -# call below passes no timeout/retries/backoff kwargs, so it prices at -# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). -# The surrounding CLI steps (solution init/import, incidents) and the tenant -# re-read through jira_is.py (whose `_run` helper hardcodes a 120s subprocess -# timeout per call) are not priced by that guard, so their sum is added by -# hand here and the criterion `timeout:` in jira_search_triage.yaml documents -# the arithmetic: -# 90 (solution init) + 180 (solution import) + 480 (debug) + 120 (incidents) -# + 240 (connection_id: 2 tenant calls @ up to 120s each) -# + 240 (2 x get_issue re-read @ up to 120s each, one per seeded issue) -# = 1350 + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1410 -# Flow's own criterion timeout (1320) does not cover the extra ephemeral- -# solution import + live tenant re-read plumbing this BPMN sequence needs -# beyond Flow's single inline `flow debug` call, so it is raised to 1440 -# (the one sanctioned deviation from "criteria identical" per LIVE-ADDENDUM's -# Budgets section -- a property of the CLI surface, not of what is graded). - - -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - -def is_search_issue_node(node_name: str, object_name: str, method: str, node_xml: str) -> bool: - """Curated (`issue_search_get`) OR generic (GET + a `jql` token anywhere - in the node) form. The `jql` token requirement keeps a generic GET node - from being mistaken for search: BATCH1-ADDENDUM's dual-form tolerance - classifies by objectName/method, but Search Issues by JQL has no distinct - generic object name of its own (unlike Get Issue's `issue`+GETBYID).""" - if SEARCH_OP_RE.search(object_name or "") or SEARCH_OP_RE.search(node_name or ""): - return True - return ( - (object_name or "").strip().lower() in GENERIC_SEARCH_OBJECTS - and (method or "").strip().upper() in GENERIC_SEARCH_METHODS - and "jql" in node_xml.lower() - ) - - -def find_search_issue_nodes(root: ET.Element) -> list[ET.Element]: - """Every element carrying an Intsvc.ActivityExecution Jira Search-Issues op. - - Scans every descendant, not a fixed tag list (registry templates may emit - a connector activity as sendTask, serviceTask, or a plain task) -- - mirrors bpmn_live.index_runtime_connectors' own scanning discipline. - """ - found = [] - for node in root.iter(): - context = connector_context(node) - if context.get("connectorKey") != JIRA_KEY: - continue - node_xml = ET.tostring(node, encoding="unicode") - if ACTIVITY_TYPE not in node_xml: - continue - node_name = node.attrib.get("name", "") - if is_search_issue_node(node_name, context.get("objectName", ""), context.get("method", ""), node_xml): - found.append(node) - return found - - -def has_loop(root: ET.Element) -> bool: - return bool(elements(root, "multiInstanceLoopCharacteristics")) - - -def main() -> None: - seed_path = Path("seed.json") - if not seed_path.is_file(): - _fail("seed.json is missing from the sandbox (pre_run seed did not run)") - seed = json.loads(seed_path.read_text(encoding="utf-8")) - - bpmn_path = find_bpmn_file(NAME_HINT) - raw = Path(bpmn_path).read_text(encoding="utf-8") - if JIRA_KEY not in raw: - _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") - print(f"OK: bpmn references {JIRA_KEY}") - - try: - root = ET.parse(bpmn_path).getroot() - except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") - - if not find_search_issue_nodes(root): - _fail("bpmn does not reference a Jira Search-Issues-by-JQL connector node (Intsvc.ActivityExecution)") - if not has_loop(root): - _fail("bpmn does not contain a multiInstanceLoopCharacteristics loop over the search results") - print("OK: bpmn references a JQL search op and a loop construct") - - project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "JiraSearchTriageLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") - - debug_data, instance_id = bpmn_live.run_debug( - imported_project, {}, LIVE_RUN_DIR / "debug.log" - ) - print(f"OK: debug completed (instance {instance_id})") - - final_status = get_ci(debug_data, "FinalStatus") - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) - - conn = jira_is.connection_id() - marker = seed["processed_comment"] - for key in seed["issue_keys"]: - fields = jira_is.get_issue(conn, key) - if not fields: - _fail(f"seeded issue {key} not found on re-read") - if marker not in json.dumps(fields.get("comment")): - _fail( - f"issue {key} is missing the triage comment {marker!r} — the " - "search-driven loop did not comment it" - ) - print(f"OK: all {len(seed['issue_keys'])} matched issues carry the triage comment") - print("PASS: all JiraSearchTriage checks passed") - - -if __name__ == "__main__": - try: - main() - except CheckFailure as error: - raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py deleted file mode 100644 index 2dd84e733c..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn.py +++ /dev/null @@ -1,275 +0,0 @@ -#!/usr/bin/env python3 -"""BellevueWeather (BPMN): a weather-HTTP node is present and live output -contains one branch message. - -Ported from Flow `multi_node/bellevue_weather/_shared` (via -`uipath-maestro-flow/_shared/check_weather_flow.py`): same scenario (fetch -today's Bellevue weather from Open-Meteo and branch the summary message on -temperature), translated from a JSON node-type scan + inline `flow debug` -payload to an XML scan over the registry-driven `Intsvc.HttpExecution` -managed-HTTP shell (see skills/uipath-maestro-bpmn/references/structural-bpmn.md, -references/registry-workflow.md) plus the BPMN live-debug surface -(`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical pattern: ephemeral -solution import, `bpmn debug`, `debug-instance variables-all`/`incidents`). -The canonical live grader this file's plumbing is modeled on is -`_shared/check_jira_get_issue.py`. - -Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per -check. - -Assertion map (Flow -> BPMN): - F check_weather_flow.py:20-21 assert_flow_has_any_node_type( - ["core.action.http", "custom-codereval-openmeteoapis"]) - -> any element carrying an Intsvc.HttpExecution - uipath:activity wrapper. BPMN has no curated - Open-Meteo Integration Service connector, so - only the managed-HTTP construct from - PORTING-BRIEF's construct-translation table - remains; the Flow grader's connector-fallback - branch has nothing to translate to. - F check_weather_flow.py:22 run_debug(timeout=240) implicitly requires - finalStatus == "Completed" (flow_check.run_debug - raises on a non-Completed status internally; - `bpmn debug` returns only an instance id, so the - check is explicit here) - -> FinalStatus in COMPLETED_STATUSES and - debug-instance incidents is empty - F check_weather_flow.py:23-24 assert_outputs_contain(payload, - ["nice day", "bring a jacket"], require_all=False) - -> either verdict string found among the root - scope's variable leaves AND every element's - Outputs in `debug-instance variables-all` - (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read - back null even when mapped correctly, so the - search is not scoped to one declared output - variable) - I locate/parse .bpmn (file exists, well-formed XML, project - directory resolved) -> bpmn_check.find_bpmn_file()/resolve_project() - I ephemeral solution init + `solution projects import` + sha256 - pin of the imported bytes against the submitted file -- - `bpmn debug` runs against an imported project, unlike - `flow debug`, which runs directly against the discovered - project directory -> LIVE-ADDENDUM canonical live pattern - (mirrors check_jira_get_issue.py) - DROPPED require_no_private_connector_values / require_sequence_integrity / - require_di_for_visible_elements / connection-binding checks -- - not in Flow; the `bpmn validate` criterion covers structure - -Flow's grader does not assert a Script node or a Decision node exists (only -the weather-API node type and the branch output are graded), so this checker -does not add a structural check for the exclusiveGateway either -- adding one -would be a BPMN-only requirement the Flow prompt/grader never had. -""" - -from __future__ import annotations - -import os -import sys -import xml.etree.ElementTree as ET -from pathlib import Path - -# .../uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 - -from _shared.bpmn_check import ( # noqa: E402 - find_bpmn_file, - has_typed_uipath_extension, - resolve_project, -) -from _shared import bpmn_live # noqa: E402 -from _shared.bpmn_live import ( # noqa: E402 - CheckFailure, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, -) - -BPMN_NS = "http://www.omg.org/spec/BPMN/20100524/MODEL" -# registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask; the eval agent emits either (CI run 35538279757). -ACTIVITY_TYPES = ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") -ACTIVITY_TYPE = ACTIVITY_TYPES[0] -NAME_HINT = "BellevueWeather" -VERDICTS = ("nice day", "bring a jacket") - -LIVE_RUN_DIR = Path("bellevue-weather-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} - -# Worst-case wall clock this checker can spend, priced the way -# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug -# call below passes no timeout/retries/backoff kwargs, so it prices at -# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). -# The surrounding CLI steps (solution init/import, variables-all, incidents) -# are not priced by that guard, so their sum is added by hand here and the -# criterion `timeout:` in bellevue_weather.yaml documents the arithmetic: -# 90 (solution init) + 180 (solution import) + 480 (debug) -# + 120 (variables-all) + 120 (incidents) = 990 -# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 -# Flow's own criterion timeout (600) does not cover this larger live-debug -# surface, so it is raised to 1050 -- the one sanctioned deviation -# LIVE-ADDENDUM allows, because the budget is a property of the CLI surface, -# not of what is graded. - - -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - -def find_http_execution_nodes(root: ET.Element) -> list[ET.Element]: - """Every element carrying an Intsvc.HttpExecution uipath:activity wrapper. - - Scans every descendant, not a fixed tag list (registry templates may emit - the managed-HTTP activity as sendTask, serviceTask, or a plain task) -- - mirrors bpmn_live.index_runtime_connectors' own scanning discipline. Skips - ``bpmn:extensionElements`` nodes themselves: ``has_typed_uipath_extension`` - matches an element whose OWN direct children include a matching - ``uipath:activity`` (the wrapper task) as well as an ``extensionElements`` - node (whose direct child literally is that ``uipath:activity``), which - would otherwise double-count every match once per node. - """ - return [ - el - for el in root.iter() - if el.tag != f"{{{BPMN_NS}}}extensionElements" - and any(has_typed_uipath_extension(el, "activity", t) for t in ACTIVITY_TYPES) - ] - - -def collect_output_haystack(variables_data: object) -> str: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- it mirrors Flow's own - assert_outputs_contain(), which flattens the whole outputs payload. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return "\n".join(str(v) for v in leaves).lower() - - -def main() -> None: - bpmn_path = find_bpmn_file(NAME_HINT) - - try: - root = ET.parse(bpmn_path).getroot() - except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") - - http_nodes = find_http_execution_nodes(root) - if not http_nodes: - _fail( - f"{bpmn_path} has no element carrying an {ACTIVITY_TYPE} uipath:activity " - "wrapper (no managed-HTTP weather node found)" - ) - print(f"OK: bpmn has a managed-HTTP node ({ACTIVITY_TYPE})") - - project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "BellevueWeatherLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") - - debug_data, instance_id = bpmn_live.run_debug( - imported_project, {}, LIVE_RUN_DIR / "debug.log" - ) - print(f"OK: debug completed (instance {instance_id})") - - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {incidents_list}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) - - haystack = collect_output_haystack(variables_data) - if not any(verdict in haystack for verdict in VERDICTS): - _fail( - f"outputs do not contain either verdict string {list(VERDICTS)!r}\n" - f"outputs: {haystack[:1000]}" - ) - print("OK: bpmn outputs contain a weather branch message") - print("PASS: all BellevueWeather checks passed") - - -if __name__ == "__main__": - try: - main() - except CheckFailure as error: - raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py deleted file mode 100644 index 7e02569e8e..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_weather_bpmn_simulated.py +++ /dev/null @@ -1,299 +0,0 @@ -#!/usr/bin/env python3 -"""BellevueWeather (BPMN, simulated): a weather-HTTP node is present and live -output contains one branch message. - -Name-agnostic sibling of ``_shared/check_weather_bpmn.py`` (the committed -non-simulated Bellevue port): the two graders make exactly the same -assertions, in the same order, over the same live-debug surface. The only -difference is that this one never pins the project/file name -- the -simulated persona (see -``uipath-maestro-flow/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml``) -withholds "BellevueWeather" until asked, so a correctly-built, differently -named process must still be gradable. This mirrors how Flow's own -``check_weather_flow_simulated.py`` relates to the retired non-simulated -``check_weather_flow.py``: "Identical assertions to the retired non-simulated -original ... Name-agnostic runtime checker for the simulated variant." - -Ported from Flow `_shared/check_weather_flow_simulated.py` (itself the -name-agnostic sibling of the retired Flow `multi_node/bellevue_weather/ -check_weather_flow.py`), translated the same way `_shared/check_weather_bpmn.py` -already translates the non-simulated pair: a JSON node-type scan + inline -`flow debug` payload becomes an XML scan over the registry-driven -`Intsvc.HttpExecution` managed-HTTP shell (see -skills/uipath-maestro-bpmn/references/structural-bpmn.md, -references/registry-workflow.md) plus the BPMN live-debug surface -(`_shared/bpmn_live.py`, per LIVE-ADDENDUM's canonical pattern: ephemeral -solution import, `bpmn debug`, `debug-instance variables-all`/`incidents`). -The canonical live grader this file's plumbing is modeled on is -`_shared/check_jira_get_issue.py`; the non-simulated sibling -`_shared/check_weather_bpmn.py` is the exact structural template. - -Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per -check. - -Assertion map (Flow -> BPMN): - F check_weather_flow_simulated.py:20-21 - assert_flow_has_any_node_type( - ["core.action.http", "custom-codereval-openmeteoapis"]) - -> any element carrying an Intsvc.HttpExecution - uipath:activity wrapper. BPMN has no curated - Open-Meteo Integration Service connector, so - only the managed-HTTP construct from - PORTING-BRIEF's construct-translation table - remains; the Flow grader's connector-fallback - branch has nothing to translate to (same as - check_weather_bpmn.py). - F check_weather_flow_simulated.py:22 - run_debug(timeout=240) implicitly requires - finalStatus == "Completed" (flow_check.run_debug - raises on a non-Completed status internally; - `bpmn debug` returns only an instance id, so the - check is explicit here) - -> FinalStatus in COMPLETED_STATUSES and - debug-instance incidents is empty - F check_weather_flow_simulated.py:23-24 - assert_outputs_contain(payload, - ["nice day", "bring a jacket"], require_all=False) - -> either verdict string found among the root - scope's variable leaves AND every element's - Outputs in `debug-instance variables-all` - (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read - back null even when mapped correctly, so the - search is not scoped to one declared output - variable) - I locate/parse .bpmn, name-agnostic (the simulated persona - withholds the project name, mirroring Flow's own - name-agnostic glob for this variant: no fixed basename hint) - -> bpmn_check.find_bpmn_file()/resolve_project() - I ephemeral solution init + `solution projects import` + sha256 - pin of the imported bytes against the submitted file -- - `bpmn debug` runs against an imported project, unlike - `flow debug`, which runs directly against the discovered - project directory -> LIVE-ADDENDUM canonical live pattern - (mirrors check_jira_get_issue.py / check_weather_bpmn.py) - DROPPED require_no_private_connector_values / require_sequence_integrity / - require_di_for_visible_elements / connection-binding checks -- - not in Flow; the `bpmn validate` criterion covers structure - -Flow's grader does not assert a Script node or a Decision node exists (only -the weather-API node type and the branch output are graded), so this checker -does not add a structural check for the exclusiveGateway either -- adding one -would be a BPMN-only requirement the Flow prompt/grader never had. -""" - -from __future__ import annotations - -import os -import sys -import xml.etree.ElementTree as ET -from pathlib import Path - -# .../uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py -sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 - -from _shared.bpmn_check import ( # noqa: E402 - find_bpmn_file, - has_typed_uipath_extension, - resolve_project, -) -from _shared import bpmn_live # noqa: E402 -from _shared.bpmn_live import ( # noqa: E402 - CheckFailure, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, -) - -BPMN_NS = "http://www.omg.org/spec/BPMN/20100524/MODEL" -# registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask; the eval agent emits either (CI run 35538279757). -ACTIVITY_TYPES = ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") -ACTIVITY_TYPE = ACTIVITY_TYPES[0] -VERDICTS = ("nice day", "bring a jacket") - -LIVE_RUN_DIR = Path("bellevue-weather-simulated-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} - -# Worst-case wall clock this checker can spend, priced the way -# _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug -# call below passes no timeout/retries/backoff kwargs, so it prices at -# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). -# The surrounding CLI steps (solution init/import, variables-all, incidents) -# are not priced by that guard, so their sum is added by hand here and the -# criterion `timeout:` in bellevue_weather_simulated.yaml documents the -# arithmetic (identical to check_weather_bpmn.py's own budget): -# 90 (solution init) + 180 (solution import) + 480 (debug) -# + 120 (variables-all) + 120 (incidents) = 990 -# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 -# Flow's own criterion timeout (600) does not cover this larger live-debug -# surface, so it is raised to 1050 -- the one sanctioned deviation -# LIVE-ADDENDUM allows, because the budget is a property of the CLI surface, -# not of what is graded. - - -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - -def find_http_execution_nodes(root: ET.Element) -> list[ET.Element]: - """Every element carrying an Intsvc.HttpExecution uipath:activity wrapper. - - Scans every descendant, not a fixed tag list (registry templates may emit - the managed-HTTP activity as sendTask, serviceTask, or a plain task) -- - mirrors bpmn_live.index_runtime_connectors' own scanning discipline. Skips - ``bpmn:extensionElements`` nodes themselves: ``has_typed_uipath_extension`` - matches an element whose OWN direct children include a matching - ``uipath:activity`` (the wrapper task) as well as an ``extensionElements`` - node (whose direct child literally is that ``uipath:activity``), which - would otherwise double-count every match once per node. - """ - return [ - el - for el in root.iter() - if el.tag != f"{{{BPMN_NS}}}extensionElements" - and any(has_typed_uipath_extension(el, "activity", t) for t in ACTIVITY_TYPES) - ] - - -def collect_output_haystack(variables_data: object) -> str: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- it mirrors Flow's own - assert_outputs_contain(), which flattens the whole outputs payload. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return "\n".join(str(v) for v in leaves).lower() - - -def main() -> None: - # Name-agnostic: no hint. The simulated persona withholds the project - # name unless asked, so the submitted .bpmn may not be named - # "BellevueWeather*" -- find_bpmn_file() falls back to "exactly one - # .bpmn" or "the one with project.uiproj beside it" when several exist. - bpmn_path = find_bpmn_file() - - try: - root = ET.parse(bpmn_path).getroot() - except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") - - http_nodes = find_http_execution_nodes(root) - if not http_nodes: - _fail( - f"{bpmn_path} has no element carrying an {ACTIVITY_TYPE} uipath:activity " - "wrapper (no managed-HTTP weather node found)" - ) - print(f"OK: bpmn has a managed-HTTP node ({ACTIVITY_TYPE})") - - project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "BellevueWeatherSimulatedLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") - - debug_data, instance_id = bpmn_live.run_debug( - imported_project, {}, LIVE_RUN_DIR / "debug.log" - ) - print(f"OK: debug completed (instance {instance_id})") - - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {incidents_list}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) - - haystack = collect_output_haystack(variables_data) - if not any(verdict in haystack for verdict in VERDICTS): - _fail( - f"outputs do not contain either verdict string {list(VERDICTS)!r}\n" - f"outputs: {haystack[:1000]}" - ) - print("OK: bpmn outputs contain a weather branch message") - print("PASS: all BellevueWeather (simulated) checks passed") - - -if __name__ == "__main__": - try: - main() - except CheckFailure as error: - raise SystemExit(f"FAIL: {error}") from error diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml deleted file mode 100644 index 0ab76bbfb6..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/ceql_where/ceql_where.yaml +++ /dev/null @@ -1,76 +0,0 @@ -task_id: skill-bpmn-ceql-where -description: > - Tests the CEQL where query IS feature — agent plans a structured filter - tree for the Microsoft Entra (Azure AD) connector's "List groups" - operation with a `displayName = "active"` filter, following the canonical - shape documented in the Filter Trees (CEQL) section of the - uipath-platform skill. The process must reference the registered connector - key (`uipath-microsoft-azureactivedirectory`) and use an exclusive gateway - + Terminate end event for routing. This is an offline structure test: - product validation is not graded because the two authoring loops persist - an unconfigured connector differently. - Ported from Flow `connector_features/ceql_where.yaml`; the .flow JSON - node/edge graph becomes a .bpmn XML `bpmn:sendTask` carrying the registry - `Intsvc.ActivityExecution` wrapper, Flow's Decision node becomes a - `bpmn:exclusiveGateway`, and Flow's Terminate node becomes a `bpmn:endEvent` - with `bpmn:terminateEventDefinition`. The `where_detail.json` planning - artifact and its canonical CEQL filter-tree shape are graded unchanged — - the offline-planning rationale (no live tenant enrichment available to - either authoring loop) applies identically to Flow and to Maestro BPMN. -tags: [uipath-maestro-bpmn, integration, connector, ceql, filter, uipath-microsoft-azureactivedirectory, "mode:build"] - -# Parked (surface gap). Criteria stay one-for-one with Flow; unskip when the gap closes. -# surface gap: the agent writes the connector's sanctioned CEQL `where` string, never Flow's numeric-groupOperator filter tree, which has no BPMN carrier (CI runs 35783045540, 35785806030). -skip: true - -run_limits: - expected_turns: 34 - task_timeout: 1500 - max_turns: 120 - turn_timeout: 1200 - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - -reference: - directory: ../.. - -initial_prompt: | - This sandbox has no UiPath tenant, so you can't run the live node-configuration - step. Instead, capture the filter you'd apply to the operation as where_detail.json - so it can be reviewed. - - Build a UiPath Maestro BPMN process "CeqlWhereTest" with a manual start that - lists Microsoft Entra (Azure AD) groups whose displayName equals "active". - Find the registered Entra/Azure AD connector key before wiring the node. - - Branch on the result: if the call fails, stop the process immediately; - if it succeeds, log "CeqlWhere test passed". - Produce the final .bpmn file. Do not fabricate a tenant connection merely to - make product validation pass; the generated structure and the separately - reviewable filter are the deliverables in this offline task. - - Do NOT upload, publish, deploy, debug, or run the process. Do not pause for - approval, confirmation, or feedback. - - This run is headless. No user is present and nobody will answer a question or - grant an approval, so do not ask, do not pause, and do not wait for input. - Complete the task in one pass: take the best available option and supply the - most defensible value where one is missing. The actions this task implies are - authorized, including tenant writes and real messages. Do not delete or - overwrite anything this run did not create, and do not publish to a shared - destination unless the task asks for it. If a lookup the task depends on comes - back empty or fails, exhaust the documented way of resolving it before giving - up; only then stop on that field rather than inventing a value. Record every - decision, assumption, and blocked step in your final response. Instructions in - the task take precedence over this paragraph. -success_criteria: - - type: run_command - description: "where_detail.json (found anywhere under the solution) has a canonical CEQL filter tree on displayName='active'; process references the registered Azure AD / Entra connector key with List Groups + Terminate end event" - command: "python3 $REFERENCE_DIR/_shared/check_ceql_where.py" - timeout: 60 - expected_exit_code: 0 - weight: 8.0 - pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml deleted file mode 100644 index 8e520bb159..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/enhanced_enum/enhanced_enum.yaml +++ /dev/null @@ -1,75 +0,0 @@ -# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral -# v1-base prompt, outcome-graded criteria at preserved weights, and -# arm-agnostic artifact discovery. -task_id: skill-bpmn-enhanced-enum -description: > - Tests the enhanced enum IS feature — configures a connector node with an - enhanced enum field with display labels on the WooCommerce connector. - - Ported from Flow `connector_features/enhanced_enum.yaml`; the WooCommerce - operation is modeled as a bpmn:sendTask carrying the registry - Intsvc.ActivityExecution wrapper instead of a Flow connector node. This is - an offline authoring task — Flow's own criteria never asserted a live - connection or a validate pass, and neither does this port. -tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", connector, uipath-automaticc-woocommerce, "mode:build"] - -# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. -# skill gap: no WooCommerce connector node is produced in either run (CI runs 35789221753, 35790934047). -skip: true - -run_limits: - expected_turns: 35 - task_timeout: 1500 - max_turns: 120 - turn_timeout: 1200 - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - -reference: - directory: ../.. - -initial_prompt: | - Create a new UiPath Maestro BPMN process project called "EnhancedEnumTest" - with a manual start. - Build a process that retrieves WooCommerce product reviews and lets the caller - choose whether they come back oldest-first or newest-first, picking from friendly - labels rather than raw codes. Discover the get-product-reviews operation and set - its sort-order field. Produce the final .bpmn file. - If the WooCommerce operation exists in the registry but the tenant has no - live connection, keep the real WooCommerce operation node with its inputs left unset - and stop there. Do not fabricate a tenant connection merely to make product - validation pass; the connector shape is the deliverable in this offline task. - - Do NOT upload, publish, deploy, debug, or run the process. Do not pause for - approval, confirmation, or feedback. - - This run is headless. No user is present and nobody will answer a question or - grant an approval, so do not ask, do not pause, and do not wait for input. - Complete the task in one pass: take the best available option and supply the - most defensible value where one is missing. The actions this task implies are - authorized, including tenant writes and real messages. Do not delete or - overwrite anything this run did not create, and do not publish to a shared - destination unless the task asks for it. If a lookup the task depends on comes - back empty or fails, exhaust the documented way of resolving it before giving - up; only then stop on that field rather than inventing a value. Record every - decision, assumption, and blocked step in your final response. Instructions in - the task take precedence over this paragraph. -success_criteria: - - type: run_command - description: "BPMN file exists and is well-formed XML" - command: "python3 $REFERENCE_DIR/_shared/check_enhanced_enum.py check_exists" - timeout: 10 - expected_exit_code: 0 - weight: 3.0 - pass_threshold: 1.0 - - - type: run_command - description: "BPMN process has a connector node referencing uipath-automattic-woocommerce" - command: "python3 $REFERENCE_DIR/_shared/check_enhanced_enum.py check_connector" - timeout: 10 - expected_exit_code: 0 - weight: 3.0 - pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py deleted file mode 100644 index e5e2b601ee..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/jira_is.py +++ /dev/null @@ -1,67 +0,0 @@ -#!/usr/bin/env python3 -"""Minimal live Jira helper for this task — wraps `uip is resources run`. - -Self-contained (no shared module). Assumes `uip` is on PATH and logged in and -that the connection + CE project exist — this is a tenant-gated e2e task. - -The connection is scoped to the curated single-record ops, so we create by -body / get by id / delete by id — never a JQL search. -""" - -from __future__ import annotations - -import json -import subprocess - -CONNECTOR = "uipath-atlassian-jira" -FOLDER_PATH = "Shared/uipath-maestro-flow" -CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" -PROJECT_KEY = "CE" # "Coder Eval" project on uipath-sandbox-380 -ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project - - -def _run(*args: str) -> dict: - out = subprocess.run( - ["uip", *args, "--output", "json"], - capture_output=True, text=True, timeout=120, - ).stdout - return json.loads(out) - - -def connection_id() -> str: - folder_key = _run("or", "folders", "get", FOLDER_PATH)["Data"]["Key"] - conns = _run("is", "connections", "list", CONNECTOR, "--folder-key", folder_key, "--refresh")["Data"] - return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) - - -def create_issue(conn_id: str, summary: str) -> str: - body = {"fields": {"project": {"key": PROJECT_KEY}, "issuetype": {"id": ISSUETYPE_ID}, "summary": summary}} - return _run( - "is", "resources", "run", "create", CONNECTOR, "curated_create_issue", - "--connection-id", conn_id, "--body", json.dumps(body), - )["Data"]["key"] - - -def get_issue(conn_id: str, key: str) -> dict | None: - """Return the issue's `fields` dict (includes `summary`, `status`, and - `comment`), or None if it doesn't exist (404).""" - env = _run( - "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", - "--connection-id", conn_id, - "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", - ) - if env.get("Result") == "Failure": - return None - return env["Data"].get("fields", {}) - - -def delete_issue(conn_id: str, key: str) -> None: - """Delete an issue by key. A 404 (already gone) is a no-op.""" - _run( - "is", "resources", "run", "delete", CONNECTOR, "issue", - "--connection-id", conn_id, "--query", f"issueId={key}", - # The CLI never prompts and REFUSES an irreversible delete without this - # flag ("Confirmation required … Re-run with --yes"). Without it every - # teardown since 08-19 printed WARN and left its ticket in the CE project. - "--yes", - ) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py deleted file mode 100644 index 2875ac11f3..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/seed_jira.py +++ /dev/null @@ -1,39 +0,0 @@ -#!/usr/bin/env python3 -"""pre_run: write seed.json with a small batch of issues to create and the -per-branch comment markers. No live issue is created here — the agent's flow -creates one issue per list item when the check runs `flow debug`. - -The batch mixes `priority` values so the flow's Switch node must route each -item to a different Add-Comment branch: - - priority == "High" -> comment carries `escalated_marker` - otherwise -> comment carries `routine_marker` - -Every summary and both markers embed the unique per-run `tag`, so the check can -locate exactly this run's issues (no JQL search — the connection is curated-ops -only) and tell the two branches apart. -""" - -import json -import secrets -from pathlib import Path - -import jira_is - -tag = secrets.token_hex(4) -issues = [ - {"summary": f"coder-eval jira lifecycle {tag} item1", "priority": "High"}, - {"summary": f"coder-eval jira lifecycle {tag} item2", "priority": "Low"}, - {"summary": f"coder-eval jira lifecycle {tag} item3", "priority": "High"}, -] -seed = { - "tag": tag, - "project_key": jira_is.PROJECT_KEY, - "issuetype_id": jira_is.ISSUETYPE_ID, - "issues": issues, - "escalated_marker": f"ESCALATED {tag}", - "routine_marker": f"ROUTINE {tag}", -} -Path("seed.json").write_text(json.dumps(seed, indent=2)) -highs = sum(1 for i in issues if i["priority"] == "High") -print(f"OK: wrote {len(issues)} seed issues ({highs} High / {len(issues) - highs} other), tag={tag}") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py deleted file mode 100644 index 91055ad017..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/_setup/teardown_jira.py +++ /dev/null @@ -1,22 +0,0 @@ -#!/usr/bin/env python3 -"""post_run: delete every issue the run created (keys in `.created_keys`). -Idempotent and never fails the task.""" - -import sys -from pathlib import Path - -import jira_is - -try: - kf = Path(".created_keys") - keys = kf.read_text().split() if kf.is_file() else [] - if keys: - conn = jira_is.connection_id() - for key in keys: - jira_is.delete_issue(conn, key) - print(f"OK: deleted {key}") - else: - print("OK: nothing to delete") -except Exception as e: # noqa: BLE001 — teardown must not fail the task - print(f"WARN: teardown ignored error: {e}") -sys.exit(0) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml deleted file mode 100644 index c552f31fd4..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_lifecycle/jira_lifecycle.yaml +++ /dev/null @@ -1,135 +0,0 @@ -task_id: skill-bpmn-jira-lifecycle -description: > - E2E live Jira coverage of a composite multi-instance-loop-and-gateway BPMN - process. The agent builds ONE UiPath Maestro BPMN process (manual start) - that iterates a seeded batch of issues and, per item, creates an Atlassian - Jira issue and then branches on the item's `priority` (an exclusive - gateway) to a branch-specific "Add Comment": `High` items get the seeded - `escalated_marker`, all others the `routine_marker`. Grading imports the - submitted project into an ephemeral solution and executes it against a real - Jira sandbox connection (`bpmn debug`), then re-reads the tenant: it asserts - the loop created every seeded issue and the gateway routed each to the - correct comment branch. The batch + markers are unique per run - (`seed.json`), so a fabricated or hardcoded output cannot pass. Every - created issue is cleaned up in post_run. - - Tenant prerequisite: a `uipath-atlassian-jira` connection in folder - `Shared/uipath-maestro-flow` (currently the single Jira connection there), - reaching the `CE` / "Coder Eval" project (issue type Task). Targets live in - the task's `jira_is.py`. - - Ported from Flow `e2e/jira_lifecycle/jira_lifecycle.yaml`; the flow's - `core.logic.loop` node is modeled as a `bpmn:multiInstanceLoopCharacteristics` - loop (over `=vars.Var_Issues`, item read as `iterator[0].item`), its - Switch/Decision node as a `bpmn:exclusiveGateway` with a `conditionExpression` - on `iterator[0].item.priority`, and each connector node as a - `bpmn:sendTask` carrying the registry `Intsvc.ActivityExecution` wrapper - (`uip is activities list uipath-atlassian-jira`: curated objectName - `curated_create_issue` for Create Issue, `curated_add_comment` for Add - Comment, both method POST) instead of a Flow connector node. `flow debug`'s - single-call inline payload becomes an ephemeral-solution import + `bpmn - debug` + `debug-instance variables-all`/`incidents` read, and the loop's - created-issue keys are recovered by scanning the `variables-all` payload - text for the seeded project's issue-key pattern instead of one inline debug - payload. -tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", "node:loop", "node:switch", "node:decision", connector, e2e, uipath-atlassian-jira, "mode:build"] - -# Parked (three different runtime failures in three runs (CI runs 35503094182, 35524004307, 35525387843); Flow's own version is flaky. Needs a live investigation before it can gate anything.). Criteria stay one-for-one with Flow; unskip when the gap closes. -# three different runtime failures in three runs (CI runs 35503094182, 35524004307, 35525387843); Flow's own version is flaky. Needs a live investigation before it can gate anything. -skip: true - -run_limits: - expected_turns: 55 - task_timeout: 2700 - max_turns: 150 - turn_timeout: 900 - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - - type: template_dir - path: ../../_setup - mount_point: _setup - - type: template_dir - path: _setup - mount_point: _setup - -pre_run: - - command: "python3 _setup/seed_jira.py" - timeout: 60 - -reference: - directory: ../.. - -post_run: - # Repo-standard solution sweep: the grader keeps its ephemeral solution - # under the sandbox CWD precisely so this glob finds the .uipx. - - command: "python3 _setup/cleanup_solutions.py" - timeout: 120 - # Connector record the solution sweep cannot reach (jira_is.py precedent). - - command: "python3 _setup/teardown_jira.py" - timeout: 180 - -initial_prompt: | - Create a new UiPath Maestro BPMN process project called "JiraLifecycle" - with a manual start, inside a solution of the same name. - - Read `seed.json` in the current directory. It contains a list of `issues` - (each with a `summary` and a `priority`), the `project_key` and `issuetype_id` - to create issues under, and two comment bodies: `escalated_marker` and - `routine_marker`. - - Build a process that loops over the `issues` list and, for EACH item: - 1. Creates a Jira issue with the item's `summary`, using the Atlassian Jira - "Create Issue" connector activity (with `project_key` / `issuetype_id`). - 2. Branches on the item's `priority`: when it is "High", add a comment whose - body is `escalated_marker`; otherwise add a comment whose body is - `routine_marker`. Use the Atlassian Jira "Add Comment" activity against - the issue created in step 1. - - Pass every summary and comment body verbatim from seed.json. - - Use the Atlassian Jira connection available in the `Shared/uipath-maestro-flow` - folder. If the Atlassian Jira connector isn't available in your registry, stop - rather than falling back to a generic HTTP request. - - Validate the final BPMN file. - - This run is headless. No user is present and nobody will answer a question or - grant an approval, so do not ask, do not pause, and do not wait for input. - Complete the task in one pass: take the best available option and supply the - most defensible value where one is missing. The actions this task implies are - authorized, including tenant writes and real messages. Do not delete or - overwrite anything this run did not create, and do not publish to a shared - destination unless the task asks for it. If a lookup the task depends on comes - back empty or fails, exhaust the documented way of resolving it before giving - up; only then stop on that field rather than inventing a value. Record every - decision, assumption, and blocked step in your final response. Instructions in - the task take precedence over this paragraph. -success_criteria: - - type: run_command - description: "BPMN validates successfully" - command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" - timeout: 180 - expected_exit_code: 0 - weight: 2.0 - pass_threshold: 1.0 - - - type: command_executed - description: "Advisory: live-v1 agent ran bpmn debug" - tool_name: "Bash" - # `bpmn` exists only under `maestro` -- `uip bpmn debug` is an unknown - # command, so the `maestro` segment must not be optional. - command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+maestro\s+bpmn\s+debug' - min_count: 1 - weight: 1.5 - pass_threshold: 0.0 - - - type: run_command - description: "JiraLifecycle checks: valid BPMN with a multi-instance loop + exclusive gateway over the Jira connector, debug creates every seeded issue, and each carries the comment its priority branch should have posted" - command: "python3 $REFERENCE_DIR/_shared/check_jira_lifecycle.py" - timeout: 1320 - expected_exit_code: 0 - weight: 5.0 - pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py deleted file mode 100644 index d734b13884..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/jira_is.py +++ /dev/null @@ -1,65 +0,0 @@ -#!/usr/bin/env python3 -"""Minimal live Jira helper for this task — wraps `uip is resources run`. - -Self-contained (no shared module). Assumes `uip` is on PATH and logged in and -that the connection + CE project exist — this is a tenant-gated e2e task. -This helper uses the create / get / delete ops to seed and verify. -""" - -from __future__ import annotations - -import json -import subprocess - -CONNECTOR = "uipath-atlassian-jira" -FOLDER_PATH = "Shared/uipath-maestro-flow" -CONNECTION_NAME = "is-sandboxes-test@uipath.com-uipath-sandbox-380" -PROJECT_KEY = "CE" # "Coder Eval" project on uipath-sandbox-380 -ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project - - -def _run(*args: str) -> dict: - out = subprocess.run( - ["uip", *args, "--output", "json"], - capture_output=True, text=True, timeout=120, - ).stdout - return json.loads(out) - - -def connection_id() -> str: - folder_key = _run("or", "folders", "get", FOLDER_PATH)["Data"]["Key"] - conns = _run("is", "connections", "list", CONNECTOR, "--folder-key", folder_key, "--refresh")["Data"] - return next(c["Id"] for c in conns if c["Name"] == CONNECTION_NAME) - - -def create_issue(conn_id: str, summary: str) -> str: - body = {"fields": {"project": {"key": PROJECT_KEY}, "issuetype": {"id": ISSUETYPE_ID}, "summary": summary}} - return _run( - "is", "resources", "run", "create", CONNECTOR, "curated_create_issue", - "--connection-id", conn_id, "--body", json.dumps(body), - )["Data"]["key"] - - -def get_issue(conn_id: str, key: str) -> dict | None: - """Return the issue's `fields` dict (includes `summary` and `comment`), - or None if it doesn't exist (404).""" - env = _run( - "is", "resources", "run", "get", CONNECTOR, "curated_get_issue", - "--connection-id", conn_id, - "--query", f"project={PROJECT_KEY}&issuetype={ISSUETYPE_ID}&issueId={key}", - ) - if env.get("Result") == "Failure": - return None - return env["Data"].get("fields", {}) - - -def delete_issue(conn_id: str, key: str) -> None: - """Delete an issue by key. A 404 (already gone) is a no-op.""" - _run( - "is", "resources", "run", "delete", CONNECTOR, "issue", - "--connection-id", conn_id, "--query", f"issueId={key}", - # The CLI never prompts and REFUSES an irreversible delete without this - # flag ("Confirmation required … Re-run with --yes"). Without it every - # teardown since 08-19 printed WARN and left its ticket in the CE project. - "--yes", - ) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py deleted file mode 100644 index 3e3cc46838..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/seed_jira.py +++ /dev/null @@ -1,26 +0,0 @@ -#!/usr/bin/env python3 -"""pre_run: create a couple of real issues carrying a unique tag in their -summary, and write seed.json with the JQL that selects exactly them plus the -comment the flow should stamp on each match. The created keys are recorded to -`.created_keys` so teardown deletes them regardless of the run outcome. -""" - -import json -import secrets -from pathlib import Path - -import jira_is - -tag = secrets.token_hex(4) -conn = jira_is.connection_id() -keys = [jira_is.create_issue(conn, f"coder-eval jira search-triage {tag} #{n}") for n in (1, 2)] -seed = { - "tag": tag, - "project_key": jira_is.PROJECT_KEY, - "jql": f'project = {jira_is.PROJECT_KEY} AND summary ~ "{tag}"', - "processed_comment": f"TRIAGED {tag}", - "issue_keys": keys, -} -Path("seed.json").write_text(json.dumps(seed, indent=2)) -Path(".created_keys").write_text("\n".join(keys) + "\n") -print(f"OK: seeded {len(keys)} issues {keys} (tag={tag}) for JQL search") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py deleted file mode 100644 index 91055ad017..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/_setup/teardown_jira.py +++ /dev/null @@ -1,22 +0,0 @@ -#!/usr/bin/env python3 -"""post_run: delete every issue the run created (keys in `.created_keys`). -Idempotent and never fails the task.""" - -import sys -from pathlib import Path - -import jira_is - -try: - kf = Path(".created_keys") - keys = kf.read_text().split() if kf.is_file() else [] - if keys: - conn = jira_is.connection_id() - for key in keys: - jira_is.delete_issue(conn, key) - print(f"OK: deleted {key}") - else: - print("OK: nothing to delete") -except Exception as e: # noqa: BLE001 — teardown must not fail the task - print(f"WARN: teardown ignored error: {e}") -sys.exit(0) diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml deleted file mode 100644 index 1da8807818..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_search_triage/jira_search_triage.yaml +++ /dev/null @@ -1,111 +0,0 @@ -task_id: skill-bpmn-jira-search-triage -description: > - E2E live Jira coverage of a JQL-search-driven triage process: a - manual-start Maestro BPMN process that searches for issues matching a - seeded JQL and, for each match, adds a triage comment. pre_run seeds two - real issues carrying a unique tag; grading runs the process (`bpmn debug`) - and asserts both seeded issues come back carrying the triage comment. - - Tenant prerequisite: a `uipath-atlassian-jira` connection in folder - `Shared/uipath-maestro-flow` reaching the `CE` / "Coder Eval" project, with - issue-search scope. Targets live in `jira_is.py`. - - Ported from Flow `e2e/jira_search_triage/jira_search_triage.yaml`; the - search-and-loop-and-comment shape is modeled as a bpmn:sendTask carrying - the registry Intsvc.ActivityExecution wrapper for the Search Issues by JQL - operation (`uip is activities list uipath-atlassian-jira`: curated - objectName `issue_search_get`, method GET) feeding a - bpmn:multiInstanceLoopCharacteristics loop over the matches, whose body is - an Add Comment connector activity (curated objectName - `curated_add_comment`), instead of a Flow search node feeding a - `core.logic.loop` node. `flow debug`'s single inline payload becomes an - ephemeral-solution import + `bpmn debug` + `debug-instance incidents` read. - The second criterion's timeout is raised from Flow's 1320s to 1440s: the - BPMN sequence adds the ephemeral solution init/import and an explicit - incidents read on top of Flow's single inline debug call, on the same - tenant re-read Flow itself performs (arithmetic in the grader's docstring). -tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", "node:loop", connector, e2e, uipath-atlassian-jira, "mode:build"] - -# Parked (platform/skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. -# platform/skill gap: multiInstanceLoopCharacteristics over a connector response fails at runtime with 400008 (CI run 35525387843). -skip: true - -run_limits: - expected_turns: 45 - task_timeout: 2400 - max_turns: 120 - turn_timeout: 900 - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - - type: template_dir - path: ../../_setup - mount_point: _setup - - type: template_dir - path: _setup - mount_point: _setup - -pre_run: - - command: "python3 _setup/seed_jira.py" - timeout: 90 - -reference: - directory: ../.. - -post_run: - # Repo-standard solution sweep: the grader keeps its ephemeral solution - # under the sandbox CWD precisely so this glob finds the .uipx. - - command: "python3 _setup/cleanup_solutions.py" - timeout: 120 - # Connector record the solution sweep cannot reach (jira_is.py precedent). - - command: "python3 _setup/teardown_jira.py" - timeout: 120 - -initial_prompt: | - Create a new UiPath Maestro BPMN process project called "JiraSearchTriage" - with a manual start, inside a solution of the same name. - - Read `seed.json` in the current directory for the `jql` query and the - `processed_comment` body. - - Build a process that: - 1. Searches Jira for issues matching `jql`, using the Atlassian Jira - "Search Issues by JQL" connector activity. - 2. Loops over the returned issues and adds a comment to each whose body is - `processed_comment` (pass it verbatim), using the "Add Comment" activity. - - Use the Atlassian Jira connection available in the `Shared/uipath-maestro-flow` - folder. If the Atlassian Jira connector isn't available in your registry, stop - rather than falling back to a generic HTTP request. - - Validate the final BPMN file. - - This run is headless. No user is present and nobody will answer a question or - grant an approval, so do not ask, do not pause, and do not wait for input. - Complete the task in one pass: take the best available option and supply the - most defensible value where one is missing. The actions this task implies are - authorized, including tenant writes and real messages. Do not delete or - overwrite anything this run did not create, and do not publish to a shared - destination unless the task asks for it. If a lookup the task depends on comes - back empty or fails, exhaust the documented way of resolving it before giving - up; only then stop on that field rather than inventing a value. Record every - decision, assumption, and blocked step in your final response. Instructions in - the task take precedence over this paragraph. -success_criteria: - - type: run_command - description: "BPMN validates successfully" - command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" - timeout: 180 - expected_exit_code: 0 - weight: 2.0 - pass_threshold: 1.0 - - - type: run_command - description: "JiraSearchTriage checks: valid BPMN with a JQL search feeding a loop of Add-Comment, debug completes, and both seeded issues carry the triage comment" - command: "python3 $REFERENCE_DIR/_shared/check_jira_search_triage.py" - timeout: 1440 - expected_exit_code: 0 - weight: 5.0 - pass_threshold: 1.0 diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml deleted file mode 100644 index ca621a8c9d..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml +++ /dev/null @@ -1,142 +0,0 @@ -# Same-ground alignment expansion plan (Tao, 2026-08-21): loop-neutral -# v1-base prompt, outcome-graded criteria at preserved weights, and -# arm-agnostic artifact discovery. -task_id: skill-bpmn-bellevue-weather-simulated -description: > - Bellevue weather process (HTTP -> script -> decision), but driven by a - simulated non-technical user who withholds requirements until asked. Tests - the agent's ability to clarify an ambiguous ask before building. - - Ported from Flow - `interactive/bellevue_weather_simulated/bellevue_weather_simulated.yaml`; - the persona/goal/constraints are unchanged apart from product-noun swaps - ("UiPath Flow" -> "UiPath Maestro BPMN process", "a `.flow` file" -> "a - `.bpmn` file"), the Managed HTTP construct and live-debug sequence follow - the same translation as the committed non-simulated port - (`multi_node/bellevue_weather/bellevue_weather.yaml` + - `_shared/check_weather_bpmn.py`), and the live-check criterion timeout - grows from Flow's 600s to 1050s to fund that larger live-debug surface - (ephemeral solution import + `bpmn debug` + `debug-instance - variables-all`/`incidents`, per LIVE-ADDENDUM's budget rule); `run_limits` - are otherwise Flow's verbatim, except `task_timeout`, which is raised from - 2400s to 2550s (turn_timeout 1200 + worst-case grading 1245 + 60s margin) - so the larger live-check budget still fits under the single turns+grading - watchdog. The graded assertions themselves (a weather-HTTP node exists; the - run's outputs contain one of the two verdict strings; the project was named - as elicited) are unchanged from Flow's own grader. -tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", "node:decision", "feature:http", simulation] - -# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. -# skill gap: same `temperature_2m`-of-undefined runtime fault as multi_node/bellevue_weather (CI run 35785806030). -skip: true - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - - type: template_dir - path: ../../_setup - mount_point: _setup - -run_limits: - # Ceiling on agent turns across the whole dialog, with generous headroom so a - # long build is never truncated mid-flow. The dialog length itself is bounded - # by simulation.max_turns below. Unchanged from Flow. - max_turns: 70 - turn_timeout: 1200 - # Flow's task_timeout (2400) does not cover this task's larger live-debug - # criterion (1050s vs Flow's 600s): turn_timeout (1200) + worst-case grading - # (180 validate + 1050 live check + 15 name check = 1245) + 60s margin = - # 2505; rounded up for per-criterion spawn overhead. task_timeout wraps - # turns AND grading under one watchdog (tests/README.md), so it must cover - # both, not grading alone. - task_timeout: 2550 - -# No initial_prompt — the simulator generates turn 1 from its persona + goal. - -reference: - directory: ../.. - -simulation: - enabled: true - - persona: | - You are a non-technical operations analyst. You have heard of UiPath but - have never built a process yourself. You speak in plain English, never in - code. You are friendly but impatient: if the agent dithers, prod it to - just build something. You cannot read or write code — if the agent shows - you code, say "I trust you, just make it work." - - goal: | - You want a UiPath Maestro BPMN process that checks today's weather for - the Bellevue office and tells you whether it's a nice day. - - OPENING MESSAGE: open the way you would to a colleague — a couple of short - sentences covering the outcome you want plus the rules that actually - matter to you: it is the Bellevue, Washington office; strictly above 60 - degrees Fahrenheit is a nice day and the result should say 'nice day', at - or below 60 should say 'bring a jacket'; and you want a short written - summary of the weather rather than raw numbers. Say that it has to be - built as a UiPath Maestro BPMN process — not an API workflow, not a - standalone script — inside a solution of the same name, and that you want - it validated before you sign off. Do not name files or paths and do not - use technical wording. Close by inviting the agent to ask you anything it - needs before it starts building. Keep the withheld items below to - yourself until it asks. - - WITHHELD — reveal each of these ONLY if the agent asks about it: - - The project name should be "BellevueWeather". - - Where the weather data comes from: the open-meteo public API, called - directly over HTTP — NOT an Integration Service connector. - - You do not require the project to execute before signing off. - - constraints: - - "Never paste or read code. If shown code, say 'I trust you, just make it work.'" - - "Do not mention 'HTTP', 'open-meteo', 'connector', 'node', 'script', or 'decision' unless the agent brings them up first." - - "Do not reveal the project name 'BellevueWeather' unless asked what to name it." - - "If the agent proposes getting the weather through an Integration Service connector, tell it to call the public open-meteo API directly over HTTP instead." - - "The one thing worth pushing back on: if the agent reports it built something that is not a Maestro BPMN process at all (e.g. an API workflow), tell it you specifically want a Maestro BPMN process — i.e. a `.bpmn` file in the workspace — and are not satisfied until there is one." - - "If the agent asks you anything before it starts building, answer every outstanding question in ONE message — include any withheld fact it asked about, stated plainly — and then let it build." - - "The moment the agent reports it has built the process, the conversation is over: reply with exactly <<>> on its own line. Do not ask for a rebuild, polish, re-checks, or alternative designs, and do not raise a requirement you have not mentioned yet — the grader verifies the result from here." - max_turns: 8 - stop_token: "<<>>" - stop_on_criteria_pass: false - check_criteria: end_of_dialog - n_trials: 1 - -success_criteria: - # Name-agnostic: any unambiguous generated .bpmn must validate. The exact - # project name is scored separately below. - - type: run_command - description: "uip maestro bpmn validate passes (any .bpmn)" - command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" - timeout: 180 - expected_exit_code: 0 - weight: 3.0 - pass_threshold: 1.0 - - # Runtime check: - - type: run_command - description: "BPMN debug runs; weather-HTTP node executed and output contains a verdict" - command: "python3 $REFERENCE_DIR/_shared/check_weather_bpmn_simulated.py" - timeout: 1050 - expected_exit_code: 0 - weight: 5.0 - pass_threshold: 1.0 - - # The project name is a WITHHELD, COSMETIC requirement (persona reveals it only - # if asked what to name it) — the process behaves identically whatever it's - # called. Non-gating: reports whether the agent elicited the exact name - # (score 1/0), but pass_threshold 0 keeps a cosmetic miss from failing the - # task and the low weight keeps it from dominating the score. - - type: run_command - description: "Project/BPMN file is named BellevueWeather (elicited from the user)" - command: "find . -iname 'BellevueWeather*.bpmn' | grep -q ." - timeout: 15 - expected_exit_code: 0 - weight: 1.0 - pass_threshold: 0 - -post_run: - - command: "python3 _setup/cleanup_solutions.py" - timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml deleted file mode 100644 index ccaf3ad26f..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml +++ /dev/null @@ -1,192 +0,0 @@ -task_id: skill-bpmn-slack-channel-description-simulated -description: > - Single-connector Slack BPMN process (read a channel's description and output - it), driven by a simulated non-technical user who withholds the channel and - project name until asked. Tests the agent's ability to clarify an ambiguous - ask before building. Executes: builds a bpmn:sendTask carrying the registry - Intsvc.ActivityExecution wrapper for the uipath-salesforce-slack connector, - validates, then runs an ephemeral-solution import + `bpmn debug` and asserts - the fetched channel description (the Bellevue office address) lands in the - runtime outputs — needs a live Slack connection. - Ported from Flow `interactive/slack_channel_description_simulated/slack_channel_description_simulated.yaml`; - the connector node is modeled as a bpmn:sendTask carrying the registry - Intsvc.ActivityExecution wrapper (connectorKey uipath-salesforce-slack; - `uip is activities list uipath-salesforce-slack`: curated GetConversationInfo, - objectName ConversationsInfo_GET, method GETBYID) instead of a Flow - connector node, `flow debug`'s single-call inline payload becomes an - ephemeral-solution import + `bpmn debug` + `debug-instance - variables-all`/`incidents` read (mirroring the CI-passing non-simulated BPMN - sibling, multi_node/slack_channel_description), and the live criterion's - timeout is raised from Flow's 600s to 1050s for the same reason that - sibling's was: the extra CLI steps (`solution init`/`solution projects - import`, separate `variables-all`/`incidents` reads) that BPMN's live - surface needs beyond Flow's single `flow debug` call. All five of Flow's - success criteria are carried over one-for-one, in Flow's order, with Flow's - weights and pass_thresholds: validate, an advisory `command_executed` on - `bpmn debug` (translating Flow's advisory on `flow debug`), the live domain - grader, a static `office-bellevue`/channel-ID check (translating Flow's - `flow_contains.py --regex` call into an inline `python3 -c` scan of every - `.bpmn` file, since BPMN has no `flow_contains.py` equivalent), and the - withheld-project-name advisory. -tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", connector, simulation] - -# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. -# skill gap: the agent lists only page 1 of Slack conversations, so the target channel never appears (CI run 35789221753); Flow's version fully passes 4/12 nightlies. -skip: true - -run_limits: - # Ceiling on agent turns across the whole dialog, with generous headroom so a - # long build is never truncated mid-flow. The dialog length itself is bounded - # by simulation.max_turns below. Flow's run_limits, verbatim. - max_turns: 70 - task_timeout: 2400 - turn_timeout: 1200 - -# No initial_prompt — the simulator generates turn 1 from its persona + goal. - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - - type: template_dir - path: ../../_setup - mount_point: _setup - -reference: - directory: ../.. - -simulation: - enabled: true - - persona: | - You are a non-technical office manager. You have heard of UiPath but have - never built a process yourself. You speak in plain English, never in code. - You are friendly but impatient: if the agent dithers, prod it to just - build something. You cannot read or write code — if the agent shows you - code, say "I trust you, just make it work." - - goal: | - You want a UiPath Maestro BPMN process that grabs the description text off - one of your team's Slack channels and shows it to you. - - OPENING MESSAGE: open the way you would to a colleague — a couple of short - sentences covering the outcome you want plus the details that matter to - you: the channel is #office-bellevue in Slack, you want that channel's - description text as the result, and it has to work on its own without you - supplying the channel every time it runs. Say that it has to be built as a - UiPath Maestro BPMN process — not an API workflow, not a standalone - script — inside a solution of the same name, and that you want it - validated before you sign off. Do not name files or paths and do not use - technical wording. Close by inviting the agent to ask you anything it - needs before it starts building. Keep the withheld items below to - yourself until it asks. - - WITHHELD — reveal each of these ONLY if the agent asks about it: - - The project name should be "SlackChannelDescription". - - You do not require the project to execute before signing off. - - constraints: - - "Never paste or read code. If shown code, say 'I trust you, just make it work.'" - - "Do not mention 'connector' or 'node' unless the agent brings them up first." - - "The process must work for #office-bellevue without you supplying anything at run time. If the agent asks you to provide a channel ID when it runs, tell it to wire the #office-bellevue channel in directly so it runs unattended." - - "Do not reveal the project name 'SlackChannelDescription' unless asked what to name it." - - "The one thing worth pushing back on: if the agent reports it built something that is not a Maestro BPMN process at all (e.g. an API workflow), tell it you specifically want a Maestro BPMN process — i.e. a `.bpmn` file in the workspace — and are not satisfied until there is one." - - "If the agent asks you anything before it starts building, answer every outstanding question in ONE message — include any withheld fact it asked about, stated plainly — and then let it build." - - "The moment the agent reports it has built the process, the conversation is over: reply with exactly <<>> on its own line. Do not ask for a rebuild, polish, re-checks, or alternative designs, and do not raise a requirement you have not mentioned yet — the grader verifies the result from here." - max_turns: 8 - stop_token: "<<>>" - stop_on_criteria_pass: false - check_criteria: end_of_dialog - n_trials: 1 - -pre_run: - - command: 'python3 "_setup/preflight_connections.py" uipath-salesforce-slack=uipath-maestro-flow' - timeout: 120 - -success_criteria: - # Name-agnostic: any unambiguous generated .bpmn must validate. The exact - # project name is scored separately below. - - type: run_command - description: "uip maestro bpmn validate passes (any .bpmn)" - command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" - timeout: 180 - expected_exit_code: 0 - weight: 3.0 - pass_threshold: 1.0 - - # Advisory (non-gating): did the agent itself run `bpmn debug` during the - # session, translating Flow's "Advisory: live-v1 agent ran flow debug" - # command_executed check (same weight/pass_threshold/description). - - type: command_executed - description: "Advisory: live-v1 agent ran flow debug" - tool_name: "Bash" - command_pattern: '(uip|\$UIP|\$\{?UIP\}?)\s+maestro\s+bpmn\s+debug' - min_count: 1 - weight: 1.5 - pass_threshold: 0.0 - - # Runtime check (mirrors the non-simulated BPMN sibling's - # check_channel_description.py, and Flow's own check_channel_description_simulated.py): - # a node targets the Slack connector AND bpmn debug retrieves the channel - # description into the runtime outputs (the Bellevue office address). - # Supersedes a static connector-key grep, which a process that never - # executes would still pass. THIS is the authoritative proof the correct - # channel was targeted — a wrong or unresolved channel fails here regardless - # of whether the process still names #office-bellevue anywhere. - # Timeout raised from Flow's 600s to 1050s — see the module docstring in - # check_channel_description_simulated.py for the arithmetic (solution init + - # import + debug + variables-all + incidents), the same raise the - # non-simulated sibling already made for the identical live sequence. - - type: run_command - description: "BPMN debug runs; Slack connector executed and output has the channel description" - command: "python3 $REFERENCE_DIR/_shared/check_channel_description_simulated.py" - timeout: 1050 - expected_exit_code: 0 - weight: 5.0 - pass_threshold: 1.0 - - # The channel #office-bellevue is a withheld requirement the persona reveals only - # when asked which channel. The runtime debug check above already PROVES the right - # channel was read (its description IS the Bellevue office address); this static - # check separately reports whether a specific channel was elicited and wired in. - # - # The Slack "Get Channel Info" activity (/ConversationsInfo/{conversationsInfoId}) - # is keyed by channel ID, so a correct process resolves "#office-bellevue" -> its - # channel ID (e.g. C0B50H7DE2F) and stores the ID — the human-readable name - # legitimately never survives into the .bpmn. A bare `office-bellevue` substring - # grep therefore FALSE-FAILS every correct implementation. So we pass on EITHER: - # (a) the process still names #office-bellevue (e.g. kept on an element label), OR - # (b) the connector node has a resolved Slack channel ID hardwired into the - # conversationsInfoId path parameter (a specific channel, not a run-time - # input the persona explicitly forbade). Case (b) is not coupled to any one - # workspace's ID — it matches the Slack channel-ID shape, and the runtime - # check above confirms it's the *correct* channel. - # Non-gating (pass_threshold 0): a legitimate ID-based encoding must never - # hard-fail the whole task — see the block comment; scored for signal only. - # Translates Flow's `flow_contains.py --regex '...'` call (identical regex) into - # an inline python3 scan of every .bpmn under the sandbox, since BPMN has no - # flow_contains.py equivalent in _shared/. - - type: run_command - description: "Flow targets the #office-bellevue channel (elicited from the user)" - command: "python3 -c 'import glob, re, sys; pat = re.compile(r\"(?i)(office-bellevue|conversationsInfoId[^A-Za-z0-9]+C[A-Z0-9]{6,})\"); paths = [p for p in glob.glob(\"**/*.bpmn\", recursive=True) if \"node_modules\" not in p]; sys.exit(0 if any(pat.search(open(p, encoding=\"utf-8\").read()) for p in paths) else 1)'" - timeout: 15 - expected_exit_code: 0 - weight: 3.0 - pass_threshold: 0 - - # The project name is a WITHHELD, COSMETIC requirement (persona reveals it - # only if asked what to name it) — the process behaves identically whatever - # it's called. Non-gating: reports whether the agent elicited the exact name - # (score 1/0), but pass_threshold 0 keeps a cosmetic miss from failing the - # task and the low weight keeps it from dominating the score. - - type: run_command - description: "Project/BPMN file is named SlackChannelDescription (elicited from the user)" - command: 'find . -name "SlackChannelDescription.bpmn" | grep -q .' - timeout: 15 - expected_exit_code: 0 - weight: 1.0 - pass_threshold: 0 - -post_run: - - command: "python3 _setup/cleanup_solutions.py" - timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml deleted file mode 100644 index 1cf3f8fde4..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/bellevue_weather/bellevue_weather.yaml +++ /dev/null @@ -1,95 +0,0 @@ -task_id: skill-bpmn-bellevue-weather -description: > - Create a UiPath Maestro BPMN process that fetches today's weather in - Bellevue from open-meteo, formats a summary, and branches on temperature: - if > 60F output 'nice day', otherwise 'bring a jacket'. Exercises a managed - HTTP activity and a decision gateway, then grades by executing the process - (`bpmn debug`) and asserting one of the two verdict strings appears in the - process outputs. - - Ported from Flow `multi_node/bellevue_weather/bellevue_weather.yaml`; the - Managed HTTP Request node (core.action.http.v2) becomes a bpmn:sendTask - carrying the registry Intsvc.HttpExecution wrapper in manual - (connectionless) mode -- BPMN has no curated Open-Meteo connector, so the - Flow grader's connector-fallback option has nothing to translate to and is - dropped -- the Decision node becomes a bpmn:exclusiveGateway, Flow's single - inline `flow debug` payload read becomes an ephemeral-solution import + - `bpmn debug` + `debug-instance variables-all`/`incidents` read, and the - live-check criterion timeout grows from 600s to 1050s to fund that larger - live-debug surface (LIVE-ADDENDUM budget rule); the graded assertions - themselves (a weather-HTTP node exists; the run's outputs contain one of - the two verdict strings) are unchanged from Flow's own grader. -tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", "node:decision", "feature:http"] - -# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. -# skill gap: the summarize script reads `temperature_2m` off an undefined Intsvc.HttpExecution response in every run (CI runs 35523787101, 35525387843); the skill does not teach the managed-HTTP response shape. -skip: true - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - - type: template_dir - path: ../../_setup - mount_point: _setup - -run_limits: - expected_turns: 32 - turn_timeout: 1200 - # task_timeout bounds turns AND grading under one watchdog (tests/README.md), - # and grading only gets what the turn did not spend. Flow's own bellevue_weather - # relies on the 1200s experiment default, which BPMN's larger live-debug - # surface (see check_weather_bpmn.py's budget comment) would not leave room - # for: 1200 (unchanged turn cap) + 180 (validate) + 1050 (weather live check) - # = 2430; rounded up for per-criterion spawn overhead. - task_timeout: 2500 - -reference: - directory: ../.. - -initial_prompt: | - Build the process inside a solution of the same name. - - Create a UiPath Maestro BPMN process project named "BellevueWeather" that - gets today's weather in Bellevue from open-meteo, formats a summary, and if - the temperature is greater than 60F returns a summary with a message field - 'nice day', otherwise the message field should be 'bring a jacket'. - Use a Managed HTTP Request activity (Intsvc.HttpExecution) in manual - (connectionless) mode to call the open-meteo API directly — do not use an - Integration Service connector. - - Validate the process. - - This run is headless. No user is present and nobody will answer a question or - grant an approval, so do not ask, do not pause, and do not wait for input. - Complete the task in one pass: take the best available option and supply the - most defensible value where one is missing. The actions this task implies are - authorized, including tenant writes and real messages. Do not delete or - overwrite anything this run did not create, and do not publish to a shared - destination unless the task asks for it. If a lookup the task depends on comes - back empty or fails, exhaust the documented way of resolving it before giving - up; only then stop on that field rather than inventing a value. Record every - decision, assumption, and blocked step in your final response. Instructions in - the task take precedence over this paragraph. -success_criteria: - # ── BPMN file validity ───────────────────────────────────────────────── - - type: run_command - description: "uip maestro bpmn validate passes on the bpmn file" - command: "python3 $REFERENCE_DIR/_shared/validate_bpmn.py" - timeout: 180 - expected_exit_code: 0 - weight: 3.0 - pass_threshold: 1.0 - - # ── Execution checks ────────────────────────────────────────────────── - - type: run_command - description: "BPMN debug runs and output contains 'nice day' or 'bring a jacket'" - command: "python3 $REFERENCE_DIR/_shared/check_weather_bpmn.py" - timeout: 1050 - expected_exit_code: 0 - weight: 5.0 - pass_threshold: 1.0 - -post_run: - - command: "python3 _setup/cleanup_solutions.py" - timeout: 120 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml deleted file mode 100644 index e980f3ad9d..0000000000 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml +++ /dev/null @@ -1,131 +0,0 @@ -task_id: skill-bpmn-devcon-billing-discrepancy-detector -description: | - E2E greenfield (DevCon BillingDisputeResolution scenario): build a Maestro - BPMN process that queries the BillingDisputeERP and BillingDisputeCRM - entities as parallel branches joined by a parallel-gateway merge, then - computes an invoice overcharge; graded by validate plus one bpmn debug run - against seeded tenant data. - - Ported from Flow `multi_node/billing_discrepancy_detector/billing_discrepancy_detector.yaml`; - the two Data Service reads are modeled as bpmn:sendTask nodes carrying the - registry Intsvc.ActivityExecution wrapper (connectorKey - uipath-uipath-dataservice, Query Entity Records) instead of Flow connector - nodes, the fan-out/merge as a bpmn:parallelGateway fork and join instead of - `core.logic.merge`, and `flow debug`'s single-call inline payload becomes an - ephemeral-solution import + `bpmn debug` + `debug-instance - variables-all`/`incidents` read. The live grading criterion's timeout is - raised from Flow's 600s to 1050s to cover the extra CLI steps (`solution - init`/`solution projects import`, separate `variables-all`/`incidents` - reads) that BPMN's live surface needs beyond Flow's single `flow debug` - call, matching multi_node/slack_weather_pipeline's precedent for the same - deviation. -tags: - - uipath-maestro-bpmn - - e2e - - mode:build - - lifecycle:execute - - shape:multi-node - - connector - - path-to-ga - -# Parked (skill gap). Criteria stay one-for-one with Flow; unskip when the gap closes. -# skill gap: the Data Service where clause built from a process variable is malformed in both runs (400 'Expected a field name expression', then an empty interpolated value; CI runs 35538279757, 35783045540). -skip: true - -run_limits: - max_turns: 120 - turn_timeout: 2400 - task_timeout: 3840 - -sandbox: - template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - - type: template_dir - path: ../../_setup - mount_point: _setup - -reference: - directory: ../.. - -post_run: - - command: python3 _setup/cleanup_solutions.py - timeout: 120 - -initial_prompt: | - Build a UiPath Maestro BPMN process "BillingDiscrepancyDetector" (in a - solution of the same name) with a manual start, that checks a disputed - invoice line against the contracted amount and pulls the customer's account - tier. - - - Inputs: process variables `invoiceNumber` (string), `accountNumber` - (string), `disputedLineNumber` (number), `disputedUnitPrice` (number), - `disputedQuantity` (number). - - Look up two existing Data Service entities: - - `BillingDisputeERP` — rows for `invoiceNumber`; each has a `lineNumber` - and contracted `amount`. - - `BillingDisputeCRM` — the account row for `accountNumber`; it has an - `accountTier`. - - The two lookups are independent. Run them as parallel branches that fan out - from the start event and rejoin at a parallel-gateway join before computing - the result — do not chain them one after the other. - - Compute: invoiced = `disputedUnitPrice` x `disputedQuantity`; contracted = - the `amount` of the ERP row whose `lineNumber` equals `disputedLineNumber`; - `totalOvercharge` = invoiced minus contracted when positive, else 0; - `discrepancyCount` = 1 when overcharge is positive, else 0. - - Outputs: `totalOvercharge` (number), `discrepancyCount` (number), - `matchedInvoiceNumber` (string, first ERP row), `accountTier` (string, from - CRM). - - Build both lookups on the UiPath Data Service Integration Service connector - (`uipath-uipath-dataservice`) Query Entity Records activity, not a native - Data Fabric node: the tenants this ships to do not all enable one. - - Validate the final BPMN file — the task is not complete until `uip maestro - bpmn validate` passes. Grading then runs the process under `bpmn debug` - against live tenant data, so build it to actually run, not merely validate. - - This run is headless. No user is present and nobody will answer a question or - grant an approval, so do not ask, do not pause, and do not wait for input. - Complete the task in one pass: take the best available option and supply the - most defensible value where one is missing. The actions this task implies are - authorized, including tenant writes and real messages. Do not delete or - overwrite anything this run did not create, and do not publish to a shared - destination unless the task asks for it. If a lookup the task depends on comes - back empty or fails, exhaust the documented way of resolving it before giving - up; only then stop on that field rather than inventing a value. Record every - decision, assumption, and blocked step in your final response. Instructions in - the task take precedence over this paragraph. -pre_run: - - command: 'python3 "_setup/preflight_connections.py" uipath-uipath-dataservice' - timeout: 120 - -success_criteria: - - type: run_command - description: uip maestro bpmn validate passes on the BPMN file - command: python3 $REFERENCE_DIR/_shared/validate_bpmn.py - timeout: 180 - expected_exit_code: 0 - weight: 3.0 - pass_threshold: 1.0 - - type: run_command - description: Check requires a parallel-gateway join (parallel branches) and debugs once; process computed overcharge=1610, count=1, invoice MCS-2026-04872, tier Enterprise - command: python3 $REFERENCE_DIR/_shared/check_billing_discrepancy_detector.py detector - timeout: 1050 - expected_exit_code: 0 - weight: 5.0 - pass_threshold: 1.0 - - type: run_command - description: 'Advisory: packed bindings_v2.json Connection resources exist and carry no stub UUIDs' - command: python3 $REFERENCE_DIR/_shared/check_billing_discrepancy_detector.py bindings - timeout: 120 - expected_exit_code: 0 - weight: 1.0 - pass_threshold: 0.0 - - type: run_command - description: 'Advisory: STRUCTURAL (covering): two DS queries (ERP + CRM) on MUTUALLY UNREACHABLE branches from the start event, converging on one bpmn:parallelGateway join that continues downstream; each filter computed from its own input; no answer literals; each output read from its own side' - command: python3 $REFERENCE_DIR/_shared/check_billing_discrepancy_detector.py advisory - timeout: 30 - expected_exit_code: 0 - weight: 1.0 - pass_threshold: 0.0 From 353e65b9175a067de6f7825ea0d41643f0a1e9e0 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Wed, 23 Sep 2026 12:28:20 -0700 Subject: [PATCH 21/35] test(bpmn): define the body reader and node-input helpers once in bpmn_check body_object/body_fields come over verbatim from #3476 so the two branches merge cleanly; eight graders drop their private copies of context_value, context_inputs, has_type, all_node_values and body readers in favour of the shared ones. Assertions unchanged; every touched grader replays green on its passing CI artifact. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../uipath-maestro-bpmn/_shared/bpmn_check.py | 118 ++++++++++++++++++ .../_shared/check_billing_invoice_lookup.py | 36 +----- .../_shared/check_databricks_query.py | 21 +--- .../_shared/check_df_smoke_error.py | 19 +-- .../_shared/check_enum_flow.py | 41 ++---- .../_shared/check_slack_http_fallback.py | 16 +-- .../_shared/check_slack_multiselect.py | 43 +------ .../check_testmanager_crud_grounded.py | 13 +- .../_shared/check_webhook_waitfor_parallel.py | 12 +- .../_shared/test_bpmn_check.py | 103 +++++++++++++++ 10 files changed, 247 insertions(+), 175 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py index 922cb9b50d..11233ce255 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py @@ -4,6 +4,7 @@ from __future__ import annotations import glob +import json import os import re import sys @@ -182,6 +183,123 @@ def all_node_values(element: ET.Element) -> list[str]: return values +def body_fields(element: ET.Element) -> list[ET.Element]: + """Every ``uipath:input`` under ``element`` carrying ``target="body"``. + + The raw elements, for a grader that needs to assert on the inputs + themselves (presence, count) rather than on the request body they encode. + Use :func:`body_object` for the body. + """ + return [inp for inp in context_inputs(element) if inp.attrib.get("target") == "body"] + + +def _input_payload(inp: ET.Element) -> str: + """One input's literal payload: CDATA/text first, then ``value``. + + Agents write the same field either way -- a CDATA body blob, or a + ``value`` attribute for a short scalar -- and the two never both carry + content on one input. + """ + text = (inp.text or "").strip() + if text: + return text + return (inp.attrib.get("value") or "").strip() + + +def _coerce_body_value(raw: str, declared: str, parsed, parsed_ok: bool): + """One per-field body input's value, coerced by its declared ``type``. + + ``number``/``integer`` become an ``int`` when the literal is integral and + a ``float`` otherwise, ``boolean`` becomes a ``bool``, ``json`` becomes the + parsed payload, and anything else stays the raw string. A value the + declared type cannot parse stays the raw string rather than failing -- + the grader that cares asserts the shape itself. + """ + if declared in ("number", "integer", "decimal", "double", "float", "long", "int"): + try: + number = float(raw) + except ValueError: + return raw + return int(number) if number.is_integer() else number + if declared in ("boolean", "bool"): + lowered = raw.lower() + if lowered in ("true", "false"): + return lowered == "true" + return raw + if declared == "json": + return parsed if parsed_ok else raw + return raw + + +def body_object(element: ET.Element) -> dict: + """The request body ``element``'s ``target="body"`` inputs encode, as a dict. + + Agents emit a connector request body in two shapes, both valid, and this + is the one definition that reads either (CI run 35777886090 produced the + second on tasks whose earlier runs produced the first): + + * **one JSON blob** -- ````, sometimes split across + several inputs whose objects merge; + * **one typed input per field** -- ````. + + Every ``target="body"`` input at any depth is read, in document order, + and classified: + + 1. An input named ``body`` is the whole request body. Its payload MUST + be a JSON object -- a payload that does not parse, or that parses to + something other than an object, fails the check, exactly as each + grader's own body parser did before this helper existed. + 2. Any other input whose payload is a JSON object (declared + ``type="json"`` or not) is merged into the body wholesale. + 3. Everything else is one field, keyed by ``name`` and coerced by + ``type`` (see :func:`_coerce_body_value`). + + Later inputs win on a key collision, matching the runtime's + last-one-wins behaviour. An expression payload (``=vars.X``, ``=js:...``) + stays the string it is whatever the declared type says, so a grader can + still tell a bound expression from a literal. + + Returns ``{}`` when there is no ``target="body"`` input; a grader that + must distinguish "no body input at all" from "an empty body" checks + :func:`body_fields` as well. + """ + body: dict = {} + for inp in body_fields(element): + raw = _input_payload(inp) + name = inp.attrib.get("name") or "" + declared = (inp.attrib.get("type") or "").strip().lower() + if not raw: + continue + try: + parsed = json.loads(raw) + parsed_ok = True + except (json.JSONDecodeError, ValueError): + parsed, parsed_ok = None, False + + if name == "body": + if not parsed_ok: + fail(f'target="body" input is not valid JSON: raw={raw!r}') + if not isinstance(parsed, dict): + fail( + f'target="body" JSON must be an object, got ' + f"{type(parsed).__name__}: raw={raw!r}" + ) + body.update(parsed) + continue + if parsed_ok and isinstance(parsed, dict): + body.update(parsed) + continue + if not name: + continue + if raw.startswith("="): + body[name] = raw + continue + body[name] = _coerce_body_value(raw, declared, parsed, parsed_ok) + return body + + def has_type(element: ET.Element, token: str) -> bool: """True when ``token`` appears anywhere in ``element``'s serialised XML.""" return token in ET.tostring(element, encoding="unicode") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py index 6dbad06022..50718ee0b9 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py @@ -133,6 +133,9 @@ from _shared.bpmn_check import ( # noqa: E402 NS, + all_node_values, + context_inputs, + context_value, elements, fail, find_bpmn_file, @@ -203,39 +206,10 @@ # check_df_contractregistry_crud_filters.py -- no new _shared module) ───── -def activity_root(task: ET.Element) -> ET.Element | None: - return task.find(".//uipath:activity", NS) - - -def all_inputs(task: ET.Element) -> list[ET.Element]: - root_el = activity_root(task) - if root_el is None: - return [] - return root_el.findall(".//uipath:input", NS) - - def input_val(inp: ET.Element) -> str: return inp.attrib.get("value") or (inp.text or "") -def context_value(task: ET.Element, name: str) -> str: - for inp in all_inputs(task): - if inp.attrib.get("name") == name: - return input_val(inp) - return "" - - -def all_node_values(task: ET.Element) -> list[str]: - values: list[str] = [] - for inp in all_inputs(task): - v = inp.attrib.get("value") - if v: - values.append(v) - if inp.text and inp.text.strip(): - values.append(inp.text.strip()) - return values - - def output_vars(task: ET.Element) -> list[str]: return [out.attrib["var"] for out in task.findall(".//uipath:output", NS) if out.attrib.get("var")] @@ -482,7 +456,7 @@ def lookup() -> None: def has_nonempty_filter(task: ET.Element) -> bool: - for inp in all_inputs(task): + for inp in context_inputs(task): name = (inp.attrib.get("name") or "").strip().lower() value = input_val(inp) if name in FILTER_INPUT_NAMES and str(value).strip() not in ("", "{}", "[]"): @@ -613,7 +587,7 @@ def derives_from(var_id: str, targets: set, var_sources: dict, extra_refs: dict, def filter_reference_ids(task: ET.Element) -> set: ids: set = set() - for inp in all_inputs(task): + for inp in context_inputs(task): value = input_val(inp) ids.update(re.findall(r"vars\.([A-Za-z0-9_]+)", value)) parsed = parse_json_maybe(value) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py index dcd7415b05..b38b6c32bd 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py @@ -33,7 +33,7 @@ I locate/parse .bpmn (file exists, well-formed XML) -> parse_bpmn("DatabricksQuery") T inputs collected at any depth under uipath:activity (Flow read a single JSON `inputs.detail` dict; BPMN may nest context/path/query/ - body inputs) -> node_inputs() / node_text_blob() walk `.//uipath:input` + body inputs) -> context_inputs() / node_text_blob() walk `.//uipath:input` T curated objectName/method spelling as an alternate to a literal operation-name token match (registry curated activity; no generic-CRUD alternate form exists for this operation) -> references_op() @@ -81,7 +81,7 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) -from _shared.bpmn_check import NS, elements, fail, parse_bpmn # noqa: E402 +from _shared.bpmn_check import NS, context_inputs, context_value, elements, fail, has_type, parse_bpmn # noqa: E402 JDBC_KEY = "uipath-uipath-jdbc" NATIVE_DATABRICKS_KEY = "uipath-databricks-databricks" @@ -93,23 +93,6 @@ def _normalize(text: str) -> str: return re.sub(r"[^a-z0-9]", "", text.lower()) -def has_type(el: ET.Element, token: str) -> bool: - return token in ET.tostring(el, encoding="unicode") - - -def node_inputs(task: ET.Element) -> list[ET.Element]: - # Any depth under the sendTask: agents sometimes nest body/query/path - # inputs inside uipath:context rather than as its siblings. - return task.findall(".//uipath:input", NS) - - -def context_value(task: ET.Element, name: str) -> str: - for inp in node_inputs(task): - if inp.attrib.get("name") == name: - return inp.attrib.get("value") or (inp.text or "") - return "" - - def node_text_blob(task: ET.Element) -> str: # Serialize only the uipath:activity payload (context + inputs/outputs), # not the enclosing bpmn:sendTask's own id/name attributes -- a node's diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index 95078183bd..42c4a6fcf5 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -57,6 +57,8 @@ from _shared.bpmn_check import ( # noqa: E402 NS, + context_inputs, + context_value, elements, fail, has_typed_uipath_extension, @@ -74,23 +76,8 @@ _LIST_OP_RE = re.compile(r"^list$", re.IGNORECASE) -def node_inputs(task: ET.Element) -> list[ET.Element]: - return task.findall(".//uipath:input", NS) - - -def input_val(inp: ET.Element) -> str: - return inp.attrib.get("value") or (inp.text or "") - - -def context_value(task: ET.Element, name: str) -> str: - for inp in node_inputs(task): - if inp.attrib.get("name") == name: - return input_val(inp) - return "" - - def mentions_entity(task: ET.Element, entity: str) -> bool: - for inp in node_inputs(task): + for inp in context_inputs(task): value = inp.attrib.get("value") or "" text = inp.text or "" if entity in value or entity in text: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py index 7bb89cc944..0f347ca9e7 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py @@ -57,7 +57,7 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) -from _shared.bpmn_check import context_inputs, elements, fail, parse_bpmn # noqa: E402 +from _shared.bpmn_check import body_object, context_inputs, elements, fail, parse_bpmn # noqa: E402 # Kept identical to Flow's own _EXPECTED_BODY (check_enum_flow.py:49-52): only # `to` and `importance` are asserted, matching the code, not its docstring. @@ -93,38 +93,15 @@ def check_structure(name_hint: str) -> None: def body_fields(node: ET.Element) -> dict[str, str] | None: - """The node's request body as a field->value map, in either accepted form. - - Form 1: exactly one `target="body"` input whose text/value parses as a - JSON object -- the canonical single-blob shape. - Form 2: one or more `target="body"` inputs, each named after the field it - carries -- the CLI manifest's stale separateInputs shape. - """ - body_inputs = [inp for inp in context_inputs(node) if inp.attrib.get("target") == "body"] - if not body_inputs: - return None - - if len(body_inputs) == 1: - inp = body_inputs[0] - raw = (inp.text or inp.attrib.get("value") or "").strip() - if raw: - try: - parsed = json.loads(raw) - except json.JSONDecodeError: - parsed = None - if isinstance(parsed, dict): - return {str(k): v for k, v in parsed.items()} - name = inp.attrib.get("name") - if name and name != "body": - return {name: raw} + """The node's request body as a field->value map, in either registry form + (one JSON blob, or one typed input per field), via bpmn_check.body_object; + None when the node has no target="body" input at all.""" + if not any(inp.attrib.get("target") == "body" for inp in context_inputs(node)): return None - - fields: dict[str, str] = {} - for inp in body_inputs: - name = inp.attrib.get("name") - if not name: - continue - fields[name] = (inp.attrib.get("value") or inp.text or "").strip() + fields = { + str(k): (v if isinstance(v, str) else json.dumps(v)) + for k, v in body_object(node).items() + } return fields or None diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py index 5790eb4046..98e52d117a 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py @@ -60,7 +60,7 @@ tolerance and same open GUESS as check_non_catalog_http_fallback.py) → ACTIVITY_TYPES tuple checked via has_type() T collect uipath:input elements at any depth under the node - → all_inputs() uses `.//uipath:input` + → context_inputs() uses `.//uipath:input` T endpoint value found in a context field, a sibling uipath:input's own value/text, or inside the target="body" JSON payload (the skill does not pin where the endpoint lands) @@ -102,7 +102,7 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) -from _shared.bpmn_check import NS, fail, find_bpmn_file, parse_bpmn, resolve_project # noqa: E402 +from _shared.bpmn_check import NS, context_inputs, fail, find_bpmn_file, has_type, parse_bpmn, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, @@ -146,19 +146,9 @@ # (LIVE-ADDENDUM: a property of the CLI surface, not of what is graded). -def has_type(el: ET.Element, token: str) -> bool: - return token in ET.tostring(el, encoding="unicode") - - -def all_inputs(el: ET.Element) -> list[ET.Element]: - # `.//` walks every uipath:input under the node at any depth -- agents - # sometimes nest body/query/path inputs inside uipath:context. - return el.findall(".//uipath:input", NS) - - def node_blob(el: ET.Element) -> str: parts = [ET.tostring(el, encoding="unicode")] - for inp in all_inputs(el): + for inp in context_inputs(el): parts.append(inp.attrib.get("name") or "") parts.append(inp.attrib.get("value") or "") parts.append(inp.text or "") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py index 0d99787097..5a3a616a58 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py @@ -30,10 +30,7 @@ BPMN puts the whole request in `uipath:input` elements) -> body_object() T the registry's two observed body forms both count: one whole-body `target="body"` JSON blob (name="body"), or one typed `target="body"` - input per field (name=) -- registry-workflow.md documents the - first as canonical; the second is what `bpmn_check.body_object()` on - main (not yet on this branch) tolerates, so it is reimplemented locally - here per the porting brief -> body_object() + input per field (name=) -> bpmn_check.body_object() DROPPED require_no_private_connector_values, require_sequence_integrity, require_di_for_visible_elements (not in Flow; `bpmn validate` criterion covers structure) @@ -55,6 +52,7 @@ from _shared.bpmn_check import ( # noqa: E402 NS, + body_object, context_value, elements, fail, @@ -68,10 +66,6 @@ _USERS_KEY_RE = re.compile(r"users(\[.*\])?") -def node_inputs(task: ET.Element) -> list[ET.Element]: - return task.findall(".//uipath:input", NS) - - def slack_tasks(root: ET.Element) -> list[ET.Element]: return [ task @@ -81,39 +75,6 @@ def slack_tasks(root: ET.Element) -> list[ET.Element]: ] -def body_object(task: ET.Element) -> dict: - """Merge every `target="body"` input on `task` into one dict, accepting - either registry-observed form: a single `name="body"` input holding the - whole request as a JSON object, or one typed input per field (its own - `name`, value/text is that field's value).""" - obj: dict = {} - for inp in node_inputs(task): - if inp.attrib.get("target") != "body": - continue - name = inp.attrib.get("name") - raw = inp.attrib.get("value") - if raw is None: - raw = inp.text - raw = (raw or "").strip() - if not raw: - continue - if name == "body": - try: - parsed = json.loads(raw) - except json.JSONDecodeError: - continue - if isinstance(parsed, dict): - obj.update(parsed) - continue - if not name: - continue - try: - obj[name] = json.loads(raw) - except json.JSONDecodeError: - obj[name] = raw - return obj - - def is_users_key(key) -> bool: """True for the 'users' multiselect key in any of its encodings. diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py index 2f3dbd60d5..927d0640ea 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_testmanager_crud_grounded.py @@ -54,7 +54,7 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) -from _shared.bpmn_check import NS, elements, parse_bpmn # noqa: E402 +from _shared.bpmn_check import NS, context_value, elements, has_type, parse_bpmn # noqa: E402 CONNECTOR_KEY = "uipath-uipath-testmanager" ACTIVITY_TYPE = "Intsvc.ActivityExecution" @@ -84,17 +84,6 @@ def load_json(path: str, what: str) -> dict: return {} # unreachable: fail() exits, but satisfies static analysis -def has_type(el: ET.Element, token: str) -> bool: - return token in ET.tostring(el, encoding="unicode") - - -def context_value(task: ET.Element, name: str) -> str: - for inp in task.findall(".//uipath:input", NS): - if inp.attrib.get("name") == name: - return (inp.attrib.get("value") or inp.text or "").strip() - return "" - - def connector_tasks(root: ET.Element) -> list[ET.Element]: return [ task diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py index 2739cf4c9f..170bf61046 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py @@ -59,6 +59,7 @@ from _shared.bpmn_check import ( # noqa: E402 NS, attr, + context_value, elements, fail, has_typed_uipath_extension, @@ -88,17 +89,6 @@ def manual_start_events(root: ET.Element) -> list[ET.Element]: return out -def node_inputs(el: ET.Element) -> list[ET.Element]: - return el.findall(".//uipath:input", NS) - - -def context_value(el: ET.Element, name: str) -> str: - for inp in node_inputs(el): - if inp.attrib.get("name") == name: - return inp.attrib.get("value") or (inp.text or "") - return "" - - def _is_empty_json_field(value: str) -> bool: """True when a headers/parameters context field carries nothing -- absent, blank, `{}`/`[]`, or `null` (an unfilled registry template placeholder is diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py index 39992e7b54..f38de59b5a 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py @@ -423,3 +423,106 @@ def test_find_bpmn_file_without_hint_skips_an_untyped_draft(tmp_path, monkeypatc (tmp_path / "ProjSolution" / "Proj" / "Proj.bpmn").write_text(typed.replace("Intsvc", "BPMN"), encoding="utf-8") with pytest.raises(SystemExit): bpmn_check.find_bpmn_file() + + +def _send_task(payload: str) -> ET.Element: + return ET.fromstring( + f'' + f"{payload}" + ) + + +def test_body_object_reads_a_single_json_blob() -> None: + task = _send_task( + '" + ) + assert bpmn_check.body_object(task) == { + "title": "T", + "score": 7.25, + "active": True, + "viewCount": 3, + } + + +def test_body_object_merges_several_json_blobs_later_wins() -> None: + task = _send_task( + '' + '' + '' + '' + ) + assert bpmn_check.body_object(task) == {"a": 1, "b": 2, "c": 3} + + +def test_body_object_coerces_per_field_inputs_by_type() -> None: + task = _send_task( + '' + '' + '' + '' + '' + '' + '' + '' + '' + '' + ) + body = bpmn_check.body_object(task) + assert body == { + "score": 7.25, + "viewCount": 350, + "rank": 4, + "ratio": 9, + "active": True, + "archived": False, + "tags": ["x", "y"], + "title": "AllTypesTest-Smoke", + "code": "350", + "releaseDate": "2024-03-10", + } + assert type(body["score"]) is float + assert type(body["viewCount"]) is int and type(body["ratio"]) is int + assert type(body["active"]) is bool + + +def test_body_object_mixes_a_blob_with_per_field_inputs() -> None: + task = _send_task( + '' + '' + '' + ) + assert bpmn_check.body_object(task) == {"title": "T", "score": 9} + + +def test_body_object_leaves_expressions_as_strings() -> None: + task = _send_task( + '' + '' + ) + assert bpmn_check.body_object(task) == { + "score": "=vars.Score", + "active": "=js:vars.flag", + } + + +def test_body_object_is_empty_without_body_inputs() -> None: + task = _send_task('') + assert bpmn_check.body_object(task) == {} + assert bpmn_check.body_fields(task) == [] + + +def test_body_object_fails_a_malformed_body_blob() -> None: + bad_json = _send_task( + '' + ) + with pytest.raises(SystemExit, match="not valid JSON"): + bpmn_check.body_object(bad_json) + + not_object = _send_task( + '' + ) + with pytest.raises(SystemExit, match="must be an object"): + bpmn_check.body_object(not_object) From 49869bd96666c261a52989c5474fce47574cf0fa Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Wed, 23 Sep 2026 12:30:53 -0700 Subject: [PATCH 22/35] test(bpmn): grader hygiene and a handoff that fits on one screen Answer the GUESS blocks CI has since settled (connector-mode HTTP is Intsvc.ActivityExecution; the Slack fallback uses emoji_list_GET; the Slack send objectName is send_message_to_channel_v2), wrap docstring and code lines over 120 characters, and cut the handoff to status, methodology, recipe, runtime facts, skill findings and resume steps; batch narratives live in the ledger. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/LIVE-HANDOFF.md | 131 ++++++------------ .../_shared/check_billing_invoice_lookup.py | 19 ++- .../_shared/check_complex_array.py | 6 +- .../_shared/check_databricks_query.py | 37 +++-- .../_shared/check_escalation_jira_ticket.py | 108 ++++++++++----- .../check_escalation_orchestrator_paths.py | 63 ++++++--- .../_shared/check_escalation_slack_alert.py | 37 +++-- .../_shared/check_jira_get_issue.py | 3 +- .../_shared/check_managed_http_fallback.py | 41 +++--- .../_shared/check_multiselect.py | 3 +- .../check_paginated_reference_lookup.py | 4 +- .../_shared/check_slack_http_fallback.py | 31 +++-- .../_shared/check_slack_multiselect.py | 12 +- .../_shared/check_slack_weather_pipeline.py | 3 +- 14 files changed, 279 insertions(+), 219 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index 532e2b2ad5..ee35defafd 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -1,100 +1,47 @@ # Flow → BPMN eval porting: live-tier handoff -Branch `test/bpmn-port-live` (PR #3502) holds the live-tier and field-shape ports that pass; branch `test/bpmn-port-parked` (stacked on it) holds the eight that do not, each `skip: true` with its evidence. The structural bucket is PR #3426. Methodology files sit beside this document; per-task history is `parity-ledger.md`. - -## Context - -The Flow suite (`tests/tasks/uipath-maestro-flow/`) has 131 tasks; the BPMN suite had 82. A per-task map put 62 Flow tasks in scope (`parity-ledger.md`): 29 structural (authoring + `validate`), 17 live (the Flow grader runs `flow debug`), 16 feasibility probes; 30 are Flow-only surface (IXP, evaluate, voice, conversational, bindings, native Data Fabric nodes). - -Reading the Flow graders during the loop reclassified four "structural" tasks as live (their graders debug): `slack_http_fallback`, `bellevue_weather_simulated`, `cli_dice_roller_simulated`, `slack_channel_description_simulated`; and four "live" tasks as structural (their graders never debug): `smoke_error`, `jdbc_databricks_query`, `webhook_waitfor_parallel`, `testmanager_crud_grounded` (self-reported result file, as Flow). The live bucket is therefore 21 tasks. - -## Status (as of 2026-09-22, after batch 15) - -Live bucket (21) and field-shape probes (9). Every row is a CI result on the alpha tenant, codex driver; green rows are in PR #3502, parked rows on `test/bpmn-port-parked`. - -| Task | State | Evidence | -|---|---|---| -| e2e/customer_escalation_triage | green (already on main) | pre-existing | -| e2e/jira_get_issue | green | run 35501830119 | -| e2e/jira_create_issue | green | run 35503094182 | -| e2e/escalation_jira_ticket | green | run 35503094182 | -| e2e/escalation_orchestrator_paths | green | run 35503094182 | -| e2e/escalation_slack_alert | green it.2 | run 35524004307 | -| multi_node/slack_channel_description | green it.2 | run 35525387843 | -| connector_features/generic_dynamic_node | green | run 35538279757 | -| connector_features/jdbc_databricks_query (structural) | green | run 35538279757 | -| connector_features/datafabric_connector/smoke_error (structural) | green | run 35538279757 | -| connector_features/slack_http_fallback | green | run 35783045540 | -| connector_trigger/webhook_waitfor_parallel (structural) | green | run 35783045540 | -| connector_features/testmanager_crud_grounded | green | run 35783045540 | -| interactive/cli_dice_roller_simulated | green | run 35783045540 | -| multi_node/billing_invoice_lookup | green | run 35785806030 | -| multi_node/slack_weather_pipeline | green it.3 | run 35785806030 | -| connector_features/enum | green | run 35789221753 | -| connector_features/query_params | green | run 35789221753 | -| connector_features/multiselect | green | run 35789221753 | -| connector_features/searchable_joins | green | run 35789221753 | -| connector_features/complex_array | green 0.875 (advisory miss only) | run 35789221753 | -| connector_features/path_params | green it.2 | run 35790934047 | -| connector_features/paginated_reference_lookup | green it.3 | run 35791969905 | -| multi_node/bellevue_weather | parked on `test/bpmn-port-parked`, skill gap | HttpExecution response shape (`temperature_2m` of undefined), runs 35523787101 + 35525387843 | -| interactive/bellevue_weather_simulated | parked on `test/bpmn-port-parked`, same gap | run 35785806030 | -| e2e/jira_search_triage | parked on `test/bpmn-port-parked`, platform/skill gap | multi-instance over a connector response, 400008 (run 35525387843) | -| e2e/jira_lifecycle | parked on `test/bpmn-port-parked`, needs live investigation | three different runtime failures; Flow flaky | -| multi_node/billing_discrepancy_detector | parked on `test/bpmn-port-parked`, skill gap | Data Service where clause from a process variable, two different 400s (runs 35538279757, 35783045540) | -| interactive/slack_channel_description_simulated | parked on `test/bpmn-port-parked`, skill gap | Slack channel pagination: page 1 only (run 35789221753); Flow 4/12 | -| connector_features/ceql_where | parked on `test/bpmn-port-parked`, surface gap | agent writes the CEQL `where` string, never Flow's filter tree (runs 35783045540, 35785806030) | -| connector_features/enhanced_enum | parked on `test/bpmn-port-parked`, skill gap | no WooCommerce connector node in either run (runs 35789221753, 35790934047) | - -Total: 23 green, 8 parked. Not started: the 4 probes that need tenant fixtures (billing_dispute_analyst / _resolution / _writer need a published agent substitute for Flow inline agents; single_node/file_attachment needs a file-typed process variable). - -## Probe bucket (16): pilot ported, 11 decided, 4 blocked - -`connector_features/ceql_where` is ported (commit fd6312fde), not yet run. The probe confirmed the filter carrier exists: `Intsvc.ActivityExecution` enrichment for the Entra `groups` List operation exposes a `where` parameter (type `query`, `FilterBuilder`, `hasCEQL: true`), and the CI-passing Data Fabric artifact carries the same tree as a `target="query" name="queryExpression" type="json"` input. As in Flow, the sandbox has no live tenant for enrichment, so the port grades the same standalone `where_detail.json` planning artifact plus the connector node and terminate end. No `bpmn validate` gate, matching Flow. Its one review flag: the groups-operation tolerance (objectName contains `group`, or `groups` + GET/list) has no CI-passed fixture yet. - -Verdict for the other 11 field-shape evals, from that probe: - -| Eval | Verdict | Carrier | -|---|---|---| -| path_params, query_params | portable, high confidence | `target="path"` / `target="query"` inputs, proven live | -| paginated_reference_lookup | portable | same query carrier (`pageSize`, `nextPage` seen in the Entra enrichment) | -| complex_array, multiselect | portable | nested JSON in the single `target="body"` CDATA, or array-valued query inputs | -| enum | portable | any literal `uipath:input` graded against the allowed set | -| enhanced_enum, searchable_joins | plausible, unverified | needs the `metadata` json context field or a `where`/`queryExpression` carrier; verify with a live `registry get` for the target connector first | -| generate_schema | uncertain | BPMN's schema surface is the opaque `jsonSchema` output contract that the skill says not to pre-empt offline; side-artifact grading or park | -| dtl_load_by_default ×2 | uncertain | `design.loadByDefault` is discovery-time metadata, not wire XML; portable only if Flow's grader checks the wire value | - -The remaining 4 probes need a published agent substitute (billing_dispute_analyst / _resolution / _writer use Flow inline agents) or a file-typed process variable (single_node/file_attachment). - -## Methodology (how each port is made) - -1. **One Sonnet subagent per task** with `PORTING-BRIEF.md` (faithfulness table, construct translation, mandatory grading contract), `BATCH1-ADDENDUM.md` (connector node forms, criterion translations, staging paths), and for live ports `LIVE-ADDENDUM.md`. Ports are born normalized: every grader assertion is tagged F (translation of a cited Flow assertion), I (artifact plumbing) or T (a listed tolerance); anything else is not written. `NORMALIZATION.md` is the pass that retro-fitted the first 15 ports to that rule. -2. **Review = mechanical cross-check + assertion map.** The cross-check (a small script used throughout; see the parity ledger notes) compares criteria type/order/weight/threshold/timeout, run_limits, tags, prompt literals, pre_run/post_run, relative paths against the Flow source. Deviations allowed: CLI verbs, grader implementation, live criterion timeouts sized to the priced BPMN CLI sequence, `task_timeout` = turn_timeout + grading + 60 when Flow's does not cover it. -3. **One CI dispatch per batch** (`gh workflow run run-coder-eval.yml --ref -f task_globs='…'`), default codex driver, alpha tenant. Results: `gh run download ` → `**/task.json` → `success_criteria_results`; artifacts under `**/00/artifacts/` hold the agent's `.bpmn` for grader regression. -4. **Iteration rule:** fix only port defects (grader over-strictness, wrong construct name); max 3 graded iterations; a repeat runtime/skill failure parks the task with evidence in the ledger. Never weaken a Flow assertion to go green. - -## Live-grader recipe (proven on 7 tasks) - -Canonical: `_shared/check_jira_get_issue.py`. Sequence via `_shared/bpmn_live.py`: `uip solution init /` under the sandbox CWD → `uip solution projects import --solutionFile <.uipx>` (assert sha256 of the imported `.bpmn` equals the submitted one) → `run_debug(project, inputs, log, timeout)` → `debug-instance variables-all ` → `debug-instance incidents `. Grade from `variables-all`: root scope Globals plus every element's `Outputs` (root public output values have read back `null`; element outputs are reliable; connector responses via `connector_response_values`). post_run `_setup/cleanup_solutions.py` sweeps the ephemeral solution. - -Budget: `_shared/test_criterion_budgets.py` prices every `run_debug` call; criterion `timeout` ≥ solution init 90 + import 180 + debug 480 + variables-all 120 + incidents 120 + margin 60 = 1050 for one debug run (per-case loops multiply the debug/variables terms; annotate `# budget-guard: manual xN` when the loop count is not a module literal). - -## Runtime and CLI facts learned (grade around them) - -- `uip maestro bpmn debug` polls at most 300 times; `run_debug` derives `--poll-interval` from its timeout and raises a clear failure on the CLI's poll-timeout envelope or a `subprocess.TimeoutExpired` (carrying the CLI's stderr log). -- Incident 102010 "Value cannot be null (Parameter 'Folder')" on a Slack activity = missing `folderKey` binding (agent defect). 102009 "Parameter '' has null or empty value" = missing activity parameter (agent defect). 400008 on a multi-instance marker = input collection over a connector response does not evaluate (platform/skill). -- Actions.HITL needs a deployed Action App to pass `bpmn validate`; HITL ports reach parity on every other criterion (decision: leave as is). -- The eval agent emits connector nodes in two forms (curated `objectName` with path/query inputs and a structured tree in `metadata`; or generic entity-CRUD with `objectName` = entity and the verb in `operation`/`method`). Graders accept both; inputs at any depth; body optional outside Create/Update. -- `_shared/validate_bpmn.py` validates every `.bpmn` in the sandbox: the old "any file validates" loop passed on a stray `bpmn init` scaffold. +Branch `test/bpmn-port-live` (PR #3502) holds the live-tier and field-shape ports that pass CI. Branch `test/bpmn-port-parked`, stacked on it, holds the eight that do not, each `skip: true` with its evidence in the YAML. The structural bucket is PR #3426. Per-task history, run ids and iteration notes are in `parity-ledger.md`; methodology in `PORTING-BRIEF.md`, `BATCH1-ADDENDUM.md`, `NORMALIZATION.md` and `LIVE-ADDENDUM.md` beside this file. + +## Where it stands (2026-09-23) + +| Bucket | Ported | Green | Parked | Not started | +|---|---|---|---|---| +| Live (Flow grader runs `flow debug`) | 21 | 15 | 6 | 0 | +| Field-shape probes (Integration Service parameters) | 9 | 7 | 2 | 0 | +| Fixture-dependent probes | 0 | 0 | 0 | 4 | + +Parked, with the gap each hits (details in the ledger): bellevue_weather and bellevue_weather_simulated (managed-HTTP response shape), jira_search_triage (multi-instance over a connector response, 400008), jira_lifecycle (three different runtime failures; Flow flaky too), billing_discrepancy_detector (Data Service where clause from a process variable), slack_channel_description_simulated (Slack channel pagination), ceql_where (Flow filter tree has no BPMN carrier), enhanced_enum (no WooCommerce connector node). + +Not started: billing_dispute_analyst / _resolution / _writer need a published agent to stand in for Flow's inline agents; single_node/file_attachment needs a file-typed process variable. Also undecided: generate_schema (opaque `jsonSchema` output contract) and dtl_load_by_default ×2 (design-time metadata, not wire XML). + +## Methodology + +1. One Sonnet subagent per task with the brief and addenda; ports are born normalized (every grader assertion tagged F, I or T in the module docstring; nothing else is written). +2. Review is a mechanical cross-check (criteria type/order/weight/threshold/timeout, run_limits, tags, prompt literals, pre_run/post_run, relative paths) plus reading the assertion map. Sanctioned deviations: CLI verbs, grader implementation, live criterion timeouts sized to the BPMN CLI sequence, `task_timeout` = turn_timeout + grading + 60. +3. One CI dispatch per batch: `gh workflow run run-coder-eval.yml --ref -f task_globs='…'`, codex driver, alpha tenant. Read results with `gh run download ` → `**/task.json`; agent artifacts under `**/00/artifacts/` feed grader regression replays. +4. Fix only port defects; three graded iterations at most; a repeat runtime or skill failure parks the task with evidence. Never weaken a Flow assertion. + +## Live-grader recipe + +Canonical: `_shared/check_jira_get_issue.py`, via `_shared/bpmn_live.py`: `uip solution init` under the sandbox CWD → `uip solution projects import` (assert the imported `.bpmn` sha256 equals the submitted one) → `run_debug(project, inputs, log, timeout)` → `debug-instance variables-all` → `debug-instance incidents`. Grade from root-scope globals plus every element's `Outputs` (root public outputs have read back null). post_run `_setup/cleanup_solutions.py` sweeps the ephemeral solution; a later criterion in the same task must pass `resolve_project(exclude_under=[LIVE_RUN_DIR])` so that import is not read as a second project. + +Budget: `_shared/test_criterion_budgets.py` prices every `run_debug` call; one debug run needs criterion `timeout` ≥ 90 + 180 + 480 + 120 + 120 + 60 = 1050. + +## Runtime and grader facts + +- `uip maestro bpmn debug` polls at most 300 times; `run_debug` sizes `--poll-interval` to 80% of its budget and surfaces the CLI's own timeout envelope. +- Incident 102010 "Parameter 'Folder' null" on a Slack node = missing `folderKey` binding; 102009 = missing activity parameter; 400008 on a multi-instance marker = input collection over a connector response. +- Connector nodes come in two forms (curated `objectName` with path/query inputs, or generic entity-CRUD with the verb in `operation`/`method`); request bodies in two forms (one JSON `target="body"` blob, or one typed input per field). Graders use `bpmn_check.body_object`, `context_value`, `context_inputs` and accept both. +- Managed HTTP is `Intsvc.HttpExecution` or `Intsvc.UnifiedHttpRequest`; connector-mode HTTP is `Intsvc.ActivityExecution`. Wait-for-event may be a `receiveTask` or an `intermediateCatchEvent`; classify by the `uipath:type` wrapper, never the BPMN tag. +- `find_bpmn_file` treats byte-identical copies as one artifact and drops drafts with no registry-typed node; `validate_bpmn.py` validates every `.bpmn` in the sandbox. +- Actions.HITL needs a deployed Action App to pass `bpmn validate`; the curated Data Fabric query template has no sort-field parameter. ## Skill findings to report upstream -Connector trigger parameters (Data Fabric entity, Outlook `parentFolderId`) are omitted on first attempts; `uip is triggers objects/describe` discovery is not taught; Slack `folderKey` binding omitted; `Intsvc.HttpExecution` response shape unclear to downstream scripts; Data Service query filter grammar (400 "Expected a field name expression"); Slack channel lookup misses existing channels (pagination); multi-instance over connector output fails at runtime; no "existing solutions → ask" greenfield rule; Actions.HITL requires a tenant Action App. +Slack channel-id resolution and pagination (three tasks); Slack `folderKey` binding omitted; connector trigger parameters (Data Fabric entity, Outlook `parentFolderId`) omitted and `uip is triggers` discovery not taught; managed-HTTP response shape unclear to downstream scripts; Data Service where-clause grammar from variables; multi-instance over connector output; fixed literal values parametrised into unbound variables; no "existing solutions → ask" greenfield rule; WooCommerce connector node not produced. ## Resuming -1. `git checkout test/bpmn-port-live` (stacked on the PR branch; rebase after the PR merges). -2. Dispatch the ten-task batch listed under "Batch 10 and after"; record results in `parity-ledger.md` and this table. -3. New ports follow the spawn pattern "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; create ; gates; report with assertion map". -4. Dispatch new ports as one batch; iterate per the rule; park with evidence. -5. Port the 7 field-shape probes marked portable; verify the carrier live before the 2 plausible ones; the 4 agent/file-typed probes need tenant fixtures first. +1. Rebase `test/bpmn-port-parked` onto main after #3502 merges; unskip a task when its gap closes and re-dispatch it. +2. For the four fixture-dependent probes, build the tenant fixture first (published agent, file-typed variable), then port with the same brief. +3. New ports: spawn "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; gates; report with assertion map", cross-check, dispatch as one batch, iterate per rule 4. diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py index 50718ee0b9..db5d8938b9 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py @@ -527,7 +527,9 @@ def bindings() -> None: doc = json.loads(archive.read(by_basename["bindings_v2.json"])) resources = [ - r for r in (doc.get("resources") or []) if isinstance(r, dict) and str(r.get("resource") or "").lower() == "connection" + r + for r in (doc.get("resources") or []) + if isinstance(r, dict) and str(r.get("resource") or "").lower() == "connection" ] # Flow's own check only requires non-empty + non-stub, never an exact # count (a connector port always needs >=1, unlike Flow's native-read @@ -536,7 +538,10 @@ def bindings() -> None: fail("packed bindings_v2.json declares no Connection resources") stubbed = [r.get("key") for r in resources if not is_real_connection_key(r.get("key"))] if stubbed: - fail(f"packed bindings_v2.json Connection keys must be real connection ids, not unresolved stubs: {stubbed}") + fail( + "packed bindings_v2.json Connection keys must be real connection ids, " + f"not unresolved stubs: {stubbed}" + ) print(f"OK: {len(resources)} connection binding(s) across bindings_v2.json, all non-stub") @@ -668,7 +673,10 @@ def advisory() -> None: http_nodes = [ task.attrib.get("id") for task in (*elements(root, "sendTask"), *elements(root, "serviceTask")) - if any(has_typed_uipath_extension(task, "activity", t) for t in ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest")) + if any( + has_typed_uipath_extension(task, "activity", t) + for t in ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") + ) ] if http_nodes: fail(f"the process calls Data Service over raw HTTP ({http_nodes}); use the connector action") @@ -695,7 +703,10 @@ def advisory() -> None: fail(f"the bpmn contains the literal {CANONICAL!r} -- the invoice number must be COMPUTED, never written in") for bad in RAW_INPUTS: if bad in raw: - fail(f"the bpmn contains the test input {bad!r} as a literal -- normalising by matching known inputs generalises to nothing") + fail( + f"the bpmn contains the test input {bad!r} as a literal -- " + "normalising by matching known inputs generalises to nothing" + ) # 5. outputs declared with the contract's names AND types (F: :98-108) outputs = declared_outputs(root) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py index 16babd6350..dc6e94074f 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_complex_array.py @@ -11,10 +11,12 @@ Assertion map (Flow -> BPMN): F criterion 1 flow_contains.py --flow-name ComplexArrayTest '"nodes"' - '"edges"' (file exists and is valid JSON) -> parse_bpmn("ComplexArrayTest") locates and parses the .bpmn + '"edges"' (file exists and is valid JSON) + -> parse_bpmn("ComplexArrayTest") locates and parses the .bpmn I locate/parse .bpmn with the ComplexArrayTest name hint -> parse_bpmn("ComplexArrayTest") F criterion 3 flow_contains.py --flow-name ComplexArrayTest - 'U0B7Y855WGG' 'U05Q882RHFZ' (advisory, threshold 0) -> check_ids(): same two literals searched in the located .bpmn's raw text (same weight/threshold) + 'U0B7Y855WGG' 'U05Q882RHFZ' (advisory, threshold 0) + -> check_ids(): same two literals searched in the located .bpmn's raw text (same weight/threshold) Usage: python3 check_complex_array.py # criterion 1: locate + parse diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py index b38b6c32bd..3c85f4fc1e 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py @@ -22,18 +22,26 @@ Assertion map (Flow -> BPMN): F check_databricks_query.py:66 assert_flow_uses_connector_target(JDBC_KEY) -- native `uipath.connector..*` node-type match - (flow_check.py:743-744) -> connector_tasks() finds >=1 bpmn:sendTask carrying Intsvc.ActivityExecution - with context connectorKey == uipath-uipath-jdbc + (flow_check.py:743-744) + -> connector_tasks() finds >=1 bpmn:sendTask carrying Intsvc.ActivityExecution + with context connectorKey == + uipath-uipath-jdbc F check_databricks_query.py:72-78 _references_op (node type OR - inputs.detail JSON contains "execute-query-synchronously") -> references_op() matches the catalog's curated objectName+method - ("query"/"POST") OR an "executequerysynchronously" token anywhere in - the node's inputs (name/value/text, any depth) + inputs.detail JSON contains "execute-query-synchronously") + -> references_op() matches the catalog's curated objectName+method + ("query"/"POST") OR an + "executequerysynchronously" token + anywhere in + the node's inputs (name/value/text, + any depth) F check_databricks_query.py:83-86 native-Databricks guard - (NATIVE_DATABRICKS_KEY not in any node type) -> no sendTask context connectorKey == uipath-databricks-databricks + (NATIVE_DATABRICKS_KEY not in any node type) + -> no sendTask context connectorKey == uipath-databricks-databricks I locate/parse .bpmn (file exists, well-formed XML) -> parse_bpmn("DatabricksQuery") T inputs collected at any depth under uipath:activity (Flow read a single JSON `inputs.detail` dict; BPMN may nest context/path/query/ - body inputs) -> context_inputs() / node_text_blob() walk `.//uipath:input` + body inputs) + -> context_inputs() / node_text_blob() walk `.//uipath:input` T curated objectName/method spelling as an alternate to a literal operation-name token match (registry curated activity; no generic-CRUD alternate form exists for this operation) -> references_op() @@ -46,13 +54,18 @@ exercises returns on connectorKey match alone (flow_check.py:743-744), with no binding check. (not enforced by Flow for this node shape) DROPPED require_no_private_connector_values / require_sequence_integrity - / require_di_for_visible_elements (not in Flow; `bpmn validate` criterion covers structure) + / require_di_for_visible_elements (not in Flow; `bpmn validate` criterion + covers structure) DROPPED a distinct "HTTP-fallback with connector auth" acceptance path (Flow's assert_flow_uses_connector_target has one for - core.action.http nodes) (no separate BPMN wrapper for connector-authenticated HTTP exists -- - see check_non_catalog_http_fallback.py's GUESS note: all catalog - connector activities, including HTTP-connector-mode ones, route - through the same Intsvc.ActivityExecution shell) + core.action.http nodes) (no separate BPMN wrapper for + connector-authenticated HTTP exists -- + confirmed by CI run 35500726138: + all catalog + connector activities, including + HTTP-connector-mode ones, route + through the same + Intsvc.ActivityExecution shell) Checks performed: 1. BPMN file exists and is well-formed XML. diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py index df30859435..612d7065f2 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py @@ -23,47 +23,82 @@ ticket, not a fabricated key. Assertion map (Flow -> BPMN): - I locate/parse .bpmn (no pinned path -- prompt names only the process) -> bpmn_check.parse_bpmn("EscalationJiraTicket") + resolve_project() - F check_escalation_jira_ticket.py:53-56 .flow references JIRA_KEY -> .bpmn source text contains JIRA_KEY substring - I uipath:variables output ids / Jira Create-Issue element ids / -> resolve_contract() (mirrors check_customer_escalation_behavior.py's Contract, Jira-only) + I locate/parse .bpmn (no pinned path -- prompt names only the process) + -> bpmn_check.parse_bpmn("EscalationJiraTicket") + resolve_project() + F check_escalation_jira_ticket.py:53-56 .flow references JIRA_KEY + -> .bpmn source text contains JIRA_KEY substring + I uipath:variables output ids / Jira Create-Issue element ids / + -> resolve_contract() (mirrors check_customer_escalation_behavior.py's Contract, Jira-only) scriptTask ids needed to address runtime evidence by id - T flow_check.run_debug(inputs=..., retries=1) -- single attempt, -> ephemeral solution import (sha256-pinned) + bpmn_live.run_debug(); - finalStatus == "Completed" checked inline FinalStatus/incidents asserted explicitly afterward (LIVE-ADDENDUM - canonical live pattern; bpmn debug already makes one attempt, no backoff) - F check_escalation_jira_ticket.py:64-68 no whole-run retries -> bpmn_live.run_debug has no retry/backoff parameter to begin with + T flow_check.run_debug(inputs=..., retries=1) -- single attempt, + -> ephemeral solution import (sha256-pinned) + bpmn_live.run_debug(); + finalStatus == "Completed" checked inline FinalStatus/incidents asserted + explicitly afterward (LIVE-ADDENDUM + canonical live pattern; bpmn debug + already makes one attempt, no + backoff) + F check_escalation_jira_ticket.py:64-68 no whole-run retries + -> bpmn_live.run_debug has no retry/backoff parameter to begin with (a retried Create-Issue would duplicate the ticket) - F check_escalation_jira_ticket.py:70-90 except-branch: on a debug -> on subprocess.TimeoutExpired from run_debug, scrape partial - timeout, best-effort scrape partial output for -\\d+ stdout/stderr for -\\d+ candidates, keep only ones whose - candidates, keep only ones owned (summary carries correlationId) summary carries correlationId, journal them, then fail - F check_escalation_jira_ticket.py:104-106 Jira Create-Issue node -> Jira Create-Issue element has exactly one Completed + F check_escalation_jira_ticket.py:70-90 except-branch: on a debug + -> on subprocess.TimeoutExpired from run_debug, scrape partial + timeout, best-effort scrape partial output for -\\d+ stdout/stderr for -\\d+ + candidates, keep only ones whose + candidates, keep only ones owned (summary carries correlationId) summary carries correlationId, journal + them, then fail + F check_escalation_jira_ticket.py:104-106 Jira Create-Issue node + -> Jira Create-Issue element has exactly one Completed specifically must have completed (not merely any Jira node) ElementExecutions record - F check_escalation_jira_ticket.py:108-111 candidate keys from -> connector_response_values() on the Create-Issue element's OWN - collect_outputs()/raw debug text, ISSUE_KEY_RE-shaped Outputs (element_output_records), value "key" - F check_escalation_jira_ticket.py:118-123 persist proven-created keys -> journal the harvested key to `.created_keys` BEFORE the tenant - BEFORE the fallible tenant reread reread, mirroring the exemplar's harvest-before-assert order - F check_escalation_jira_ticket.py:129-140 tenant reread: get_issue(), -> jira_is.get_issue(conn, key) (copied verbatim from Flow), summary + F check_escalation_jira_ticket.py:108-111 candidate keys from + -> connector_response_values() on the Create-Issue element's OWN + collect_outputs()/raw debug text, ISSUE_KEY_RE-shaped Outputs (element_output_records), + value "key" + F check_escalation_jira_ticket.py:118-123 persist proven-created keys + -> journal the harvested key to `.created_keys` BEFORE the tenant + BEFORE the fallible tenant reread reread, mirroring the exemplar's + harvest-before-assert order + F check_escalation_jira_ticket.py:129-140 tenant reread: get_issue(), + -> jira_is.get_issue(conn, key) (copied verbatim from Flow), summary summary contains correlationId, never an unrelated pre-existing issue contains correlationId - F check_escalation_jira_ticket.py:143-149 created key must be in the -> inherent by construction: jira_key is sourced ONLY from the - executed Create-Issue node's OWN output Create-Issue element's own Outputs (see harvest above) - F check_escalation_jira_ticket.py:154-155 assert_named_equals( -> declared output `jiraIssueKey` resolved via its uipath:variables - "jiraIssueKey", match, case_sensitive=True) output id (root scope Globals), compared case-sensitively - F check_escalation_jira_ticket.py:158-160 assert_named_equals per -> same, via declared output ids; severity case-insensitive, - seed["expected"] (severity case-insensitive, caseKey case-sensitive) caseKey case-sensitive (CASE_SENSITIVE set, ported verbatim) - F check_escalation_jira_ticket.py:162-186 severity AND engineeringNeeded -> same binding, over bpmn:scriptTask elements' own Outputs - bound to the SAME executed Script node, not split across two nodes (element_output_records); a node's response value-pool must - contain both the expected severity and engineeringNeeded value - T check_escalation_jira_ticket.py's severity/engineeringNeeded lookup -> matched against the VALUES of the scriptTask's own response - is by normalized field NAME (find_node_output_value) dict rather than by key name (mirrors check_customer_escalation_ - behavior.py's `carries()`, the CI-proven pattern for this exact - runtime shape, since BPMN scriptTask output key-naming is agent- - chosen and not part of the registry contract) - - DROPPED check_customer_escalation_behavior.py's OUTPUT_TYPES exact_type() (not in Flow -- assert_named_equals never type-checks output values) + F check_escalation_jira_ticket.py:143-149 created key must be in the + -> inherent by construction: jira_key is sourced ONLY from the + executed Create-Issue node's OWN output Create-Issue element's own Outputs + (see harvest above) + F check_escalation_jira_ticket.py:154-155 assert_named_equals( + -> declared output `jiraIssueKey` resolved via its uipath:variables + "jiraIssueKey", match, case_sensitive=True) output id (root scope Globals), + compared case-sensitively + F check_escalation_jira_ticket.py:158-160 assert_named_equals per + -> same, via declared output ids; severity case-insensitive, + seed["expected"] (severity case-insensitive, caseKey case-sensitive) caseKey case-sensitive + (CASE_SENSITIVE set, ported verbatim) + F check_escalation_jira_ticket.py:162-186 severity AND engineeringNeeded + -> same binding, over bpmn:scriptTask elements' own Outputs + bound to the SAME executed Script node, not split across two nodes (element_output_records); a node's + response value-pool must + contain both the expected severity + and engineeringNeeded value + T check_escalation_jira_ticket.py's severity/engineeringNeeded lookup + -> matched against the VALUES of the scriptTask's own response + is by normalized field NAME (find_node_output_value) dict rather than by key name (mirrors + check_customer_escalation_ + behavior.py's `carries()`, the + CI-proven pattern for this exact + runtime shape, since BPMN scriptTask + output key-naming is agent- + chosen and not part of the registry + contract) + + DROPPED check_customer_escalation_behavior.py's OUTPUT_TYPES exact_type() (not in Flow -- assert_named_equals + never type-checks output values) check on output values - DROPPED check_customer_escalation_behavior.py's Jira project.key/ (not in Flow -- Flow only checks the summary contains correlationId) + DROPPED check_customer_escalation_behavior.py's Jira project.key/ (not in Flow -- Flow only checks the + summary contains correlationId) issuetype.id remote-field re-check - DROPPED assert_live_target() tenant-lock guard (not in Flow's jira_is.py, which this task's _setup/jira_is.py is a - verbatim copy of; not adding it keeps that copy faithful) + DROPPED assert_live_target() tenant-lock guard (not in Flow's jira_is.py, which this + task's _setup/jira_is.py is a + verbatim copy of; not adding it keeps + that copy faithful) Budget arithmetic (LIVE-ADDENDUM): bpmn_live.debug_budget() default (480) + SOLUTION_INIT_TIMEOUT (90) + SOLUTION_IMPORT_TIMEOUT (180) + @@ -351,7 +386,8 @@ def main() -> None: detail.append(f"non-completed elements: {faulted}") if records: detail.append(f"incidents: {json.dumps(records)[:1500]}") - _fail(f"bpmn debug did not complete (finalStatus={final_status})" + ("; " + "; ".join(detail) if detail else "")) + suffix = "; " + "; ".join(detail) if detail else "" + _fail(f"bpmn debug did not complete (finalStatus={final_status})" + suffix) print("OK: bpmn debug completed") records = incident_records(incidents_data) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py index e34a6cb210..c9bd212792 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py @@ -15,23 +15,49 @@ when a message is really delivered, not merely reached. Assertion map (Flow -> BPMN): - F check_escalation_orchestrator_paths.py:106 assert_flow_uses_connector_target(SLACK_KEY) -> Contract.slack_ids non-empty (resolve_contract) - F check_escalation_orchestrator_paths.py:144 assert_connector_error_handlers(SLACK_KEY, ...) -> assert_error_handlers(): boundary errorEventDefinition on each Slack sendTask reaches a connector-free, acyclic, terminating path - F check_escalation_orchestrator_paths.py:145 assert_connector_send_identity(SLACK_KEY, "user", ...) -> assert_send_identity(): every Slack sendTask carries target=query name=send_as value=user - F check_escalation_orchestrator_paths.py:55 assert_named_equals(payload, name, expected, ...) -> public_value_present(): normalized value present among root Globals leaves + non-classifier element Outputs leaves (see NOTE below) - F check_escalation_orchestrator_paths.py:65 assert_slack_message_posted(payload, "slackMessageId", ...) -> assert_slack_posted(): fired Slack sendTask's own Outputs.response carries a ts-shaped id, the seeded channel, and correlationId + escalationPath in its text - F check_escalation_orchestrator_paths.py:78-90 completed_node_ids_of_type(payload,"script") + is_classifier -> classifier candidate set per case: a scriptTask whose OWN Outputs.response dict carries all four CLASSIFICATION_FIELDS matching expected - F check_escalation_orchestrator_paths.py:95 assert_node_type_executed(payload,"core.logic.decision") -> per-case fired_gateways non-empty - F check_escalation_orchestrator_paths.py:97-98 completed_node_ids_of_type(payload, "core.logic.decision"/"core.control.end") -> per-case fired_gateways / fired_ends via debug_data.ElementExecutions - F check_escalation_orchestrator_paths.py:132-139 common_classifier = intersection(per_case_classifiers) -> same intersection, same failure message shape - F check_escalation_orchestrator_paths.py:149-158 escalation/triage Slack-node disjointness -> same set overlap check - F check_escalation_orchestrator_paths.py:167-169 routing_decisions + assert_decision_branches_reach -> assert_decision_branches_reach(): exclusiveGateway's two outgoing sequenceFlows separate the fired Slack node sets (graph.reachable) - F check_escalation_orchestrator_paths.py:179 assert_distinct_branch_ends(escalation_nodes, triage_nodes) -> assert_distinct_branch_ends(): each fired-Slack-node set reaches its own endEvent, the other's not reachable (graph.reachable) - F check_escalation_orchestrator_paths.py:180-189 runtime escalation_ends/triage_ends disjointness -> same runtime disjointness check over per-case fired_ends - I locate/parse .bpmn, resolve project directory -> resolve_project() / resolve_contract() - I ephemeral solution init + import + sha256 pin, run bpmn debug per case, read variables-all -> LIVE-tier canonical pattern (bpmn_live.py; copied from e2e/customer_escalation_triage/check_customer_escalation_behavior.py) - T finalStatus/elementExecutions completion check (flow_check.run_debug does this inline for `flow debug`; `bpmn debug` does not) -> per-case FinalStatus check in verify_case() - T vars. / element Outputs in place of Flow's globals[".output"] -> element_output_records() / root_scope() (bpmn_live.py) + F check_escalation_orchestrator_paths.py:106 assert_flow_uses_connector_target(SLACK_KEY) + -> Contract.slack_ids non-empty (resolve_contract) + F check_escalation_orchestrator_paths.py:144 assert_connector_error_handlers(SLACK_KEY, ...) + -> assert_error_handlers(): boundary errorEventDefinition on each Slack sendTask reaches a connector-free, + acyclic, terminating path + F check_escalation_orchestrator_paths.py:145 assert_connector_send_identity(SLACK_KEY, "user", ...) + -> assert_send_identity(): every Slack sendTask carries target=query name=send_as value=user + F check_escalation_orchestrator_paths.py:55 assert_named_equals(payload, name, expected, ...) + -> public_value_present(): normalized value present among root Globals leaves + non-classifier element Outputs + leaves (see NOTE below) + F check_escalation_orchestrator_paths.py:65 assert_slack_message_posted(payload, "slackMessageId", ...) + -> assert_slack_posted(): fired Slack sendTask's own Outputs.response carries a ts-shaped id, the seeded + channel, and correlationId + escalationPath in its text + F check_escalation_orchestrator_paths.py:78-90 completed_node_ids_of_type(payload,"script") + is_classifier + -> classifier candidate set per case: a scriptTask whose OWN Outputs.response dict carries all four + CLASSIFICATION_FIELDS matching expected + F check_escalation_orchestrator_paths.py:95 assert_node_type_executed(payload,"core.logic.decision") + -> per-case fired_gateways non-empty + F check_escalation_orchestrator_paths.py:97-98 completed_node_ids_of_type(payload, + "core.logic.decision"/"core.control.end") + -> per-case fired_gateways / fired_ends via debug_data.ElementExecutions + F check_escalation_orchestrator_paths.py:132-139 common_classifier = intersection(per_case_classifiers) + -> same intersection, same failure message shape + F check_escalation_orchestrator_paths.py:149-158 escalation/triage Slack-node disjointness + -> same set overlap check + F check_escalation_orchestrator_paths.py:167-169 routing_decisions + assert_decision_branches_reach + -> assert_decision_branches_reach(): exclusiveGateway's two outgoing sequenceFlows separate the fired Slack node + sets (graph.reachable) + F check_escalation_orchestrator_paths.py:179 assert_distinct_branch_ends(escalation_nodes, triage_nodes) + -> assert_distinct_branch_ends(): each fired-Slack-node set reaches its own endEvent, the other's not reachable + (graph.reachable) + F check_escalation_orchestrator_paths.py:180-189 runtime escalation_ends/triage_ends disjointness + -> same runtime disjointness check over per-case fired_ends + I locate/parse .bpmn, resolve project directory + -> resolve_project() / resolve_contract() + I ephemeral solution init + import + sha256 pin, run bpmn debug per case, read variables-all + -> LIVE-tier canonical pattern (bpmn_live.py; copied from + e2e/customer_escalation_triage/check_customer_escalation_behavior.py) + T finalStatus/elementExecutions completion check (flow_check.run_debug does this inline for `flow + debug`; `bpmn debug` does not) + -> per-case FinalStatus check in verify_case() + T vars. / element Outputs in place of Flow's globals[".output"] + -> element_output_records() / root_scope() (bpmn_live.py) T NOTE (LIVE-ADDENDUM): a BPMN process with two end events may declare the SAME public output name twice (once per end event, elementId-scoped per structural-bpmn.md), and the branch that did not run has been observed to read back null even when the OTHER @@ -41,7 +67,8 @@ The classifier's OWN Outputs are excluded from that search (see CLASSIFICATION_FIELDS binding above) so a value that was only ever computed, never mapped to a public output, cannot satisfy this check by coincidence. - DROPPED Flow's exact_type check on public outputs (customer_escalation_triage-style) (not in this Flow task's grader) + DROPPED Flow's exact_type check on public outputs (customer_escalation_triage-style) (not in this + Flow task's grader) """ from __future__ import annotations diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py index aefec7855c..1d837a99e1 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py @@ -13,18 +13,31 @@ posted message and this task's Flow `post_run` never sweeps Slack either). Assertion map (Flow → BPMN): - F check_escalation_slack_alert.py:44 assert_flow_uses_connector_target(SLACK_KEY) -> resolve_contract(): ids_for(*SLACK_SEND) via index_runtime_connectors - F check_escalation_slack_alert.py:45 assert_connector_send_identity(key, "user", ...) -> resolve_contract(): send_as input == "user" on every slack_send_id - F check_escalation_slack_alert.py:58 run_debug(inputs=case["inputs"], retries=1) -> ephemeral `solution init` + `solution projects import` (sha256-pinned) + bpmn_live.run_debug(project_dir, inputs, log) - F flow_check.py:551-552 (run_debug's own exact-match status gate) -> assert_outcome(): FinalStatus == "Completed" - F check_escalation_slack_alert.py:63-64 assert_named_equals(payload, name, expected) -> assert_outcome(): assert_named_equals(actual, name, expected) per output (severity, engineeringNeeded, caseKey) - F flow_check.py:1445-1464 completed_node_ids_of_type(payload, "script") -> resolve_contract(): classifier_ids = every bpmn:scriptTask id - F check_escalation_slack_alert.py:66-88 sev_scripts / next_steps binding (same node, both fields) -> bind_classifier(): severity AND nextSteps must come from the SAME scriptTask's own output - F flow_check.py:1328-1334 slackMessageId ts-shape gate -> assert_outcome(): SLACK_TS_RE match on the exposed output - F flow_check.py:1339-1355 >=1 Completed connector node in the debug trace -> assert_outcome(): >=1 "completed" element among contract.slack_send_ids - F flow_check.py:1357-1384 mapped id must equal an executed send's OWN response ts -> assert_outcome(): match slackMessageId against a slack_send_ids element's response["ts"] - F flow_check.py:1386-1394 posted channel must equal expected_channel -> assert_outcome(): matched response["channel"] == SLACK_CHANNEL - F flow_check.py:1395-1405 must_contain: correlationId, severity, nextSteps -> assert_outcome(): matched response["message"]["text"] contains all three + F check_escalation_slack_alert.py:44 assert_flow_uses_connector_target(SLACK_KEY) + -> resolve_contract(): ids_for(*SLACK_SEND) via index_runtime_connectors + F check_escalation_slack_alert.py:45 assert_connector_send_identity(key, "user", ...) + -> resolve_contract(): send_as input == "user" on every slack_send_id + F check_escalation_slack_alert.py:58 run_debug(inputs=case["inputs"], retries=1) + -> ephemeral `solution init` + `solution projects import` (sha256-pinned) + bpmn_live.run_debug(project_dir, + inputs, log) + F flow_check.py:551-552 (run_debug's own exact-match status gate) + -> assert_outcome(): FinalStatus == "Completed" + F check_escalation_slack_alert.py:63-64 assert_named_equals(payload, name, expected) + -> assert_outcome(): assert_named_equals(actual, name, expected) per output (severity, engineeringNeeded, caseKey) + F flow_check.py:1445-1464 completed_node_ids_of_type(payload, "script") + -> resolve_contract(): classifier_ids = every bpmn:scriptTask id + F check_escalation_slack_alert.py:66-88 sev_scripts / next_steps binding (same node, both fields) + -> bind_classifier(): severity AND nextSteps must come from the SAME scriptTask's own output + F flow_check.py:1328-1334 slackMessageId ts-shape gate + -> assert_outcome(): SLACK_TS_RE match on the exposed output + F flow_check.py:1339-1355 >=1 Completed connector node in the debug trace + -> assert_outcome(): >=1 "completed" element among contract.slack_send_ids + F flow_check.py:1357-1384 mapped id must equal an executed send's OWN response ts + -> assert_outcome(): match slackMessageId against a slack_send_ids element's response["ts"] + F flow_check.py:1386-1394 posted channel must equal expected_channel + -> assert_outcome(): matched response["channel"] == SLACK_CHANNEL + F flow_check.py:1395-1405 must_contain: correlationId, severity, nextSteps + -> assert_outcome(): matched response["message"]["text"] contains all three I locate/parse .bpmn; resolve the project dir; import into an ephemeral solution (sha256-pinned); read runtime evidence via `debug-instance variables-all`/`incidents` — `bpmn debug` returns an instance id rather than inline variables, so this whole sequence stands in for Flow's single diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py index 7d3a2ea45d..54a6e4bd3e 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py @@ -17,7 +17,8 @@ directly. Assertion map (Flow → BPMN): - F check_jira_get_issue.py:43 JIRA_KEY not in raw ('"nodes"' marker dropped -- XML has no JSON "nodes" key, see I below) + F check_jira_get_issue.py:43 JIRA_KEY not in raw ('"nodes"' marker dropped -- XML has no JSON "nodes" key, see I + below) → JIRA_KEY not in raw text of the .bpmn F check_jira_get_issue.py:46 GET_OP_RE.search(raw) (Get-Issue op referenced) → find_get_issue_nodes(): a sendTask carrying Intsvc.ActivityExecution diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py index 1d214513ac..9a104977dd 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py @@ -13,17 +13,23 @@ Assertion map (Flow -> BPMN): F check_managed_http_fallback.py:110 "connector_key in full_text" (native -> whole-document blob search for connector present, whole-flow blob search) connector_key (main(), same leniency) - F check_managed_http_fallback.py:57-66 _check_path_params required -> FALLBACK_CHECKS["path_params"]["required"] - evidence ["engce-00000","method","get"] checked against each candidate node's blob - F check_managed_http_fallback.py:69-81 _check_query_params required -> FALLBACK_CHECKS["query_params"]["required"] + F check_managed_http_fallback.py:57-66 _check_path_params required + -> FALLBACK_CHECKS["path_params"]["required"] + evidence ["engce-00000","method","get"] checked against each candidate node's + blob + F check_managed_http_fallback.py:69-81 _check_query_params required + -> FALLBACK_CHECKS["query_params"]["required"] evidence [...] F check_managed_http_fallback.py:84-95 _check_enum required evidence -> FALLBACK_CHECKS["enum"]["required"] ["gmail.googleapis.com","method","post","importance","medium"] (Flow's (kept verbatim, including "medium" -- - own literal list, kept as-is even though the scenario asks for "high": see GUESS below) + own literal list, kept as-is even though the scenario asks for "high": see the note below) faithfulness contract forbids "fixing" a Flow assertion during a port) - F check_managed_http_fallback.py:32-40 http-fallback node selection -> is_managed_http_node(): Intsvc.HttpExecution - (flow node type == core.action.http.v2) or Intsvc.UnifiedHttpRequest (construct- - translation table's "Managed HTTP" row) + F check_managed_http_fallback.py:32-40 http-fallback node selection + -> is_managed_http_node(): Intsvc.HttpExecution + (flow node type == core.action.http.v2) or Intsvc.UnifiedHttpRequest + (construct- + translation table's "Managed HTTP" + row) I locate/parse .bpmn -> parse_bpmn(NAME_HINT) T curated OR generic connectorKey match for the native -> is_native_connector_node(): matches on branch (BATCH1-ADDENDUM: classify by connectorKey, not connectorKey alone, regardless of @@ -41,23 +47,20 @@ T collect uipath:input elements at any depth under the node -> context_inputs()/all_node_values() via node_blob() -GUESS (flag for reviewer): Flow's `_check_enum` requires the literal substring -"medium" in the HTTP fallback node's blob even though the enum.yaml scenario's -fixed value is "importance": "high". This reads like a latent bug or an -intentional check that the fallback node's schema/enum documentation (which -may enumerate "low, medium, high") is present, not that the request body VALUE -is "medium". The faithfulness contract's "Never" column forbids weakening a -criterion to make it pass, and there is no directive here to fix a suspected -Flow defect during a port, so this checker keeps "medium" as a required -substring verbatim. Report this upstream if the intent was "importance" + -"high". +Note on the enum fallback list: Flow's `_check_enum` requires the literal +substring "medium" in the HTTP fallback node's blob although the scenario's +fixed value is "importance": "high" (a schema/enum-documentation check, or a +latent Flow bug). Kept verbatim: the port never weakens or corrects a Flow +assertion. In CI the native-connector branch has carried every run so far +(run 35789221753), so the fallback list has not been exercised. No Flow assertions dropped: the native-connector short-circuit, all three required-evidence lists, and the http-fallback-node-must-exist precondition all have a BPMN counterpart above. Usage (from a task's run_command, cwd = sandbox root): - python3 $REFERENCE_DIR/_shared/check_managed_http_fallback.py + python3 $REFERENCE_DIR/_shared/check_managed_http_fallback.py + """ from __future__ import annotations @@ -95,7 +98,7 @@ GENERIC_HTTP_CONNECTOR_KEY = "uipath-uipath-http" # Required evidence per check, kept verbatim from Flow's own literal lists -# (see GUESS above re: "enum"'s "medium"). +# (see the module note on "enum"'s "medium"). FALLBACK_CHECKS: dict[str, dict[str, object]] = { "path_params": { "label": "Jira Get Issue path-params HTTP fallback", diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py index 3ea5e6eadd..3ad6b921da 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_multiselect.py @@ -10,7 +10,8 @@ Assertion map (Flow -> BPMN): F criterion 1 flow_contains.py '"nodes"' '"edges"' (file exists and is - valid JSON, name-agnostic) -> parse_bpmn() locates and parses the .bpmn, no name hint + valid JSON, name-agnostic) + -> parse_bpmn() locates and parses the .bpmn, no name hint I locate/parse .bpmn, no name hint (Flow's prompt names no project) -> parse_bpmn() """ diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py index 9c89d36862..a67ee38be7 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py @@ -72,8 +72,8 @@ ACTIVITY_TYPE = "Intsvc.ActivityExecution" CHANNEL_ID = "C083AN4E61E" -# GUESS: registry-workflow.md §3's example table names `send_message_to_channel_v2` -# for this connector/operation; match loosely on the concept (any separator, +# Loose match: the agent's objectName was `send_message_to_channel_v2` (CI run +# 35791969905); match on the concept (any separator, # optional "_v2"/"v2" suffix, either "message" spelling) rather than pin one # exact objectName spelling. SEND_MESSAGE_RE = re.compile(r"send[\s_-]*messages?[\s_-]*to[\s_-]*channel", re.IGNORECASE) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py index 98e52d117a..148aecbe8b 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py @@ -37,7 +37,7 @@ node `type` string: BPMN's registry wrapper is the SAME Intsvc.ActivityExecution tag for both a curated native activity (e.g. Get Channel Info) and an HTTP fallback, unlike Flow's node - `type`, which differs per shape (see GUESS) + `type`, which differs per shape (see the wrapper note) F check_slack_http_fallback.py:EMOJI_ENDPOINT blob search (json.dumps(fallback_nodes).lower()) → references_emoji_endpoint(): 'emoji.list' found anywhere across every uipath:input name/value/text at any depth under the node, @@ -57,7 +57,7 @@ e2e/jira_get_issue and multi_node/slack_channel_description) T curated (Intsvc.ActivityExecution) OR the connector-authenticated form of Intsvc.HttpExecution -- accept either wrapper tag (BATCH1-ADDENDUM; same - tolerance and same open GUESS as check_non_catalog_http_fallback.py) + tolerance; the wrapper question is answered in the note below) → ACTIVITY_TYPES tuple checked via has_type() T collect uipath:input elements at any depth under the node → context_inputs() uses `.//uipath:input` @@ -66,18 +66,11 @@ where the endpoint lands) → references_emoji_endpoint() -GUESS (flag for reviewer): same open question as check_non_catalog_http_fallback.py -- -registry-workflow.md documents `Intsvc.HttpExecution`'s `mode` context field as -hardcoded to "manual" (connectionless) with no documented connector-authenticated -alternative; the skill's own contract splits connector-mode HTTP as -`Intsvc.ActivityExecution` (a connector object/operation, here reused via a -generic "http-request" passthrough objectName under the Slack connectorKey -- -see the real CI-passing SpotifyProfileTest.bpmn fixture, which authors exactly -this shape for the non-catalog case) and reserves `Intsvc.HttpExecution` for -connectionless/manual calls only. To stay faithful to both the porting brief -and the skill's documented contract without inventing a hard requirement on one -wrapper tag, this checker classifies purely by `connectorKey` (+ the emoji.list -endpoint match), and accepts either wrapper tag carrying them. +Wrapper tag, answered by CI: the agent's fallback node is `Intsvc.ActivityExecution` +with connectorKey uipath-salesforce-slack and objectName `emoji_list_GET` (runs +35538279757, 35783045540); the skill reserves `Intsvc.HttpExecution` for +connectionless calls. Classification is by `connectorKey` plus the emoji-list +endpoint; the wrapper tuple keeps the managed-HTTP tags as a tolerance only. No Flow assertions dropped: node existence, the HTTP-fallback shape (translated via connectorKey since BPMN's node `type` cannot distinguish curated vs raw the @@ -102,7 +95,15 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) -from _shared.bpmn_check import NS, context_inputs, fail, find_bpmn_file, has_type, parse_bpmn, resolve_project # noqa: E402 +from _shared.bpmn_check import ( # noqa: E402 + NS, + context_inputs, + fail, + find_bpmn_file, + has_type, + parse_bpmn, + resolve_project, +) from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py index 5a3a616a58..2d0b90bd17 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py @@ -15,14 +15,18 @@ task-specific criteria (locate/parse, advisory id search). Assertion map (Flow -> BPMN): - F check_multiselect_flow.py:86-91 "slack" in node.type.lower() -> slack_tasks(): sendTask carrying Intsvc.ActivityExecution with connectorKey == uipath-salesforce-slack - F check_multiselect_flow.py:93-104 find_users(node.inputs), count/populated -> find_users(body_object(task)), same count/populated check + F check_multiselect_flow.py:86-91 "slack" in node.type.lower() + -> slack_tasks(): sendTask carrying Intsvc.ActivityExecution with connectorKey == uipath-salesforce-slack + F check_multiselect_flow.py:93-104 find_users(node.inputs), count/populated + -> find_users(body_object(task)), same count/populated check F check_multiselect_flow.py:26-48 parse_users(): native list, or a string wrapping an array literal (JSON array or - a `=js:(['U1','U2'])`-style expression) -> parse_users(): identical regex + ast.literal_eval tolerance + a `=js:(['U1','U2'])`-style expression) + -> parse_users(): identical regex + ast.literal_eval tolerance F check_multiselect_flow.py:51-57 is_users_key(): 'users' with an optional array-notation suffix -> is_users_key(): identical regex - F check_multiselect_flow.py:60-76 find_users(): recursive dict/list search -> find_users(): identical recursive search over the merged body object + F check_multiselect_flow.py:60-76 find_users(): recursive dict/list search + -> find_users(): identical recursive search over the merged body object I locate/parse .bpmn, no name hint (this script is shared by both tasks; complex_array's own project-name hint is handled by its own check_complex_array.py, not here) -> parse_bpmn() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py index 8ac0444916..d4eefe6b38 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py @@ -111,7 +111,8 @@ SLACK_CONNECTOR_KEY = "uipath-salesforce-slack" ACTIVITY_TYPE = "Intsvc.ActivityExecution" -# registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask; the eval agent emits either (CI run 35538279757). +# registry-workflow.md lists Intsvc.UnifiedHttpRequest beside HttpExecution for the managed HTTP sendTask; the eval +# agent emits either (CI run 35538279757). HTTP_TYPES = ("Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") HTTP_TYPE = HTTP_TYPES[0] WEATHER_HINTS = ("open-meteo", "openmeteoapis") From 1cae5ee3824dad5156e49737f5f06b3c225cc5d7 Mon Sep 17 00:00:00 2001 From: rockymadden Date: Thu, 24 Sep 2026 14:10:42 -0600 Subject: [PATCH 23/35] test(bpmn): fix the review findings on the live-tier port - Escalation graders unpack index_runtime_connectors' 3-tuple keys. - Jira create keys come only from the Create node's outputs in the seed project and are journaled before any status check; teardown reports failed deletes and keeps going; a debug timeout still journals its key. - body_object reads the one target="body" object the runtime consumes and raises BodyShapeError on several inputs or a non-object. - enum, path_params, generic_dynamic_node, billing, weather, databricks, df smoke_error, multiselect and webhook graders no longer accept what their Flow originals rejected. - The live sequence moves into bpmn_live (import_exact, fetch_variables, fetch_incidents, require_clean_run, output_leaves, input_echo_ids). - testmanager_crud_grounded is skipped again, as in Flow. - slack_channel_description task_timeout follows the stated formula. - Porting docs drop session notes and stale planning claims. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../_porting/LIVE-ADDENDUM.md | 17 +- .../_porting/LIVE-HANDOFF.md | 40 ++-- .../_porting/parity-ledger.md | 35 ++-- .../uipath-maestro-bpmn/_shared/bpmn_check.py | 129 ++++-------- .../uipath-maestro-bpmn/_shared/bpmn_live.py | 187 +++++++++++++++++ .../_shared/check_billing_invoice_lookup.py | 119 ++--------- .../_shared/check_channel_description.py | 120 +---------- .../_shared/check_databricks_query.py | 6 +- .../_shared/check_df_smoke_error.py | 16 +- .../_shared/check_dice_runs_simulated.py | 144 ++----------- .../_shared/check_enum_flow.py | 59 +++--- .../_shared/check_escalation_jira_ticket.py | 101 +++------- .../check_escalation_orchestrator_paths.py | 63 +----- .../_shared/check_escalation_slack_alert.py | 66 ++---- .../_shared/check_generic_dynamic_node.py | 189 ++++++++---------- .../_shared/check_jira_create_issue.py | 163 ++++----------- .../_shared/check_jira_get_issue.py | 152 +++++--------- .../_shared/check_managed_http_fallback.py | 16 +- .../_shared/check_path_param_value.py | 77 ++----- .../_shared/check_slack_http_fallback.py | 75 ++----- .../_shared/check_slack_multiselect.py | 16 +- .../_shared/check_slack_weather_pipeline.py | 168 +++++----------- .../_shared/check_webhook_waitfor_parallel.py | 13 +- .../_shared/test_bpmn_check.py | 74 +++---- .../_shared/test_bpmn_live.py | 89 +++++++++ .../complex_array/complex_array.yaml | 4 +- .../connector_features/enum/enum.yaml | 6 +- .../generic_dynamic_node.yaml | 10 +- .../multiselect/multiselect.yaml | 4 +- .../slack_http_fallback.yaml | 3 +- .../testmanager_crud_grounded.yaml | 9 +- .../escalation_orchestrator_paths.yaml | 2 +- .../e2e/jira_create_issue/_setup/jira_is.py | 25 ++- .../jira_create_issue/_setup/teardown_jira.py | 6 +- .../e2e/jira_get_issue/_setup/jira_is.py | 25 ++- .../jira_get_issue/_setup/teardown_jira.py | 6 +- .../cli_dice_roller_simulated.yaml | 11 +- .../slack_channel_description.yaml | 2 +- 38 files changed, 873 insertions(+), 1374 deletions(-) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md index fc540e7cc2..5ceddae683 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md @@ -4,14 +4,13 @@ Read `PORTING-BRIEF.md` (Grading contract is mandatory) and `BATCH1-ADDENDUM.md` ## The canonical live pattern (copy it) -`tests/tasks/uipath-maestro-bpmn/e2e/customer_escalation_triage/check_customer_escalation_behavior.py` + `escalation_is.py` on this branch is the one CI-proven live BPMN grader. Its sequence, all via `_shared/bpmn_live.py`: +`_shared/check_jira_get_issue.py` is the canonical live BPMN grader. Its sequence, all via `_shared/bpmn_live.py`: -1. `uip solution init /` under the sandbox CWD (ephemeral solution; kept under CWD so the standard post_run sweep finds the `.uipx`). -2. `uip solution projects import --solutionFile `; assert the imported `.bpmn` bytes equal the submitted ones (`sha256`). +1-2. `import_exact(bpmn_path, project_dir, LIVE_RUN_DIR / )`: ephemeral `uip solution init` under the sandbox CWD (so the standard post_run sweep finds the `.uipx`) + `uip solution projects import`, asserting the imported `.bpmn` bytes equal the submitted ones. 3. `debug_data, instance_id = run_debug(imported_project_dir, inputs, log_file, timeout=…)` — `bpmn debug` returns an instance id, not inline variables. `--inputs` JSON is honoured (the escalation run seeded correlationId this way and found it in Jira). -4. `uip maestro bpmn debug-instance variables-all ` → `root_scope(variables_data)` for root variables; `element_output_records(variables_data, element_id)` for a node's `Outputs`; `connector_response_values(outputs, name)` for connector response fields. -5. `uip maestro bpmn debug-instance incidents ` → `incident_records(...)`; a completed run with incidents is a failure. -6. Side-effect ids go to a flat journal the moment they are visible; post_run replays it (teardown) — mirror the Flow task's `_setup/teardown_*.py`. +4. `fetch_variables(id)` (`debug-instance variables-all`) → `root_scope(variables_data)` for root variables; `element_output_records(variables_data, element_id)` for a node's `Outputs`; `connector_response_values(outputs, name)` for connector response fields. +5. Side-effect ids go to a flat journal the moment they are visible, before any status check; post_run replays it (teardown) — mirror the Flow task's `_setup/teardown_*.py`. +6. `fetch_incidents(id)` then `require_clean_run(debug_data, evidence)`; a completed run with incidents is a failure. `debug_evidence(id)` does 4 and 6 together for a grader with no side effects. Known runtime facts (grade around them, do not fight them): - Element-level `Outputs` (a script task's mapped output, a connector's `response`) are reliably readable in `variables-all`. Root **public output values** have been read back as `null` even when correctly mapped (see `debug/live_debug_e2e/check_live_debug.py` docstring). So when Flow asserted "some output equals X" (`assert_output_value` / `assert_outputs_contain` over `variables.globals` + element outputs), translate to: search the value leaves of the root scope's variables AND every element's `Outputs` in `variables-all`; do not require the value on a root public output specifically. @@ -38,9 +37,3 @@ Add these T rows as needed and cite the Flow helper you translate: - `T assert_connector_error_handlers` → boundary error event / error path presence — only if Flow asserted it If a Flow assertion has no readable runtime evidence on the BPMN side (e.g. a root-output-only value that the runtime returns as null and no element output carries it), STOP and report "parked: " rather than loosening the assertion. - -## Learned on CI run 35503094182 - -- `uip maestro bpmn debug` polls at most 300 times; the wait is `300 × --poll-interval`. `bpmn_live.run_debug` now derives the interval from its `timeout` so the CLI keeps polling for the whole priced budget, and raises a clear `CheckFailure` on the CLI's poll-timeout envelope (`ErrorCode: timeout`, `Data.lastStatus`). Never pass a fixed small poll interval. -- Runtime incident 102010 with `ErrorDetails: "Value cannot be null. (Parameter 'Folder')"` on a Slack `Intsvc.ActivityExecution` means the activity lacks its `folderKey` binding — an authoring defect of the eval agent, not a grader defect. Do not widen a grader for it. -- Actions.HITL cannot pass `bpmn validate` without a deployed Action App binding (MISSING_BINDING with placeholder appId). HITL ports reach parity on every other criterion; record the validate gap in the description, do not work around it. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index ee35defafd..16f6e86953 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -1,37 +1,27 @@ -# Flow → BPMN eval porting: live-tier handoff +# Flow → BPMN eval porting: live tier -Branch `test/bpmn-port-live` (PR #3502) holds the live-tier and field-shape ports that pass CI. Branch `test/bpmn-port-parked`, stacked on it, holds the eight that do not, each `skip: true` with its evidence in the YAML. The structural bucket is PR #3426. Per-task history, run ids and iteration notes are in `parity-ledger.md`; methodology in `PORTING-BRIEF.md`, `BATCH1-ADDENDUM.md`, `NORMALIZATION.md` and `LIVE-ADDENDUM.md` beside this file. +Methodology is in `PORTING-BRIEF.md`, `BATCH1-ADDENDUM.md`, `NORMALIZATION.md` and `LIVE-ADDENDUM.md` beside this file; per-task history and run ids in `parity-ledger.md`. Ports that never went green live on `test/bpmn-port-parked`, each `skip: true` with its evidence. -## Where it stands (2026-09-23) +## Rules -| Bucket | Ported | Green | Parked | Not started | -|---|---|---|---|---| -| Live (Flow grader runs `flow debug`) | 21 | 15 | 6 | 0 | -| Field-shape probes (Integration Service parameters) | 9 | 7 | 2 | 0 | -| Fixture-dependent probes | 0 | 0 | 0 | 4 | - -Parked, with the gap each hits (details in the ledger): bellevue_weather and bellevue_weather_simulated (managed-HTTP response shape), jira_search_triage (multi-instance over a connector response, 400008), jira_lifecycle (three different runtime failures; Flow flaky too), billing_discrepancy_detector (Data Service where clause from a process variable), slack_channel_description_simulated (Slack channel pagination), ceql_where (Flow filter tree has no BPMN carrier), enhanced_enum (no WooCommerce connector node). - -Not started: billing_dispute_analyst / _resolution / _writer need a published agent to stand in for Flow's inline agents; single_node/file_attachment needs a file-typed process variable. Also undecided: generate_schema (opaque `jsonSchema` output contract) and dtl_load_by_default ×2 (design-time metadata, not wire XML). - -## Methodology - -1. One Sonnet subagent per task with the brief and addenda; ports are born normalized (every grader assertion tagged F, I or T in the module docstring; nothing else is written). -2. Review is a mechanical cross-check (criteria type/order/weight/threshold/timeout, run_limits, tags, prompt literals, pre_run/post_run, relative paths) plus reading the assertion map. Sanctioned deviations: CLI verbs, grader implementation, live criterion timeouts sized to the BPMN CLI sequence, `task_timeout` = turn_timeout + grading + 60. -3. One CI dispatch per batch: `gh workflow run run-coder-eval.yml --ref -f task_globs='…'`, codex driver, alpha tenant. Read results with `gh run download ` → `**/task.json`; agent artifacts under `**/00/artifacts/` feed grader regression replays. +1. Every grader assertion is tagged F (Flow translation), I (plumbing) or T (listed tolerance) in its module docstring. +2. Criteria keep Flow's type, order, weight and threshold. Sanctioned deviations: CLI verbs, grader implementation, live criterion timeouts sized to the BPMN CLI sequence, and `task_timeout` = turn_timeout + every criterion and pre_run timeout + 60. +3. Dispatch: `gh workflow run run-coder-eval.yml --ref -f task_globs='…'`, codex driver, alpha tenant. Read results with `gh run download ` → `**/task.json`; agent artifacts under `**/00/artifacts/` feed grader regression replays. 4. Fix only port defects; three graded iterations at most; a repeat runtime or skill failure parks the task with evidence. Never weaken a Flow assertion. ## Live-grader recipe -Canonical: `_shared/check_jira_get_issue.py`, via `_shared/bpmn_live.py`: `uip solution init` under the sandbox CWD → `uip solution projects import` (assert the imported `.bpmn` sha256 equals the submitted one) → `run_debug(project, inputs, log, timeout)` → `debug-instance variables-all` → `debug-instance incidents`. Grade from root-scope globals plus every element's `Outputs` (root public outputs have read back null). post_run `_setup/cleanup_solutions.py` sweeps the ephemeral solution; a later criterion in the same task must pass `resolve_project(exclude_under=[LIVE_RUN_DIR])` so that import is not read as a second project. +Canonical: `_shared/check_jira_get_issue.py`. From `_shared/bpmn_live.py`: `import_exact` (ephemeral `uip solution init` + `projects import`, sha256-pinned) → `run_debug(project, inputs, log, timeout)` in the grader itself, so `test_criterion_budgets.py` can price it → `debug_evidence` (`variables-all`, `incidents`) → `require_clean_run`. Grade with `output_leaves(variables, skip=input_echo_ids(process))`: root public outputs have read back null, and an unskipped input reads as an output. A grader with side effects journals their ids before `require_clean_run`, so a faulted run still cleans up. + +post_run `_setup/cleanup_solutions.py` sweeps the ephemeral solution; a later criterion in the same task passes `resolve_project(exclude_under=[LIVE_RUN_DIR])` so that import is not read as a second project. -Budget: `_shared/test_criterion_budgets.py` prices every `run_debug` call; one debug run needs criterion `timeout` ≥ 90 + 180 + 480 + 120 + 120 + 60 = 1050. +Budget: the guard enforces criterion `timeout` ≥ `debug_budget(...)` + 60. Size it to `debug_budget(...)` + `LIVE_OVERHEAD_SECONDS` (510) + 60, e.g. 480 + 510 + 60 = 1050 for one default debug. ## Runtime and grader facts -- `uip maestro bpmn debug` polls at most 300 times; `run_debug` sizes `--poll-interval` to 80% of its budget and surfaces the CLI's own timeout envelope. +- `uip maestro bpmn debug` polls at most 300 times; `run_debug` sizes `--poll-interval` to 80% of its budget and raises `CheckFailure` on timeout after writing the CLI output to its log file. - Incident 102010 "Parameter 'Folder' null" on a Slack node = missing `folderKey` binding; 102009 = missing activity parameter; 400008 on a multi-instance marker = input collection over a connector response. -- Connector nodes come in two forms (curated `objectName` with path/query inputs, or generic entity-CRUD with the verb in `operation`/`method`); request bodies in two forms (one JSON `target="body"` blob, or one typed input per field). Graders use `bpmn_check.body_object`, `context_value`, `context_inputs` and accept both. +- Connector nodes come in two forms: curated `objectName` with path/query inputs, or generic entity-CRUD with the verb in `operation`/`method`. A request body is one JSON `target="body"` input; several do not merge at runtime, and `bpmn_check.body_object` raises `BodyShapeError` on them. - Managed HTTP is `Intsvc.HttpExecution` or `Intsvc.UnifiedHttpRequest`; connector-mode HTTP is `Intsvc.ActivityExecution`. Wait-for-event may be a `receiveTask` or an `intermediateCatchEvent`; classify by the `uipath:type` wrapper, never the BPMN tag. - `find_bpmn_file` treats byte-identical copies as one artifact and drops drafts with no registry-typed node; `validate_bpmn.py` validates every `.bpmn` in the sandbox. - Actions.HITL needs a deployed Action App to pass `bpmn validate`; the curated Data Fabric query template has no sort-field parameter. @@ -40,8 +30,6 @@ Budget: `_shared/test_criterion_budgets.py` prices every `run_debug` call; one d Slack channel-id resolution and pagination (three tasks); Slack `folderKey` binding omitted; connector trigger parameters (Data Fabric entity, Outlook `parentFolderId`) omitted and `uip is triggers` discovery not taught; managed-HTTP response shape unclear to downstream scripts; Data Service where-clause grammar from variables; multi-instance over connector output; fixed literal values parametrised into unbound variables; no "existing solutions → ask" greenfield rule; WooCommerce connector node not produced. -## Resuming +## Not yet ported -1. Rebase `test/bpmn-port-parked` onto main after #3502 merges; unskip a task when its gap closes and re-dispatch it. -2. For the four fixture-dependent probes, build the tenant fixture first (published agent, file-typed variable), then port with the same brief. -3. New ports: spawn "read PORTING-BRIEF, BATCH1-ADDENDUM, LIVE-ADDENDUM; port ; gates; report with assertion map", cross-check, dispatch as one batch, iterate per rule 4. +billing_dispute_analyst / _resolution / _writer need a published agent to stand in for Flow's inline agents; single_node/file_attachment needs a file-typed process variable. Undecided: generate_schema (opaque `jsonSchema` output contract) and dtl_load_by_default ×2 (design-time metadata, not wire XML). diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index e537cbaf47..5e201dfb00 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -1,15 +1,6 @@ # Flow → BPMN eval parity map -Flow tasks: 131 · BPMN tasks: 82 · generated 2026-09-19 - -| Bucket | Count | -|---|---| -| Ported 1:1 | 21 | -| Covered by an equivalent BPMN task | 18 | -| Portable — structural (authoring + validate) | 29 | -| Portable — live (bpmn debug + tenant re-read) | 17 | -| Portable pending a feasibility probe | 16 | -| Not portable (Flow-only surface) | 30 | +Flow tasks: 131 · BPMN tasks: 82 ## Porting ledger @@ -55,7 +46,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `connector_features/jdbc_databricks_query/…` | same | PASS (run 35538279757) | structural | | `connector_features/slack-http-fallback/…` | `connector_features/slack_http_fallback/` | PASS (run 35783045540) | it.1 grader wanted `emoji.list`; the connector's generic resource is `emoji_list_GET` | | `connector_trigger/webhook_waitfor_parallel.yaml` | same | PASS (run 35783045540) | it.1 grader classified the wait by BPMN tag; agent emits `intermediateCatchEvent` + `Intsvc.WaitForEvent` and `Intsvc.UnifiedHttpRequest` | -| `connector_features/testmanager_crud_grounded/…` | same | PASS (run 35783045540) | it.1 two byte-identical `.bpmn` copies read as ambiguity | +| `connector_features/testmanager_crud_grounded/…` | same | SKIPPED, as in Flow | grades an agent-written `result.json`; unskip once it grades the live run | | `interactive/cli_dice_roller_simulated/…` | same | PASS (run 35783045540) | | | `multi_node/billing_invoice_lookup/…` | same | PASS (run 35785806030) | it.1 grader read its own ephemeral live solution as a second project; run 35783045540 was a platform 504 | | `multi_node/slack_weather_pipeline/…` | same | PASS it.3 (run 35785806030) | it.1 channel not found, it.2 wrong Slack connection bound (401); Flow passes 6/12 nightlies | @@ -80,7 +71,11 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `connector_features/ceql_where.yaml` | `connector_features/ceql_where/` | runs 35783045540, 35785806030: agent writes the connector's CEQL `where` string, never Flow's numeric-groupOperator tree; no BPMN carrier | | `connector_features/enhanced_enum.yaml` | `connector_features/enhanced_enum/` | runs 35789221753, 35790934047: no WooCommerce connector node produced | -## Ported 1:1 (21) +## Initial classification (2026-09-19) + +The buckets the ports were chosen from. The ledger above supersedes a row once its task is ported or parked. + +### Ported 1:1 (21) | Flow task | BPMN target / note | |---|---| @@ -96,7 +91,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `hitl/quality_04_brownfield_insert.yaml` | hitl/quality_brownfield_insert | | `hitl/smoke_02_completed_port_wired.yaml` | hitl/smoke_completed_wired | | `hitl/smoke_03_multi_outcome_routing.yaml` | hitl/smoke_multi_outcome_routing | -| `interactive/customer_escalation_triage/customer_escalation_triage.yaml` | e2e/customer_escalation_triage (live, this branch) | +| `interactive/customer_escalation_triage/customer_escalation_triage.yaml` | e2e/customer_escalation_triage (live) | | `multi_node/calculator/calculator.yaml` | multi_node/calculator (structural only; Flow is live) | | `multi_node/customer_escalation/customer_escalation.yaml` | multi_node/customer_escalation | | `multi_node/dice_roller/dice_roller.yaml` | multi_node/dice_roller (structural only; Flow is live) | @@ -106,7 +101,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `multi_node/reading_list/reading_list.yaml` | multi_node/reading_list (structural only; Flow is live) | | `multi_node/wiki_pageviews/wiki_pageviews.yaml` | multi_node/wiki_pageviews (structural only; Flow is live) | -## Covered by an equivalent BPMN task (18) +### Covered by an equivalent BPMN task (18) | Flow task | BPMN target / note | |---|---| @@ -129,7 +124,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `smoke/registry_discovery.yaml` | smoke/registry_discovery | | `smoke/scheduled_trigger.yaml` | single_node/timer_start | -## Portable — structural (authoring + validate) (29) +### Portable — structural (authoring + validate) (29) | Flow task | BPMN target / note | |---|---| @@ -144,7 +139,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `connector_features/datafabric_connector/trigger_lifecycle.yaml` | Intsvc.EventTrigger Record Created/Updated + downstream ActivityExecution | | `connector_features/drive_to_slack.yaml` | two ActivityExecution nodes; binary output chaining; validate-only | | `connector_features/non-catalog-http-fallback/non_catalog_http_fallback.yaml` | generic HTTP connector (uipath-uipath-http) ActivityExecution | -| `connector_features/slack-http-fallback/slack_http_fallback.yaml` | connector-mode HTTP fallback (Intsvc.HttpExecution on Slack connection) | +| `connector_features/slack-http-fallback/slack_http_fallback.yaml` | connector-mode HTTP fallback (Intsvc.ActivityExecution on Slack connection) | | `connector_features/testmanager_attachments/testmanager_attachments.yaml` | one ActivityExecution per Test Manager operation; validate-only | | `connector_features/testmanager_execution_results/testmanager_execution_results.yaml` | one ActivityExecution per Test Manager operation; validate-only | | `connector_features/testmanager_generic_records/testmanager_generic_records.yaml` | one ActivityExecution per Test Manager operation; validate-only | @@ -163,7 +158,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `single_node/outlook_trigger_inbox/outlook_trigger_inbox.yaml` | Intsvc.EventTrigger startEvent with fresh parentFolderId reference resolution | | `single_node/outlook_waitfor_email/outlook_waitfor_email.yaml` | Intsvc.WaitForEvent receiveTask with subject filter | -## Portable — live (bpmn debug + tenant re-read) (17) +### Portable — live (bpmn debug + tenant re-read) (17) | Flow task | BPMN target / note | |---|---| @@ -172,7 +167,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml` | JDBC ActivityExecution; live | | `connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml` | Test Manager create+get round trip; live debug | | `connector_trigger/webhook_waitfor_parallel.yaml` | parallelGateway + Intsvc.WaitForEvent (webhook) + HttpExecution self-trigger; live debug | -| `e2e/escalation_jira_ticket/escalation_jira_ticket.yaml` | sibling of the live escalation port on this branch; reuse escalation_is.py | +| `e2e/escalation_jira_ticket/escalation_jira_ticket.yaml` | sibling of the live escalation port; reuse escalation_is.py | | `e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml` | exclusiveGateway paths + Orchestrator.* nodes; live debug | | `e2e/escalation_slack_alert/escalation_slack_alert.yaml` | sibling of the live escalation port; reuse escalation_is.py | | `e2e/jira_create_issue/jira_create_issue.yaml` | Jira ActivityExecution; live debug + tenant re-read; teardown journal | @@ -185,7 +180,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `multi_node/slack_channel_description/slack_channel_description.yaml` | Intsvc.ActivityExecution Slack; Slack plumbing in e2e/customer_escalation_triage/escalation_is.py | | `multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml` | HttpExecution + Slack ActivityExecution; live debug | -## Portable pending a feasibility probe (16) +### Portable pending a feasibility probe (16) | Flow task | BPMN target / note | |---|---| @@ -206,7 +201,7 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `multi_node/billing_resolution_writer/billing_resolution_writer.yaml` | inline agent → StartAgentJob substitute | | `single_node/file_attachment/file_attachment.yaml` | file-typed process variable: confirm canvas variable contract supports it | -## Not portable (Flow-only surface) (30) +### Not portable (Flow-only surface) (30) | Flow task | BPMN target / note | |---|---| diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py index 11233ce255..e2de7f8258 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py @@ -72,9 +72,10 @@ def find_bpmn_file(name_hint: str | None = None) -> str: def _has_typed_node(path: str | Path) -> bool: try: - return "uipath:type value=" in Path(path).read_text(encoding="utf-8", errors="replace") - except OSError: + root = ET.parse(path).getroot() + except (OSError, ET.ParseError): return False + return any(node.attrib.get("value") for node in root.iter(f"{{{NS['uipath']}}}type")) def _sha256(path: str | Path) -> str: @@ -206,98 +207,46 @@ def _input_payload(inp: ET.Element) -> str: return (inp.attrib.get("value") or "").strip() -def _coerce_body_value(raw: str, declared: str, parsed, parsed_ok: bool): - """One per-field body input's value, coerced by its declared ``type``. - - ``number``/``integer`` become an ``int`` when the literal is integral and - a ``float`` otherwise, ``boolean`` becomes a ``bool``, ``json`` becomes the - parsed payload, and anything else stays the raw string. A value the - declared type cannot parse stays the raw string rather than failing -- - the grader that cares asserts the shape itself. - """ - if declared in ("number", "integer", "decimal", "double", "float", "long", "int"): - try: - number = float(raw) - except ValueError: - return raw - return int(number) if number.is_integer() else number - if declared in ("boolean", "bool"): - lowered = raw.lower() - if lowered in ("true", "false"): - return lowered == "true" - return raw - if declared == "json": - return parsed if parsed_ok else raw - return raw +class BodyShapeError(ValueError): + """The node's ``target="body"`` inputs are not one JSON object.""" def body_object(element: ET.Element) -> dict: - """The request body ``element``'s ``target="body"`` inputs encode, as a dict. - - Agents emit a connector request body in two shapes, both valid, and this - is the one definition that reads either (CI run 35777886090 produced the - second on tasks whose earlier runs produced the first): - - * **one JSON blob** -- ````, sometimes split across - several inputs whose objects merge; - * **one typed input per field** -- ````. - - Every ``target="body"`` input at any depth is read, in document order, - and classified: - - 1. An input named ``body`` is the whole request body. Its payload MUST - be a JSON object -- a payload that does not parse, or that parses to - something other than an object, fails the check, exactly as each - grader's own body parser did before this helper existed. - 2. Any other input whose payload is a JSON object (declared - ``type="json"`` or not) is merged into the body wholesale. - 3. Everything else is one field, keyed by ``name`` and coerced by - ``type`` (see :func:`_coerce_body_value`). - - Later inputs win on a key collision, matching the runtime's - last-one-wins behaviour. An expression payload (``=vars.X``, ``=js:...``) - stays the string it is whatever the declared type says, so a grader can - still tell a bound expression from a literal. - - Returns ``{}`` when there is no ``target="body"`` input; a grader that - must distinguish "no body input at all" from "an empty body" checks - :func:`body_fields` as well. + """The request body ``element``'s ``target="body"`` input encodes. + + The runtime reads exactly one ``target="body"`` input as the whole body; + several do not merge (skills/uipath-maestro-bpmn/references/registry-workflow.md, + "Body shape"). So the only gradeable shape is:: + + + + Returns ``{}`` when there is no ``target="body"`` input. Raises + :class:`BodyShapeError` for several inputs, or one whose payload is an + expression or anything but a JSON object. A ``=vars.X`` value inside the + object stays the string it is. """ - body: dict = {} - for inp in body_fields(element): - raw = _input_payload(inp) - name = inp.attrib.get("name") or "" - declared = (inp.attrib.get("type") or "").strip().lower() - if not raw: - continue - try: - parsed = json.loads(raw) - parsed_ok = True - except (json.JSONDecodeError, ValueError): - parsed, parsed_ok = None, False - - if name == "body": - if not parsed_ok: - fail(f'target="body" input is not valid JSON: raw={raw!r}') - if not isinstance(parsed, dict): - fail( - f'target="body" JSON must be an object, got ' - f"{type(parsed).__name__}: raw={raw!r}" - ) - body.update(parsed) - continue - if parsed_ok and isinstance(parsed, dict): - body.update(parsed) - continue - if not name: - continue - if raw.startswith("="): - body[name] = raw - continue - body[name] = _coerce_body_value(raw, declared, parsed, parsed_ok) - return body + fields = body_fields(element) + if not fields: + return {} + if len(fields) > 1: + names = [inp.attrib.get("name") or "" for inp in fields] + raise BodyShapeError( + f'{len(fields)} target="body" inputs {names}; the runtime does not ' + f"merge them, so the request carries only the last one" + ) + + raw = _input_payload(fields[0]) + if raw.startswith("="): + raise BodyShapeError(f'target="body" input is an expression, not a literal object: {raw!r}') + try: + parsed = json.loads(raw) + except (json.JSONDecodeError, ValueError): + raise BodyShapeError(f'target="body" input is not valid JSON: raw={raw!r}') from None + if not isinstance(parsed, dict): + raise BodyShapeError( + f'target="body" JSON must be an object, got {type(parsed).__name__}: raw={raw!r}' + ) + return parsed def has_type(element: ET.Element, token: str) -> bool: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py index fc1e18bde1..ffa320d589 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py @@ -25,6 +25,8 @@ import re import subprocess import xml.etree.ElementTree as ET +from collections.abc import Collection, Iterator +from dataclasses import dataclass from pathlib import Path from typing import Any @@ -492,3 +494,188 @@ def run_debug( + f"; stderr tail: {(completed.stderr or '')[-1500:]}" ) return debug_data, instance_id + + +SOLUTION_INIT_TIMEOUT = 90 +SOLUTION_IMPORT_TIMEOUT = 180 +VARIABLES_ALL_TIMEOUT = 120 +INCIDENTS_TIMEOUT = 120 + +# Sum of the unpriced CLI steps around one run_debug: init + import + +# variables-all + incidents. A criterion timeout adds this to debug_budget(). +LIVE_OVERHEAD_SECONDS = ( + SOLUTION_INIT_TIMEOUT + SOLUTION_IMPORT_TIMEOUT + VARIABLES_ALL_TIMEOUT + INCIDENTS_TIMEOUT +) + +COMPLETED_STATUSES = frozenset({"Completed", "Successful"}) + + +def import_exact(bpmn_path: Path, project_dir: Path, solution_dir: Path) -> Path: + """Import the submitted project into a fresh solution and return the + imported project directory, failing if the imported .bpmn bytes differ.""" + + original_hash = sha256(bpmn_path) + solution_dir.parent.mkdir(parents=True, exist_ok=True) + initialized = run_cli( + ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + ) + payload_data(initialized, "initialize ephemeral solution") + + solution_files = sorted(solution_dir.glob("*.uipx")) + if len(solution_files) != 1: + raise CheckFailure( + f"solution init produced {len(solution_files)} .uipx files in " + f"{solution_dir}, expected exactly one" + ) + + imported = run_cli( + [ + "uip", + "solution", + "projects", + "import", + str(project_dir.resolve()), + "--solutionFile", + str(solution_files[0]), + ], + timeout=SOLUTION_IMPORT_TIMEOUT, + ) + payload_data(imported, "import exact BPMN project") + + imported_project = solution_dir / project_dir.name + if sha256(imported_project / bpmn_path.name) != original_hash: + raise CheckFailure("solution import changed the submitted BPMN bytes") + print(f"OK: imported exact artifact (sha256={original_hash})") + return imported_project + + +@dataclass(frozen=True) +class DebugEvidence: + variables: Any + variables_text: str + incidents: list[Any] | None + incidents_raw: Any + + +def fetch_variables(instance_id: str) -> tuple[Any, str]: + """`debug-instance variables-all`: (data, raw stdout).""" + + variables = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], + timeout=VARIABLES_ALL_TIMEOUT, + ) + _payload, variables_data = payload_data(variables, "variables-all") + return variables_data, variables.stdout or "" + + +def fetch_incidents(instance_id: str) -> tuple[list[Any] | None, Any]: + """`debug-instance incidents`: (records, raw data).""" + + incidents = run_cli( + ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], + timeout=INCIDENTS_TIMEOUT, + ) + _payload, incidents_data = payload_data(incidents, "incidents") + return incident_records(incidents_data), incidents_data + + +def debug_evidence(instance_id: str) -> DebugEvidence: + """Variables then incidents. A grader with side effects calls the two + fetches itself, journaling between them.""" + + variables_data, variables_text = fetch_variables(instance_id) + incidents, incidents_data = fetch_incidents(instance_id) + return DebugEvidence(variables_data, variables_text, incidents, incidents_data) + + +def require_clean_run(debug_data: Any, evidence: DebugEvidence) -> str: + """Raise unless the run completed with no incidents; return FinalStatus.""" + + final_status = get_ci(debug_data, "FinalStatus") + if final_status not in COMPLETED_STATUSES: + detail = [] + faulted = [ + f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and str(get_ci(item, "Status") or "").casefold() != "completed" + ] + if faulted: + detail.append(f"non-completed elements: {faulted}") + if evidence.incidents: + detail.append(f"incidents: {json.dumps(evidence.incidents)[:1500]}") + raise CheckFailure( + f"final status was {final_status!r}" + + ("; " + "; ".join(detail) if detail else "") + ) + + if evidence.incidents is None: + raise CheckFailure( + f"incidents response has an unknown shape: {evidence.incidents_raw!r}" + ) + if evidence.incidents: + raise CheckFailure(f"unexpected incidents: {evidence.incidents}") + + print(f"OK: bpmn debug completed (FinalStatus={final_status}, no incidents)") + return final_status + + +def value_leaves(value: Any) -> Iterator[Any]: + if isinstance(value, dict): + for item in value.values(): + yield from value_leaves(item) + elif isinstance(value, list): + for item in value: + yield from value_leaves(item) + elif value is not None: + yield value + + +def output_leaves(variables_data: Any, skip: Collection[str] = ()) -> list[Any]: + """Leaves of the root Globals and every element's Outputs, minus the + globals and elements named in `skip` (see :func:`input_echo_ids`).""" + + skipped = {normalized_identifier(name) for name in skip} + globals_ = get_ci(root_scope(variables_data), "Globals", {}) or {} + leaves: list[Any] = [] + if isinstance(globals_, dict): + for name, value in globals_.items(): + if normalized_identifier(name) in skipped: + continue + leaves.extend(value_leaves(value)) + + for scope in get_ci(variables_data, "Variables", []) or []: + for element in get_ci(scope, "Elements", []) or []: + if normalized_identifier(get_ci(element, "ElementId")) in skipped: + continue + leaves.extend(value_leaves(get_ci(element, "Outputs", {}))) + return leaves + + +def output_haystack(variables_data: Any, skip: Collection[str] = ()) -> str: + return "\n".join(str(v) for v in output_leaves(variables_data, skip)).lower() + + +def input_echo_ids(process: ET.Element) -> set[str]: + """Where a process input shows up unchanged in variables-all: the + `uipath:input` ids and names, the variables a start event copies them + into verbatim, and the start events themselves. + + + # on Start_1 + -> {"input_Var_Amount", "Amount", "Var_Amount", "Start_1"} + """ + + ids: set[str] = set() + for variables in process.iter(q(UIPATH_NS, "variables")): + for node in variables.iter(q(UIPATH_NS, "input")): + ids.update(v for v in (node.attrib.get("id"), node.attrib.get("name")) if v) + + copies = {f"=vars.{identifier}" for identifier in ids} + for start in process.iter(q(BPMN_NS, "startEvent")): + if start.attrib.get("id"): + ids.add(start.attrib["id"]) + for mapping in start.iter(q(UIPATH_NS, "output")): + if (mapping.attrib.get("source") or "").strip() in copies and mapping.attrib.get("var"): + ids.add(mapping.attrib["var"]) + return ids diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py index db5d8938b9..12e4998315 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_billing_invoice_lookup.py @@ -146,12 +146,11 @@ from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, + debug_evidence, + import_exact, + input_echo_ids, + output_leaves, + require_clean_run, ) NAME_HINT = "BillingInvoiceLookup" @@ -177,11 +176,6 @@ FILTER_INPUT_NAMES = {"queryexpression", "where", "filter", "filtergroup", "filtervariables"} LIVE_RUN_DIR = Path("billing-invoice-lookup-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} # Worst-case wall clock, priced the way _shared/test_criterion_budgets.py # prices a run_debug(...) call inside a static loop: bpmn_live.debug_budget(180) @@ -312,34 +306,6 @@ def input_id_for_name(root: ET.Element, name: str) -> str | None: return None -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - -def output_leaves(variables_data: object) -> list: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has read back null even when correctly mapped - (LIVE-ADDENDUM), so the search is not scoped to one declared output - variable -- mirrors check_jira_get_issue.py's collect_output_haystack, but - keeps each leaf's native type so a numeric expectation is not spuriously - matched by a digit embedded in an unrelated string (flow_check.assert_output_value's - own reason for exact numeric equality, not substring, on numerics). - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return leaves - - def assert_output_value(leaves: list, expected) -> bool: """F: flow_check.assert_output_value -- exact-equal numerics, case-insensitive substring strings.""" for v in leaves: @@ -366,38 +332,17 @@ def lookup() -> None: fail("process declares no public uipath:input variable for the invoice number") var_name = input_names[0] - project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) - original_hash = sha256(Path(bpmn_path)) + process = root.find("bpmn:process", NS) + if process is None: + fail(f"{bpmn_path} has no bpmn:process") + echoes = input_echo_ids(process) + project_dir = resolve_project(os.path.basename(bpmn_path), exclude_under=[LIVE_RUN_DIR]) LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "BillingInvoiceLookupLiveEval" try: - initialized = run_cli(["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "BillingInvoiceLookupLiveEval" ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") for raw_value, label in CASES: inputs = {var_name: raw_value} @@ -408,39 +353,13 @@ def lookup() -> None: LIVE_RUN_DIR / f"debug-{label.replace(' ', '-')}.log", timeout=180, ) - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"[{label}] final status was {final_status!r}" + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"[{label}] incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"[{label}] unexpected incidents: {incidents_list}") - - leaves = output_leaves(variables_data) + evidence = debug_evidence(instance_id) + try: + require_clean_run(debug_data, evidence) + except CheckFailure as error: + raise CheckFailure(f"[{label}] {error}") from error + + leaves = output_leaves(evidence.variables, skip=echoes) if not assert_output_value(leaves, EXPECTED_INVOICE): raise CheckFailure(f"[{label}] no output equals expected {EXPECTED_INVOICE!r}") if not assert_output_value(leaves, EXPECTED_LINE_COUNT): diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py index bfde963a73..9b9d6478e0 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py @@ -1,9 +1,8 @@ #!/usr/bin/env python3 """SlackChannelDescription (BPMN): structural + live checks. -Ported from Flow `multi_node/slack_channel_description/_shared/check_channel_description.py` -(via `tests/tasks/uipath-maestro-flow/_shared/check_channel_description.py`): same -scenario (a manual-start process retrieves the channel description of +Ported from Flow `tests/tasks/uipath-maestro-flow/_shared/check_channel_description.py`: +same scenario (a manual-start process retrieves the channel description of #office-bellevue via the Slack Integration Service connector and outputs it), translated from a JSON node walk + inline `flow debug` payload to an XML walk over the registry-driven `Intsvc.ActivityExecution` connector shell (see @@ -56,7 +55,6 @@ from __future__ import annotations -import json import os import sys import xml.etree.ElementTree as ET @@ -70,12 +68,10 @@ from _shared.bpmn_live import ( # noqa: E402 CheckFailure, connector_context, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, + debug_evidence, + import_exact, + output_haystack, + require_clean_run, ) CONNECTOR_KEY = "uipath-salesforce-slack" @@ -90,11 +86,6 @@ ] LIVE_RUN_DIR = Path("slack-channel-description-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} # Worst-case wall clock this checker can spend, priced the way # _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug @@ -118,17 +109,6 @@ def _fail(msg: str) -> None: sys.exit(f"FAIL: {msg}") -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: """Every element carrying an Intsvc.ActivityExecution targeting connector_key. @@ -149,23 +129,6 @@ def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Elemen return found -def collect_output_haystack(variables_data: object) -> str: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- it mirrors Flow's own assert_outputs_contain(), - which flattens the whole outputs payload. Element Outputs include a - connector's nested `response` object, which _leaves() flattens along with - everything else. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return "\n".join(str(v) for v in leaves).lower() - - def main() -> None: bpmn_path = find_bpmn_file(NAME_HINT) raw = Path(bpmn_path).read_text(encoding="utf-8") @@ -187,81 +150,20 @@ def main() -> None: print(f"OK: bpmn references a {CONNECTOR_KEY} connector node") project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "SlackChannelDescriptionLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "SlackChannelDescriptionLiveEval" ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") debug_data, instance_id = bpmn_live.run_debug( imported_project, {}, LIVE_RUN_DIR / "debug.log" ) print(f"OK: debug completed (instance {instance_id})") - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + evidence = debug_evidence(instance_id) + require_clean_run(debug_data, evidence) - haystack = collect_output_haystack(variables_data) + haystack = output_haystack(evidence.variables) missing = [f for f in ADDRESS_FRAGMENTS if f.lower() not in haystack] if missing: _fail( diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py index 3c85f4fc1e..ee912e7788 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_databricks_query.py @@ -151,8 +151,10 @@ def main() -> None: "(objectName=query, method=POST)" ) - native_tasks = connector_tasks(root, NATIVE_DATABRICKS_KEY) - if native_tasks: + native_nodes = [ + node for node in root.iter() if context_value(node, "connectorKey") == NATIVE_DATABRICKS_KEY + ] + if native_nodes: fail( "process references the native Databricks connector " f"({NATIVE_DATABRICKS_KEY}) -- Databricks SQL must route through " diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index 42c4a6fcf5..c8c5a8bb12 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -29,7 +29,8 @@ F check_smoke_error.py:40-42 `len(good_queries) < 2` -> fail -> `good_queries < 2` check I locate/parse .bpmn -> parse_bpmn() T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() - T entity name anywhere in inputs -> mentions_entity() + T entity as the generic objectName, or an -> mentions_entity() + exact target="path" input value DROPPED topology/parallel-branch parsing (Flow's own grader does not parse it either -- see its docstring) DROPPED require_no_private_connector_values (not in Flow) DROPPED require_sequence_integrity (not in Flow; `bpmn validate` criterion covers structure) @@ -77,12 +78,13 @@ def mentions_entity(task: ET.Element, entity: str) -> bool: - for inp in context_inputs(task): - value = inp.attrib.get("value") or "" - text = inp.text or "" - if entity in value or entity in text: - return True - return False + if is_generic_entity_object(context_value(task, "objectName"), entity): + return True + return any( + inp.attrib.get("target") == "path" + and (inp.attrib.get("value") or inp.text or "").strip() == entity + for inp in context_inputs(task) + ) def is_generic_entity_object(object_name: str, entity: str) -> bool: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py index 9c605523e1..120577bee1 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py @@ -1,16 +1,13 @@ #!/usr/bin/env python3 """DiceRoller (BPMN, simulated): a scriptTask runs and produces an integer in [1, 6]. -Name-agnostic runtime checker for the simulated variant, modeled the same way -`_shared/check_weather_bpmn_simulated.py` relates to its own non-simulated -sibling: the simulated persona +Name-agnostic runtime checker for the simulated variant: the simulated persona (`uipath-maestro-flow/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml`) withholds "DiceRoller" until asked, so a correctly-built, differently named process must still be gradable -- no name hint is passed to `find_bpmn_file()`. Ported from Flow `_shared/check_dice_runs_simulated.py`, translated the same -way `_shared/check_weather_bpmn_simulated.py` and `_shared/check_jira_get_issue.py` -already translate a Flow live check: a JSON node-type scan + inline `flow +way `_shared/check_jira_get_issue.py` already translates a Flow live check: a JSON node-type scan + inline `flow debug` payload becomes an XML scan for a `bpmn:scriptTask` element (the `BPMN.ScriptTask` registry construct -- see skills/uipath-maestro-bpmn/references/structural-bpmn.md "Script tasks -- @@ -63,7 +60,7 @@ `bpmn debug` runs against an imported project, unlike `flow debug`, which runs directly against the discovered project directory -> LIVE-ADDENDUM canonical live pattern - (mirrors check_jira_get_issue.py / check_weather_bpmn_simulated.py) + (mirrors check_jira_get_issue.py) DROPPED require_no_private_connector_values / require_sequence_integrity / require_di_for_visible_elements / connection-binding checks -- not in Flow; the `bpmn validate` criterion covers structure @@ -89,35 +86,26 @@ from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, + debug_evidence, + import_exact, + output_leaves, + require_clean_run, ) LIVE_RUN_DIR = Path("dice-roller-simulated-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} ROLL_LO, ROLL_HI = 1, 6 INT_RE = re.compile(r"-?\d+") # Worst-case wall clock this checker can spend, priced the way # _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug -# call below passes no timeout/retries/backoff kwargs, so it prices at -# bpmn_live.debug_budget() == bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT (480). -# The surrounding CLI steps (solution init/import, variables-all, incidents) -# are not priced by that guard, so their sum is added by hand here and the -# criterion `timeout:` in cli_dice_roller_simulated.yaml documents the -# arithmetic (identical to check_jira_get_issue.py's / check_weather_bpmn_simulated.py's -# own budget): -# 90 (solution init) + 180 (solution import) + 480 (debug) -# + 120 (variables-all) + 120 (incidents) = 990 -# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1050 +# call below passes timeout=600 (Flow's own run_debug timeout), so it +# prices at bpmn_live.debug_budget(600) == 600. The surrounding CLI +# steps are not priced by that guard, so bpmn_live.LIVE_OVERHEAD_SECONDS is +# added by hand here and the criterion `timeout:` in +# cli_dice_roller_simulated.yaml documents the arithmetic: +# 90 (solution init) + 180 (solution import) + 600 (debug) +# + 120 (variables-all) + 120 (incidents) = 1110 +# + bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1170 # Flow's own criterion timeout (1320) already covers this, so it is kept # verbatim rather than raised. @@ -126,35 +114,6 @@ def _fail(msg: str) -> None: sys.exit(f"FAIL: {msg}") -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - -def collect_output_leaves(variables_data: object) -> list: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- it mirrors Flow's own - assert_output_int_in_range()/collect_outputs(), which flattens the whole - outputs payload (declared globals + element outputs) rather than the - entire debug response, so a stray digit inside an id, timestamp or other - metadata field elsewhere in the payload can never produce a false match. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return leaves - - def find_int_in_range(leaves: list, lo: int, hi: int) -> int | None: """First integer in [lo, hi] found in the stringified leaf values. @@ -189,81 +148,20 @@ def main() -> None: print(f"OK: bpmn has a scriptTask ({len(script_tasks)} found)") project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "DiceRollerSimulatedLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "DiceRollerSimulatedLiveEval" ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") debug_data, instance_id = bpmn_live.run_debug( - imported_project, {}, LIVE_RUN_DIR / "debug.log" + imported_project, {}, LIVE_RUN_DIR / "debug.log", timeout=600 ) print(f"OK: debug completed (instance {instance_id})") - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {incidents_list}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + evidence = debug_evidence(instance_id) + require_clean_run(debug_data, evidence) - leaves = collect_output_leaves(variables_data) + leaves = output_leaves(evidence.variables) roll = find_int_in_range(leaves, ROLL_LO, ROLL_HI) if roll is None: haystack = "\n".join(str(v) for v in leaves) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py index 0f347ca9e7..da29073786 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_enum_flow.py @@ -9,14 +9,10 @@ registry-driven `Intsvc.*` connector shell (see skills/uipath-maestro-bpmn/ references/registry-workflow.md §3 "Body shape"). -registry-workflow.md documents the canonical hand-authored body shape as +registry-workflow.md documents the only body shape the runtime consumes: exactly ONE `target="body"` input holding the whole request as a JSON CDATA -blob. The CLI manifest's stale `InputNotes`, however, still tell an author to -add one `uipath:input` per request field, and BATCH1-ADDENDUM's own lessons -record agents emitting both shapes for other Intsvc.ActivityExecution -parameters (curated separate path/query inputs vs. one JSON blob). Per this -task's instructions, both body forms are accepted here too: one JSON blob -input, or one typed `target="body"` input per field. +blob. Several `target="body"` inputs do not merge at runtime, so that shape +fails here (bpmn_check.body_object). Assertion map (Flow -> BPMN): F check_enum_flow.py:39-44 structure: flow exists, valid JSON, -> check_structure(): locate/parse .bpmn @@ -32,13 +28,9 @@ inputs.detail.bodyParameters, keep the best (fewest-missing) partial carrying a target="body" input, keep match for the failure message the best (fewest-missing) partial match I locate/parse .bpmn -> parse_bpmn(name_hint) - T body read in both forms: one JSON blob, or one typed -> body_fields(): single JSON-parseable - input per field (task instruction; mirrors the CLI manifest's stale target="body" input -> its parsed - separateInputs InputNotes as a real, if non-canonical, shape) object; multiple target="body" inputs, - each name=field -> {name: value} map - T `=`-prefixed expression values pass any type/value check -> body_fields_match() treats a value - (grading-contract-wide translation tolerance for expression strings) starting with "=" as satisfying its - expected field unconditionally + T `=`-prefixed expression value -> body_fields_match() accepts it only + when it carries the expected literal + (`=js:'high'`) No Flow assertions dropped: `structure`'s existence/parse check and `body_params`'s to/importance field match both have a BPMN counterpart above. @@ -57,7 +49,14 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) -from _shared.bpmn_check import body_object, context_inputs, elements, fail, parse_bpmn # noqa: E402 +from _shared.bpmn_check import ( # noqa: E402 + BodyShapeError, + body_object, + context_inputs, + elements, + fail, + parse_bpmn, +) # Kept identical to Flow's own _EXPECTED_BODY (check_enum_flow.py:49-52): only # `to` and `importance` are asserted, matching the code, not its docstring. @@ -93,24 +92,22 @@ def check_structure(name_hint: str) -> None: def body_fields(node: ET.Element) -> dict[str, str] | None: - """The node's request body as a field->value map, in either registry form - (one JSON blob, or one typed input per field), via bpmn_check.body_object; - None when the node has no target="body" input at all.""" + """The node's request body as a field->value map; None when the node has + no target="body" input at all. Raises BodyShapeError for a body the + runtime cannot consume.""" if not any(inp.attrib.get("target") == "body" for inp in context_inputs(node)): return None - fields = { + return { str(k): (v if isinstance(v, str) else json.dumps(v)) for k, v in body_object(node).items() } - return fields or None def body_fields_match(fields: dict[str, object]) -> list[str]: """Missing/mismatched field descriptions; empty list means OK. - A value beginning with "=" is an expression (`=vars.X`, `=js:...`) and - passes unconditionally -- the grading-contract-wide translation tolerance - for expression strings. + An expression (`=js:'high'`) passes only when it carries the expected + literal. """ lowered = {str(k).lower(): v for k, v in fields.items()} missing: list[str] = [] @@ -120,7 +117,7 @@ def body_fields_match(fields: dict[str, object]) -> list[str]: missing.append(f"{key}={expected!r} (missing)") continue actual_str = str(actual).strip() - if actual_str.startswith("="): + if actual_str.startswith("=") and expected.lower() in actual_str.lower(): continue if actual_str.lower() != expected.lower(): missing.append(f"{key}={expected!r} (got {actual!r})") @@ -133,13 +130,18 @@ def check_body_params(name_hint: str) -> None: checked_any = False candidates = [node for tag in CANDIDATE_TAGS for node in elements(root, tag)] for node in candidates: - fields = body_fields(node) + node_id = node.attrib.get("id", "") + try: + fields = body_fields(node) + except BodyShapeError as exc: + checked_any = True + best_missing = best_missing or [f"node {node_id!r}: {exc}"] + continue if fields is None: continue checked_any = True missing = body_fields_match(fields) if not missing: - node_id = node.attrib.get("id", "") print(f"OK: body payload on node {node_id!r} carries expected to/importance") return if best_missing is None or len(missing) < len(best_missing): @@ -147,9 +149,8 @@ def check_body_params(name_hint: str) -> None: if not checked_any: fail( - f"No node in {path} has a target=\"body\" input. Hand-authored connector " - f"nodes must carry either one JSON target=\"body\" input holding the " - f"whole request object, or one target=\"body\" input per field." + f"No node in {path} has a target=\"body\" input. A hand-authored connector " + f"node carries one JSON target=\"body\" input holding the whole request object." ) fail(f"target=\"body\" input found but missing or wrong fields: {best_missing}. BPMN: {path}") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py index 612d7065f2..ad19ec3df7 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py @@ -41,8 +41,8 @@ -> bpmn_live.run_debug has no retry/backoff parameter to begin with (a retried Create-Issue would duplicate the ticket) F check_escalation_jira_ticket.py:70-90 except-branch: on a debug - -> on subprocess.TimeoutExpired from run_debug, scrape partial - timeout, best-effort scrape partial output for -\\d+ stdout/stderr for -\\d+ + -> on CheckFailure from run_debug, scrape the debug log file + timeout, best-effort scrape partial output for -\\d+ for -\\d+ candidates, keep only ones whose candidates, keep only ones owned (summary carries correlationId) summary carries correlationId, journal them, then fail @@ -117,7 +117,6 @@ import json import os import re -import subprocess import sys import xml.etree.ElementTree as ET from dataclasses import dataclass @@ -135,18 +134,19 @@ from _shared.bpmn_live import ( # noqa: E402 BPMN_NS, CheckFailure, + DebugEvidence, connector_response_values, element_output_records, + fetch_incidents, + fetch_variables, get_ci, - incident_records, + import_exact, index_runtime_connectors, - payload_data, q, + require_clean_run, resolve_runtime_key, root_scope, - run_cli, run_debug, - sha256, UIPATH_NS, ) @@ -155,13 +155,8 @@ ISSUE_KEY_RE = re.compile(r"^[A-Z][A-Z0-9]*-\d+$") CASE_SENSITIVE = {"caseKey", "jiraIssueKey"} # opaque ids -- exact-case match OUTPUT_NAMES = ("severity", "caseKey", "jiraIssueKey") -COMPLETED_STATUSES = {"Completed", "Successful"} LIVE_RUN_DIR = Path("escalation-jira-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 MAX_CANDIDATE_ISSUE_READS = 2 # headroom; one Create-Issue node normally executes once @@ -236,8 +231,8 @@ def resolve_contract(root: ET.Element) -> Contract: connectors = index_runtime_connectors(process) jira_create_ids = tuple( element_id - for (key, route), element_ids in connectors.items() - if key == JIRA_CONNECTOR and JIRA_CREATE_OP in route + for (key, path, _object_name), element_ids in connectors.items() + if key == JIRA_CONNECTOR and JIRA_CREATE_OP in path for element_id in element_ids ) if not jira_create_ids: @@ -280,7 +275,7 @@ def _recover_partial_keys(project_key: str, correlation: str, raw_text: str) -> return [] try: conn = jira_is.connection_id() - except SystemExit: + except (SystemExit, Exception): # noqa: BLE001 -- connection_id() raises SystemExit return [] owned = [] for key in cands: @@ -309,31 +304,9 @@ def main() -> None: contract = resolve_contract(root) project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "EscalationJiraLiveEval" - initialized = run_cli(["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - _fail( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", "solution", "projects", "import", str(project_dir.resolve()), - "--solutionFile", str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "EscalationJiraLiveEval" ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - _fail("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") # No whole-run retries: this process CREATES a Jira issue, so a retried # whole run on a transient error could create a duplicate ticket that this @@ -342,23 +315,20 @@ def main() -> None: log_file = LIVE_RUN_DIR / "debug.log" try: debug_data, instance_id = run_debug(imported_project, seed["inputs"], log_file) - except subprocess.TimeoutExpired as exc: - partial = "".join( - s.decode() if isinstance(s, bytes) else (s or "") for s in (exc.stdout, exc.stderr) - ) + except CheckFailure as error: + try: + partial = log_file.read_text(encoding="utf-8") + except OSError: + partial = "" owned = _recover_partial_keys(project_key, correlation, partial) _journal(owned) _fail( - f"bpmn debug timed out after {exc.timeout}s" + str(error) + (f"; recorded this-run key(s) {owned} for teardown" if owned else "") ) print(f"OK: debug completed (instance {instance_id})") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") + variables_data, variables_text = fetch_variables(instance_id) # Journal the created key BEFORE any assertion -- it was created regardless # of the verdict below, and post_run's teardown_jira.py replays the journal @@ -366,35 +336,10 @@ def main() -> None: jira_keys = _harvest_jira_keys(contract, variables_data) _journal(jira_keys) - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, + incidents, incidents_data = fetch_incidents(instance_id) + require_clean_run( + debug_data, DebugEvidence(variables_data, variables_text, incidents, incidents_data) ) - _payload, incidents_data = payload_data(incidents, "incidents") - - final_status = get_ci(debug_data, "FinalStatus") - if final_status not in COMPLETED_STATUSES: - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - records = incident_records(incidents_data) - detail = [] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if records: - detail.append(f"incidents: {json.dumps(records)[:1500]}") - suffix = "; " + "; ".join(detail) if detail else "" - _fail(f"bpmn debug did not complete (finalStatus={final_status})" + suffix) - print("OK: bpmn debug completed") - - records = incident_records(incidents_data) - if records is None: - _fail(f"incidents response has an unknown shape: {incidents_data!r}") - if records: - _fail(f"unexpected incidents: {records}") # Execution evidence: the Jira CREATE-ISSUE element specifically must have # executed (not merely any Jira element -- a read op could surface an @@ -425,7 +370,7 @@ def main() -> None: for fields in [jira_is.get_issue(conn, k)] if fields is not None and correlation in str(fields.get("summary", "")) ] - _journal(owned or jira_keys) # re-journal narrowed to confirmed-owned when possible + _journal(owned) if not owned: _fail( f"none of {jira_keys} is a Jira issue whose summary contains " diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py index c9bd212792..b297d2f3c5 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py @@ -51,8 +51,7 @@ I locate/parse .bpmn, resolve project directory -> resolve_project() / resolve_contract() I ephemeral solution init + import + sha256 pin, run bpmn debug per case, read variables-all - -> LIVE-tier canonical pattern (bpmn_live.py; copied from - e2e/customer_escalation_triage/check_customer_escalation_behavior.py) + -> LIVE-tier canonical pattern (bpmn_live.import_exact() + run_debug()) T finalStatus/elementExecutions completion check (flow_check.run_debug does this inline for `flow debug`; `bpmn debug` does not) -> per-case FinalStatus check in verify_case() @@ -88,18 +87,21 @@ from _shared.bpmn_check import NS, attr, elements, resolve_project # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 BPMN_NS, + COMPLETED_STATUSES, CheckFailure, UIPATH_NS, + VARIABLES_ALL_TIMEOUT, connector_context, element_output_records, get_ci, + import_exact, index_runtime_connectors, payload_data, q, root_scope, run_cli, run_debug, - sha256, + value_leaves, ) SLACK_KEY = "uipath-salesforce-slack" @@ -118,11 +120,7 @@ NAMED_OUTPUT_FIELDS = CLASSIFICATION_FIELDS + ("caseKey",) LIVE_RUN_DIR = Path("escalation-orchestrator-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 DEBUG_TIMEOUT_SECONDS = 300 # literal on the run_debug call below -- the budget guard reads this statically -VARIABLES_ALL_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} _SLACK_TS_RE = re.compile(r"^\d{9,11}\.\d{4,6}$") @@ -164,8 +162,8 @@ def resolve_contract(path: Path) -> Contract: connectors = index_runtime_connectors(process) slack_ids = tuple( element_id - for (key, route), element_ids in connectors.items() - if key == SLACK_KEY and SLACK_SEND_PATH_HINT in route + for (key, connector_path, _object_name), element_ids in connectors.items() + if key == SLACK_KEY and SLACK_SEND_PATH_HINT in connector_path for element_id in element_ids ) if not slack_ids: @@ -332,17 +330,6 @@ def _loose_contains(haystack: str, needle: str) -> bool: return norm(needle) in norm(haystack) -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - def public_value_present( variables_data, expected, *, exclude_ids: tuple[str, ...], case_sensitive: bool ) -> bool: @@ -351,12 +338,12 @@ def public_value_present( leaf search rather than one pinned root-output id.""" target = normalized(expected, case_fold=not case_sensitive) - candidates = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) + candidates = list(value_leaves(get_ci(root_scope(variables_data), "Globals", {}))) for scope in get_ci(variables_data, "Variables", []) or []: for element in get_ci(scope, "Elements", []) or []: if get_ci(element, "ElementId") in exclude_ids: continue - candidates.extend(_leaves(get_ci(element, "Outputs", {}))) + candidates.extend(value_leaves(get_ci(element, "Outputs", {}))) return any(normalized(v, case_fold=not case_sensitive) == target for v in candidates) @@ -508,37 +495,9 @@ def main() -> None: assert_send_identity(process, contract.slack_ids) assert_error_handlers(root, process, contract.slack_ids) - original_hash = sha256(bpmn_path) - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "EscalationOrchestratorLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, + imported_project = import_exact( + bpmn_path, project_dir, LIVE_RUN_DIR / "EscalationOrchestratorLiveEval" ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / BPMN_NAME) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") escalation_fired: set = set() triage_fired: set = set() diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py index 1d837a99e1..e01e2cbf70 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_slack_alert.py @@ -67,17 +67,20 @@ from _shared.bpmn_live import ( # noqa: E402 BPMN_NS, CheckFailure, + INCIDENTS_TIMEOUT, + SOLUTION_IMPORT_TIMEOUT, + SOLUTION_INIT_TIMEOUT, + VARIABLES_ALL_TIMEOUT, + debug_evidence, element_output_records, get_ci, + import_exact, incident_records, index_runtime_connectors, - payload_data, q, resolve_runtime_key, root_scope, - run_cli, run_debug, - sha256, UIPATH_NS, ) @@ -99,10 +102,6 @@ # cover their sum plus bpmn_live.CRITERION_MARGIN_SECONDS. DEBUG_TIMEOUT_SECONDS = 480 # bpmn_live.DEBUG_BUDGET_DEFAULT_TIMEOUT, spelled out so the # static budget guard can price this call without following the import. -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 STEP_TIMEOUTS = ( DEBUG_TIMEOUT_SECONDS, SOLUTION_INIT_TIMEOUT, @@ -177,8 +176,8 @@ def resolve_contract(root: ET.Element) -> Contract: def ids_for(connector_key: str, path_needle: str) -> tuple[str, ...]: found = tuple( element_id - for (key, route), element_ids in connectors.items() - if key == connector_key and path_needle in route + for (key, path, _object_name), element_ids in connectors.items() + if key == connector_key and path_needle in path for element_id in element_ids ) if not found: @@ -407,58 +406,19 @@ def main() -> None: raise CheckFailure("seed.json must contain exactly one case") case = cases[0] - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "EscalationSlackAlertLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], - timeout=SOLUTION_INIT_TIMEOUT, + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "EscalationSlackAlertLiveEval" ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / Path(bpmn_path).name) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") debug_data, instance_id = run_debug( imported_project, case["inputs"], LIVE_RUN_DIR / "debug.log" ) print(f"OK: debug completed (instance {instance_id})") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, + evidence = debug_evidence(instance_id) + ts = assert_outcome( + contract, case, debug_data, evidence.variables, evidence.incidents_raw ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - - ts = assert_outcome(contract, case, debug_data, variables_data, incidents_data) print( f"OK: {case['name']} completed -- Sev1 + engineering classified, " f"correlationId preserved, and the Slack alert was posted (ts={ts})" diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py index d379ef2c3a..9739ecd074 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py @@ -51,10 +51,11 @@ → FinalStatus in COMPLETED_STATUSES and debug-instance incidents is empty F check_generic_dynamic_node.py:167-200 _assert_array_output(): an array-typed global output (empty allowed), reporting a flattened sys_id-bearing record if present - → array-typed value among the root scope's Globals AND every element's - Outputs (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read back null even - when correctly mapped, so the search is not scoped to one declared - output variable — mirrors check_jira_get_issue.collect_output_haystack) + → array held by a root Global that is a declared process output + (input echoes excluded); only when every declared output reads back + null (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read back null even + when correctly mapped), an array in the Outputs of the generic list + node(s) instead I locate/parse .bpmn, resolve project directory → bpmn_check.find_bpmn_file()/resolve_project() I ephemeral solution init + `solution projects import` + sha256 pin of the imported bytes @@ -89,14 +90,20 @@ from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 + BPMN_NS, + UIPATH_NS, CheckFailure, connector_context, + debug_evidence, + element_output_records, get_ci, - incident_records, - payload_data, + import_exact, + input_echo_ids, + normalized_identifier, + q, + require_clean_run, + resolve_runtime_key, root_scope, - run_cli, - sha256, ) CONNECTOR_KEY = "uipath-servicenow-servicenow" @@ -110,11 +117,6 @@ NAME_HINT = "AcrUserList" LIVE_RUN_DIR = Path("acr-user-list-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} # Worst-case wall clock this checker can spend, priced the way # _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug @@ -180,29 +182,63 @@ def _array_leaves(value): yield from _array_leaves(v) -def collect_array_candidates(variables_data: object) -> list[tuple[str, list]]: - """Array-typed values among the root scope's Globals AND every element's - Outputs. +def declared_outputs(process: ET.Element) -> list[tuple[str, ...]]: + """(id, name) of each process-level `uipath:output`, minus input echoes.""" + variables = process.find( + f"./{q(BPMN_NS, 'extensionElements')}/{q(UIPATH_NS, 'variables')}" + ) + if variables is None: + return [] + + echoes = {normalized_identifier(name) for name in input_echo_ids(process)} + outputs = [] + for node in variables.findall(q(UIPATH_NS, "output")): + keys = tuple( + key + for key in (node.attrib.get("id"), node.attrib.get("name")) + if key and normalized_identifier(key) not in echoes + ) + if keys: + outputs.append(keys) + return outputs - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable — mirrors check_jira_get_issue.py's - collect_output_haystack, adapted to look for an array shape rather than a - substring. - """ - candidates: list[tuple[str, list]] = [] + +def global_value(globals_: dict, identifier: str) -> object: + wanted = normalized_identifier(identifier) + if not any(normalized_identifier(key) == wanted for key in globals_): + return None + return resolve_runtime_key(globals_, identifier, "declared output") + + +def collect_array_candidates( + variables_data: object, + outputs: list[tuple[str, ...]], + list_node_ids: tuple[str, ...], +) -> list[tuple[str, list]]: + """Arrays held by declared output globals; the list nodes' Outputs only + when every declared output reads back null (LIVE-ADDENDUM).""" globals_ = get_ci(root_scope(variables_data), "Globals", {}) or {} - if isinstance(globals_, dict): - for name, value in globals_.items(): - for array in _array_leaves(value): - candidates.append((f"global {name!r}", array)) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - outputs = get_ci(element, "Outputs", {}) - for array in _array_leaves(outputs): - candidates.append( - (f"element {get_ci(element, 'ElementId')!r} Outputs", array) - ) + if not isinstance(globals_, dict): + globals_ = {} + + candidates: list[tuple[str, list]] = [] + read_back = False + for keys in outputs: + value = next( + (v for v in (global_value(globals_, key) for key in keys) if v is not None), + None, + ) + if value is None: + continue + read_back = True + if isinstance(value, list): + candidates.append((f"output global {keys[0]!r}", value)) + if read_back: + return candidates + + for outputs_record in element_output_records(variables_data, list_node_ids): + for array in _array_leaves(outputs_record): + candidates.append((f"list node Outputs {sorted(list_node_ids)}", array)) return candidates @@ -236,87 +272,34 @@ def main() -> None: ) print(f"OK: found {len(list_nodes)} generic list node(s) on objectName={OBJECT_NAME!r}") - project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) + process = root.find(q(BPMN_NS, "process")) + if process is None: + _fail(f"{bpmn_path} has no bpmn:process") + outputs = declared_outputs(process) + if not outputs: + _fail("process declares no uipath:output variable to surface the records") + list_node_ids = tuple(node.attrib["id"] for node in list_nodes if node.attrib.get("id")) + project_dir = resolve_project(os.path.basename(bpmn_path)) LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "AcrUserListLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "AcrUserListLiveEval" ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") - debug_data, instance_id = bpmn_live.run_debug( imported_project, {}, LIVE_RUN_DIR / "debug.log" ) print(f"OK: debug completed (instance {instance_id})") - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + evidence = debug_evidence(instance_id) + require_clean_run(debug_data, evidence) - candidates = collect_array_candidates(variables_data) + candidates = collect_array_candidates(evidence.variables, outputs, list_node_ids) if not candidates: _fail( "No output variable holds an array — the connector result was not " - "surfaced as a process output. Checked root Globals and every " - "element's Outputs." + "surfaced as a process output. Checked declared output globals " + f"{outputs} and, on null readback, the list node Outputs " + f"{list(list_node_ids)}." ) label, value = candidates[0] if value and all(isinstance(r, dict) for r in value) and any("sys_id" in r for r in value): diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py index 891b58a588..096be6c503 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py @@ -51,19 +51,17 @@ (`collect_outputs(payload)`) + a project-scoped regex scan of the raw debug payload (`get_last_debug_raw()`) - -> collect_candidate_keys(): value leaves of the - root scope's Globals AND every element's Outputs - in `debug-instance variables-all`, plus the same - project-scoped regex scan run over the raw - variables-all response text (the nearest BPMN - analog of Flow's "everything the debug call - returned" raw text) + -> collect_candidate_keys(): the same project-scoped + regex, narrowed to the Create-Issue element(s)' + own Outputs in `debug-instance variables-all`, so + no other CE issue key reaches the journal F check_jira_create_issue.py:62-63 no candidate keys -> fail -> same F check_jira_create_issue.py:66-77 tenant re-read via jira_is.get_issue(conn, key); first candidate whose `summary` equals the seed summary wins; confirmed key journaled for teardown - -> same logic, unchanged jira_is.py (task's own + -> same logic; `.created_keys` is rewritten to the + confirmed key only; jira_is.py (task's own `_setup/jira_is.py` copy, imported via the sandbox-mounted path since this checker lives in `_shared/`, not the task dir) @@ -76,14 +74,13 @@ e2e/customer_escalation_triage/ check_customer_escalation_behavior.py and check_jira_get_issue.py) - T journal every candidate key BEFORE the tenant-confirmation loop, - not only the one that matches (LIVE-ADDENDUM: "side-effect ids go + T journal every candidate key BEFORE the status/incident checks and + the tenant-confirmation loop (LIVE-ADDENDUM: "side-effect ids go to a flat journal the moment they are visible") -- Flow's own - script only journals the confirmed match, but a decoy issue - created from a wrong body is still a real tenant record that must - not leak just because its summary didn't match - -> `.created_keys` written right after candidate - collection, one key per line + script only journals the confirmed match, but an issue created by + a run that later faults is still a real tenant record + -> `.created_keys` written right after variables-all, + one key per line DROPPED require_no_private_connector_values / require_sequence_integrity / require_di_for_visible_elements / connection-binding checks -- not in Flow; the `bpmn validate` criterion covers structure @@ -105,13 +102,13 @@ from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, + DebugEvidence, connector_context, - get_ci, - incident_records, - payload_data, - root_scope, - run_cli, - sha256, + element_output_records, + fetch_incidents, + fetch_variables, + import_exact, + require_clean_run, ) JIRA_KEY = "uipath-atlassian-jira" @@ -120,15 +117,10 @@ GENERIC_CREATE_METHODS = {"POST"} ACTIVITY_TYPE = "Intsvc.ActivityExecution" NAME_HINT = "JiraCreateIssue" -ISSUE_KEY_RE = re.compile(r"^[A-Z][A-Z0-9]*-\d+$") SEED_LITERAL_FIELDS = ("project_key", "issuetype_id", "summary", "reporter_id") LIVE_RUN_DIR = Path("jira-create-issue-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} +JOURNAL = Path(".created_keys") # Worst-case wall clock this checker can spend, priced the way # _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug @@ -165,17 +157,6 @@ def _import_jira_is(): return jira_is -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - def is_create_issue_node(node_name: str, object_name: str, method: str) -> bool: """Curated (`curated_create_issue`) OR generic (`issue` + POST) form.""" if CREATE_OP_RE.search(object_name or "") or CREATE_OP_RE.search(node_name or ""): @@ -207,20 +188,16 @@ def find_create_issue_nodes(root: ET.Element) -> list[ET.Element]: def collect_candidate_keys( - variables_data: object, raw_variables_text: str, project: str + variables_data: object, create_ids: tuple[str, ...], project: str ) -> list[str]: - """Clean output leaves (any depth) + a project-scoped regex scan of the raw - variables-all response text (covers a key buried in a nested response - blob) -- mirrors Flow's `collect_outputs(payload)` + `get_last_debug_raw()` - dual candidate collection. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - cands = [s for leaf in leaves for s in [str(leaf).strip()] if ISSUE_KEY_RE.match(s)] - cands += re.findall(rf"\b{re.escape(project)}-\d+\b", raw_variables_text) - return list(dict.fromkeys(cands)) # de-dup, keep order + outputs = element_output_records(variables_data, create_ids) + cands = re.findall(rf"\b{re.escape(project)}-\d+\b", json.dumps(outputs, default=str)) + return list(dict.fromkeys(cands)) + + +def _journal(keys: list[str]) -> None: + if keys: + JOURNAL.write_text("\n".join(keys) + "\n") def main() -> None: @@ -256,97 +233,35 @@ def main() -> None: print("OK: bpmn references the seeded project_key/issuetype_id/summary/reporter_id") project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "JiraCreateIssueLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "JiraCreateIssueLiveEval" ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") debug_data, instance_id = bpmn_live.run_debug( imported_project, {}, LIVE_RUN_DIR / "debug.log" ) print(f"OK: debug completed (instance {instance_id})") - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") + variables_data, variables_text = fetch_variables(instance_id) + create_ids = tuple(node.attrib["id"] for node in create_nodes if node.attrib.get("id")) + cands = collect_candidate_keys(variables_data, create_ids, project) + _journal(cands) - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, + incidents, incidents_data = fetch_incidents(instance_id) + require_clean_run( + debug_data, DebugEvidence(variables_data, variables_text, incidents, incidents_data) ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) - cands = collect_candidate_keys(variables_data, variables.stdout or "", project) if not cands: - _fail(f"no issue key (e.g. {project}-123) in bpmn debug outputs") + _fail(f"no {project}- issue key in the Create-Issue node outputs {list(create_ids)}") print(f"OK: candidate keys from debug: {cands}") - # Journal every candidate BEFORE the tenant read: a real issue may exist - # even if its summary doesn't end up matching below, and this journal is - # the only sweep that survives coder_eval SIGKILLing this process on the - # criterion timeout (LIVE-ADDENDUM: journal side-effect ids the moment - # they are visible). - Path(".created_keys").write_text("\n".join(cands) + "\n") - jira_is = _import_jira_is() conn = jira_is.connection_id() for key in cands: fields = jira_is.get_issue(conn, key) if fields and fields.get("summary") == seed["summary"]: + JOURNAL.write_text(key + "\n") print(f"OK: Jira issue {key} exists with the seed summary") print("PASS: all JiraCreateIssue checks passed") return diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py index 54a6e4bd3e..84f9a30bba 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py @@ -31,11 +31,13 @@ (flow_check.run_debug raises on a non-Completed status internally; `bpmn debug` returns only an instance id, so the check is explicit here) → FinalStatus in COMPLETED_STATUSES and debug-instance incidents is empty + ADDED a Get-Issue node completed in debug ElementExecutions (Flow's payload only carries + outputs of nodes that ran) F check_jira_get_issue.py:55 assert_outputs_contain(payload, seed["summary"]) - → seeded summary found among the root scope's variable leaves AND every - element's Outputs in `debug-instance variables-all` (LIVE-ADDENDUM: a - root PUBLIC OUTPUT has read back null even when mapped correctly, so the - search is not scoped to a declared output variable) + → seeded summary found in the Get-Issue node's own Outputs or the root + Globals minus input echoes in `debug-instance variables-all` + (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read back null even when + mapped correctly, so the search is not scoped to a declared output) I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) → bpmn_check.find_bpmn_file()/resolve_project() I ephemeral solution init + `solution projects import` + sha256 pin of the imported @@ -71,12 +73,15 @@ from _shared.bpmn_live import ( # noqa: E402 CheckFailure, connector_context, + debug_evidence, + element_output_records, get_ci, - incident_records, - payload_data, + import_exact, + input_echo_ids, + normalized_identifier, + require_clean_run, root_scope, - run_cli, - sha256, + value_leaves, ) JIRA_KEY = "uipath-atlassian-jira" @@ -87,11 +92,6 @@ NAME_HINT = "JiraGetIssue" LIVE_RUN_DIR = Path("jira-get-issue-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} # Worst-case wall clock this checker can spend, priced the way # _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug @@ -111,17 +111,6 @@ def _fail(msg: str) -> None: sys.exit(f"FAIL: {msg}") -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - def is_get_issue_node(node_name: str, object_name: str, method: str) -> bool: """Curated (`curated_get_issue`) OR generic (`issue` + GETBYID/GET) form.""" if GET_OP_RE.search(object_name or "") or GET_OP_RE.search(node_name or ""): @@ -154,18 +143,31 @@ def find_get_issue_nodes(root: ET.Element) -> list[ET.Element]: return found -def collect_output_haystack(variables_data: object) -> str: - """Value leaves of the root scope's Globals AND every element's Outputs. +def completed_ids(debug_data: object, element_ids: tuple[str, ...]) -> tuple[str, ...]: + return tuple( + dict.fromkeys( + get_ci(item, "ElementId") + for item in get_ci(debug_data, "ElementExecutions", []) or [] + if isinstance(item, dict) + and get_ci(item, "ElementId") in element_ids + and str(get_ci(item, "Status") or "").casefold() == "completed" + ) + ) - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- it mirrors Flow's own - assert_outputs_contain(), which flattens the whole outputs payload. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) + +def collect_output_haystack( + variables_data: object, get_ids: tuple[str, ...], skip: set[str] +) -> str: + """The Get-Issue node's own Outputs plus root Globals that are not input + echoes, so a summary typed into an input cannot pass without the Get.""" + leaves = list(value_leaves(element_output_records(variables_data, get_ids))) + skipped = {normalized_identifier(name) for name in skip} + globals_ = get_ci(root_scope(variables_data), "Globals", {}) or {} + if isinstance(globals_, dict): + for name, value in globals_.items(): + if normalized_identifier(name) in skipped: + continue + leaves.extend(value_leaves(value)) return "\n".join(str(v) for v in leaves).lower() @@ -195,84 +197,32 @@ def main() -> None: print(f"OK: bpmn references a Get-Issue op and the seeded key {issue_key}") project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "JiraGetIssueLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "JiraGetIssueLiveEval" ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") debug_data, instance_id = bpmn_live.run_debug( imported_project, {}, LIVE_RUN_DIR / "debug.log" ) print(f"OK: debug completed (instance {instance_id})") - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) + evidence = debug_evidence(instance_id) + require_clean_run(debug_data, evidence) - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") + get_ids = tuple(node.attrib["id"] for node in get_issue_nodes if node.attrib.get("id")) + ran = completed_ids(debug_data, get_ids) + if not ran: + _fail( + f"no Get-Issue node among {list(get_ids)} completed in the debug trace; " + "the issue was not actually read" ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + print(f"OK: Get-Issue node(s) {list(ran)} completed") - haystack = collect_output_haystack(variables_data) + haystack = collect_output_haystack(evidence.variables, ran, input_echo_ids(root)) if seed["summary"].lower() not in haystack: _fail( - f"outputs do not contain the seeded issue summary {seed['summary']!r}\n" + f"Get-Issue outputs and non-input globals do not contain the seeded " + f"issue summary {seed['summary']!r}\n" f"outputs: {haystack[:1000]}" ) print("OK: bpmn outputs contain the seeded issue summary") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py index 9a104977dd..c2e161ea31 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_managed_http_fallback.py @@ -31,9 +31,9 @@ translation table's "Managed HTTP" row) I locate/parse .bpmn -> parse_bpmn(NAME_HINT) - T curated OR generic connectorKey match for the native -> is_native_connector_node(): matches on - branch (BATCH1-ADDENDUM: classify by connectorKey, not connectorKey alone, regardless of - by objectName) curated/generic objectName spelling + T curated OR generic connectorKey match for the native -> main(): connectorKey anywhere in the + branch (BATCH1-ADDENDUM: classify by connectorKey, not document, regardless of curated/generic + by objectName) objectName spelling T generic HTTP connector (uipath-uipath-http) accepted as -> is_managed_http_node() also accepts an additional managed-HTTP wrapper form: query_params.yaml Intsvc.ActivityExecution whose prompt explicitly asks for "a managed HTTP fallback through connectorKey == uipath-uipath-http, @@ -122,12 +122,6 @@ } -def is_native_connector_node(node: ET.Element, connector_key: str) -> bool: - if not has_type(node, ACTIVITY_TYPE): - return False - return context_value(node, "connectorKey").strip().lower() == connector_key.lower() - - def is_managed_http_node(node: ET.Element) -> bool: if any(has_type(node, token) for token in HTTP_TYPES): return True @@ -143,7 +137,7 @@ def node_blob(node: ET.Element) -> str: return ET.tostring(node, encoding="unicode").lower() -def require_all(haystack: str, needles: list[str], label: str) -> list[str]: +def require_all(haystack: str, needles: list[str]) -> list[str]: return [needle for needle in needles if needle.lower() not in haystack] @@ -181,7 +175,7 @@ def main() -> None: required = list(check["required"]) # type: ignore[arg-type] best_missing: list[str] | None = None for node in fallback_nodes: - missing = require_all(node_blob(node), required, label) + missing = require_all(node_blob(node), required) if not missing: print(f"OK: managed HTTP fallback has {check_name} evidence in {path}") return diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py index 3ffe0b5012..68b8488e42 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_path_param_value.py @@ -9,35 +9,24 @@ registry-workflow.md §3 "Connector enrichment"). Flow looked in exactly two places: a node's `pathParameters` dict values -(exact match) or its `url`/`endpoint` string (substring match). BPMN has no -fixed home for a path parameter -- the registry does not pin whether it lands -as a `target="path"` input, a `target="query"` input, inside the single -`target="body"` JSON payload, or embedded in a managed-HTTP node's own -url/path field -- so this checker widens the search to all of those homes -(BATCH1-ADDENDUM "Where connector node values live in BPMN" + -_porting/PORTING-BRIEF.md's `T` translation-tolerance for "entity anywhere in -inputs/objectName/path"). Widening only ever makes an assertion easier to -satisfy, never harder, so this stays within the Normalization pass's -"dropping/widening only" rule. +(exact match) or its `url`/`endpoint` string (substring match). The BPMN +homes are a `target="path"` input (exact) and a url/path/endpoint field +(substring). A query or body value is not a path parameter. Assertion map (Flow -> BPMN): - F check_path_param_value.py:87-91 pathParameters dict value == needle -> path_or_query_input_match(): any - (exact match) target="path"/"query" input whose - value/text contains needle + F check_path_param_value.py:87-91 pathParameters dict value == needle -> path_input_match(): a + (exact match) target="path" input whose value + equals needle F check_path_param_value.py:92-95 url/endpoint substring match -> context_field_match(): any "url"/ "path"/"endpoint" context field containing needle, checked on every connector/HTTP node I locate/parse .bpmn -> parse_bpmn(name_hint) - T needle searched across path/query/body, not only -> body_json_match(): the needle also - pathParameters (widened breadth, kept in scope by the matched inside the single target="body" - calling task's instructions) JSON payload, at any nesting depth T collect uipath:input elements at any depth under the node -> context_inputs() (bpmn_check) used by every match function above No Flow assertions dropped: both of Flow's two search locations (path-param -values, url/endpoint) have a widened BPMN counterpart above; nothing is -required that Flow did not also accept. +values, url/endpoint) have a BPMN counterpart above. Usage (from a task's run_command, cwd = sandbox root): python3 $REFERENCE_DIR/_shared/check_path_param_value.py @@ -45,7 +34,6 @@ from __future__ import annotations -import json import os import sys import xml.etree.ElementTree as ET @@ -72,32 +60,15 @@ ) -def _flatten_strings(value: object) -> list[str]: - if isinstance(value, str): - return [value] - if isinstance(value, dict): - out: list[str] = [] - for v in value.values(): - out.extend(_flatten_strings(v)) - return out - if isinstance(value, list): - out = [] - for v in value: - out.extend(_flatten_strings(v)) - return out - return [] - - -def path_or_query_input_match(node: ET.Element, needle: str) -> str | None: +def path_input_match(node: ET.Element, needle: str) -> str | None: node_id = node.attrib.get("id", "") for inp in context_inputs(node): - if inp.attrib.get("target") not in ("path", "query"): + if inp.attrib.get("target") != "path": continue - value = (inp.attrib.get("value") or inp.text or "") - if needle.lower() in value.lower(): - target = inp.attrib.get("target") + value = (inp.attrib.get("value") or inp.text or "").strip() + if value == needle: name = inp.attrib.get("name") or "?" - return f"{target} input {name!r} of node {node_id!r}" + return f"path input {name!r} of node {node_id!r}" return None @@ -113,28 +84,10 @@ def context_field_match(node: ET.Element, needle: str) -> str | None: return None -def body_json_match(node: ET.Element, needle: str) -> str | None: - node_id = node.attrib.get("id", "") - for inp in context_inputs(node): - if inp.attrib.get("target") != "body": - continue - raw = (inp.text or inp.attrib.get("value") or "").strip() - if not raw: - continue - try: - parsed = json.loads(raw) - except json.JSONDecodeError: - continue - for leaf in _flatten_strings(parsed): - if needle.lower() in leaf.lower(): - return f"body JSON payload of node {node_id!r}" - return None - - def find_needle(root: ET.Element, needle: str) -> str | None: candidates = [node for tag in CANDIDATE_TAGS for node in elements(root, tag)] for node in candidates: - for matcher in (path_or_query_input_match, context_field_match, body_json_match): + for matcher in (path_input_match, context_field_match): location = matcher(node, needle) if location is not None: return location @@ -151,8 +104,8 @@ def main() -> None: location = find_needle(root, needle) if location is None: fail( - f"{needle!r} not found in any node's path/query input, url/path/endpoint " - f"context field, or body JSON payload in {path}" + f"{needle!r} not found in any node's path input or url/path/endpoint " + f"context field in {path}" ) print(f"OK: {needle!r} found in {location}") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py index 148aecbe8b..5daca53ad5 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_http_fallback.py @@ -86,7 +86,6 @@ from __future__ import annotations -import json import os import re import sys @@ -106,13 +105,15 @@ ) from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 + INCIDENTS_TIMEOUT, CheckFailure, + DebugEvidence, connector_context, - get_ci, + import_exact, incident_records, payload_data, + require_clean_run, run_cli, - sha256, ) NAME_HINT = "SlackEmojiListTest" @@ -127,10 +128,6 @@ ACTIVITY_TYPES = ("Intsvc.ActivityExecution", "Intsvc.HttpExecution", "Intsvc.UnifiedHttpRequest") LIVE_RUN_DIR = Path("slack-emoji-list-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} # Worst-case wall clock check_debug can spend, priced the way # _shared/test_criterion_budgets.py prices a run_debug(...) call: the call @@ -227,76 +224,28 @@ def check_fallback() -> None: def check_debug() -> None: bpmn_path = find_bpmn_file(NAME_HINT) project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "SlackEmojiListTestLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT - ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "SlackEmojiListTestLiveEval" ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") debug_data, instance_id = bpmn_live.run_debug( imported_project, {}, LIVE_RUN_DIR / "debug.log" ) print(f"OK: debug completed (instance {instance_id})") - final_status = get_ci(debug_data, "FinalStatus") incidents = run_cli( ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], timeout=INCIDENTS_TIMEOUT, ) _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print( - "OK: uip maestro bpmn debug finished with FinalStatus=%s (no incidents)" - % final_status + evidence = DebugEvidence( + variables=None, + variables_text="", + incidents=incident_records(incidents_data), + incidents_raw=incidents_data, ) + require_clean_run(debug_data, evidence) DISPATCH = { diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py index 2d0b90bd17..006d0d5573 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_multiselect.py @@ -26,15 +26,14 @@ F check_multiselect_flow.py:51-57 is_users_key(): 'users' with an optional array-notation suffix -> is_users_key(): identical regex F check_multiselect_flow.py:60-76 find_users(): recursive dict/list search - -> find_users(): identical recursive search over the merged body object + -> find_users(): identical recursive search over the body object I locate/parse .bpmn, no name hint (this script is shared by both tasks; complex_array's own project-name hint is handled by its own check_complex_array.py, not here) -> parse_bpmn() I parse a connector node's request body (Flow read a native `inputs` dict; BPMN puts the whole request in `uipath:input` elements) -> body_object() - T the registry's two observed body forms both count: one whole-body - `target="body"` JSON blob (name="body"), or one typed `target="body"` - input per field (name=) -> bpmn_check.body_object() + T the body must be the one `target="body"` JSON object the runtime + consumes; several inputs fail (they don't merge at runtime) -> bpmn_check.body_object() DROPPED require_no_private_connector_values, require_sequence_integrity, require_di_for_visible_elements (not in Flow; `bpmn validate` criterion covers structure) @@ -56,6 +55,7 @@ from _shared.bpmn_check import ( # noqa: E402 NS, + BodyShapeError, body_object, context_value, elements, @@ -151,12 +151,16 @@ def main() -> None: reasons = [] for task in tasks: node_id = task.attrib.get("id", "") - users = find_users(body_object(task)) + try: + users = find_users(body_object(task)) + except BodyShapeError as exc: + reasons.append(f"node '{node_id}': {exc}") + continue if users is None: reasons.append(f"node '{node_id}': no 'users' multiselect field found") continue if expected_count == "populated": - if users: + if any(str(user).strip() for user in users): print(f"OK: {path} — node '{node_id}' users={users}") sys.exit(0) reasons.append(f"node '{node_id}': users field is empty") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py index d4eefe6b38..e0594bbf9b 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py @@ -1,8 +1,7 @@ #!/usr/bin/env python3 """SlackWeatherPipeline (BPMN): structural + live checks. -Ported from Flow `multi_node/slack_weather_pipeline/_shared/check_slack_weather_pipeline.py` -(via `tests/tasks/uipath-maestro-flow/_shared/check_slack_weather_pipeline.py`): +Ported from Flow `tests/tasks/uipath-maestro-flow/_shared/check_slack_weather_pipeline.py`: same scenario (a manual-start process reads the #office-bellevue Slack channel description, extracts the city, fetches weather for that city from open-meteo, and decides warm/cold), translated from a JSON node walk + inline @@ -51,19 +50,12 @@ incidents is empty F check_slack_weather_pipeline.py:31 assert_output_nonempty(payload, 'weatherVerdict') + lines 32-37 exactly one of ALLOWED_VERDICTS found - → collect_output_haystack(): value leaves of the root scope's - variables AND every element's Outputs in `debug-instance - variables-all` (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read - back null even when mapped correctly, so the search is not - scoped to one declared output variable -- LIVE-ADDENDUM's - translation rule for any Flow output-value assertion). - Widening from Flow's own name-scoped lookup carries no - practical false-positive risk here: the searched strings are - the full literal multi-word verdict phrases ('warm office - today' / 'cold office today'), not a loose substring, so - neither the Slack street-address text nor the raw Open-Meteo - JSON body (numeric fields only) can coincidentally match. The - exact-one-hit check is otherwise verbatim Flow logic. + → verdict_text(): the `weatherVerdict` root Global, resolved + by its declared output id then by name. Only when it reads + back null (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read back + null even when mapped correctly) does the search widen to + bpmn_live.output_haystack(). The exact-one-hit check is + verbatim Flow logic. I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) → bpmn_check.find_bpmn_file()/resolve_project() I ephemeral solution init + `solution projects import` + sha256 pin of the imported @@ -84,7 +76,6 @@ from __future__ import annotations -import json import os import sys import xml.etree.ElementTree as ET @@ -100,13 +91,15 @@ CheckFailure, UIPATH_NS, connector_context, + debug_evidence, get_ci, - incident_records, - payload_data, + import_exact, + normalized_identifier, + output_haystack, q, + require_clean_run, + resolve_runtime_key, root_scope, - run_cli, - sha256, ) SLACK_CONNECTOR_KEY = "uipath-salesforce-slack" @@ -119,13 +112,9 @@ NAME_HINT = "SlackWeatherPipeline" ALLOWED_VERDICTS = ("warm office today", "cold office today") +VERDICT_OUTPUT = "weatherVerdict" LIVE_RUN_DIR = Path("slack-weather-pipeline-live") -SOLUTION_INIT_TIMEOUT = 90 -SOLUTION_IMPORT_TIMEOUT = 180 -VARIABLES_ALL_TIMEOUT = 120 -INCIDENTS_TIMEOUT = 120 -COMPLETED_STATUSES = {"Completed", "Successful"} # Worst-case wall clock this checker can spend, priced the way # _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug @@ -149,17 +138,6 @@ def _fail(msg: str) -> None: sys.exit(f"FAIL: {msg}") -def _leaves(value): - if isinstance(value, dict): - for v in value.values(): - yield from _leaves(v) - elif isinstance(value, list): - for v in value: - yield from _leaves(v) - elif value is not None: - yield value - - def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: """Every element carrying an Intsvc.ActivityExecution targeting connector_key. @@ -214,21 +192,30 @@ def find_weather_node(root: ET.Element) -> list[ET.Element]: return found -def collect_output_haystack(variables_data: object) -> str: - """Value leaves of the root scope's Globals AND every element's Outputs. - - A root public output has been observed to read back null even when - correctly mapped (LIVE-ADDENDUM), so the search is not scoped to one - declared output variable -- it mirrors Flow's own assert_output_nonempty() - widened per LIVE-ADDENDUM's translation rule for output-value assertions. - Element Outputs include a connector's nested `response` object, which - _leaves() flattens along with everything else. - """ - leaves = list(_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - leaves.extend(_leaves(get_ci(element, "Outputs", {}))) - return "\n".join(str(v) for v in leaves).lower() +def verdict_identifiers(root: ET.Element) -> list[str]: + """Declared output id(s) named weatherVerdict, then the name itself.""" + wanted = normalized_identifier(VERDICT_OUTPUT) + ids = [ + node.attrib["id"] + for variables in root.iter(q(UIPATH_NS, "variables")) + for node in variables.findall(q(UIPATH_NS, "output")) + if node.attrib.get("id") and normalized_identifier(node.attrib.get("name", "")) == wanted + ] + return [*ids, VERDICT_OUTPUT] + + +def verdict_text(variables_data: object, identifiers: list[str]) -> str | None: + globals_ = get_ci(root_scope(variables_data), "Globals", {}) or {} + if not isinstance(globals_, dict): + return None + for identifier in identifiers: + wanted = normalized_identifier(identifier) + if not any(normalized_identifier(key) == wanted for key in globals_): + continue + value = resolve_runtime_key(globals_, identifier, VERDICT_OUTPUT) + if value is not None: + return str(value).lower() + return None def main() -> None: @@ -256,87 +243,30 @@ def main() -> None: print(f"OK: bpmn references an API node targeting open-meteo") project_dir = resolve_project(os.path.basename(bpmn_path)) - original_hash = sha256(Path(bpmn_path)) - LIVE_RUN_DIR.mkdir(parents=True, exist_ok=True) - solution_dir = LIVE_RUN_DIR / "SlackWeatherPipelineLiveEval" - initialized = run_cli( - ["uip", "solution", "init", str(solution_dir)], timeout=SOLUTION_INIT_TIMEOUT + imported_project = import_exact( + Path(bpmn_path), project_dir, LIVE_RUN_DIR / "SlackWeatherPipelineLiveEval" ) - payload_data(initialized, "initialize ephemeral solution") - solution_files = sorted(solution_dir.glob("*.uipx")) - if len(solution_files) != 1: - raise CheckFailure( - f"solution init produced {len(solution_files)} .uipx files in " - f"{solution_dir}, expected exactly one" - ) - solution_file = solution_files[0] - imported = run_cli( - [ - "uip", - "solution", - "projects", - "import", - str(project_dir.resolve()), - "--solutionFile", - str(solution_file), - ], - timeout=SOLUTION_IMPORT_TIMEOUT, - ) - payload_data(imported, "import exact BPMN project") - imported_project = solution_dir / project_dir.name - if sha256(imported_project / os.path.basename(bpmn_path)) != original_hash: - raise CheckFailure("solution import changed the submitted BPMN bytes") - print(f"OK: imported exact artifact (sha256={original_hash})") debug_data, instance_id = bpmn_live.run_debug( imported_project, {}, LIVE_RUN_DIR / "debug.log" ) print(f"OK: debug completed (instance {instance_id})") - final_status = get_ci(debug_data, "FinalStatus") - variables = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "variables-all", instance_id], - timeout=VARIABLES_ALL_TIMEOUT, - ) - _payload, variables_data = payload_data(variables, "variables-all") - - incidents = run_cli( - ["uip", "maestro", "bpmn", "debug-instance", "incidents", instance_id], - timeout=INCIDENTS_TIMEOUT, - ) - _payload, incidents_data = payload_data(incidents, "incidents") - incidents_list = incident_records(incidents_data) - - if final_status not in COMPLETED_STATUSES: - detail = [] - faulted = [ - f"{get_ci(item, 'ElementId')}={get_ci(item, 'Status')}" - for item in get_ci(debug_data, "ElementExecutions", []) or [] - if isinstance(item, dict) - and str(get_ci(item, "Status") or "").casefold() != "completed" - ] - if faulted: - detail.append(f"non-completed elements: {faulted}") - if incidents_list: - detail.append(f"incidents: {json.dumps(incidents_list)[:1500]}") - raise CheckFailure( - f"final status was {final_status!r}" - + ("; " + "; ".join(detail) if detail else "") - ) - if incidents_list is None: - raise CheckFailure(f"incidents response has an unknown shape: {incidents_data!r}") - if incidents_list: - raise CheckFailure(f"unexpected incidents: {incidents_list}") - print("OK: bpmn debug completed (FinalStatus=%s, no incidents)" % final_status) + evidence = debug_evidence(instance_id) + require_clean_run(debug_data, evidence) - haystack = collect_output_haystack(variables_data) + haystack = verdict_text(evidence.variables, verdict_identifiers(root)) + source = VERDICT_OUTPUT + if haystack is None: + haystack = output_haystack(evidence.variables) + source = f"outputs ({VERDICT_OUTPUT} read back null)" hits = [v for v in ALLOWED_VERDICTS if v in haystack] if len(hits) != 1: found = "both verdicts" if len(hits) > 1 else "neither verdict" _fail( - f"outputs must contain exactly one of {list(ALLOWED_VERDICTS)}; " - f"found {found}\noutputs: {haystack[:1000]}" + f"{source} must contain exactly one of {list(ALLOWED_VERDICTS)}; " + f"found {found}\n{source}: {haystack[:1000]}" ) print(f"OK: bpmn outputs carry {hits[0]!r}") print("PASS: all SlackWeatherPipeline checks passed") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py index 170bf61046..8d7ab77d53 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py @@ -65,7 +65,7 @@ has_typed_uipath_extension, parse_bpmn, ) -from _shared.graph import reachable # noqa: E402 +from _shared.graph import reachable, reaches # noqa: E402 BPMN_NS = NS["bpmn"] WAIT_TYPE = "Intsvc.WaitForEvent" @@ -201,6 +201,17 @@ def main() -> None: ) print("OK: manual GET HttpExecution sendTask to webhook URL, no headers/query") + wait_id, get_id = attr(event_nodes[0], "id"), attr(http_nodes[0], "id") + from_start = reachable(root, start_id) + if wait_id not in from_start or get_id not in from_start: + fail("the wait-for-event and HTTP-request nodes must both be reachable from the manual start") + if reaches(root, wait_id, get_id) or reaches(root, get_id, wait_id): + fail( + f"{wait_id!r} and {get_id!r} sit in series on one branch; the wait and the GET " + "must run on parallel branches or the wait never sees the request" + ) + print("OK: wait-for-event and HTTP-request run on parallel branches") + end_ids = {attr(e, "id") for e in elements(root, "endEvent")} if not end_ids: fail("no end event") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py index f38de59b5a..5a84a2edad 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py @@ -411,7 +411,7 @@ def test_find_bpmn_file_without_hint_skips_an_untyped_draft(tmp_path, monkeypatc """Two different .bpmn files, both beside a project.uiproj: the one with a registry-typed node is the deliverable; the other is an abandoned draft (CI run 35785806030). Two typed candidates stay ambiguous.""" - typed = '' + typed = f'' for d, body in (("Real/Proj", typed), ("ProjSolution/Proj", "")): (tmp_path / d).mkdir(parents=True) (tmp_path / d / "Proj.bpmn").write_text(body, encoding="utf-8") @@ -447,60 +447,38 @@ def test_body_object_reads_a_single_json_blob() -> None: } -def test_body_object_merges_several_json_blobs_later_wins() -> None: - task = _send_task( - '' - '' - '' - '' - ) - assert bpmn_check.body_object(task) == {"a": 1, "b": 2, "c": 3} +@pytest.mark.parametrize( + "payload", + [ + '' + '', + '' + '', + ], +) +def test_body_object_rejects_several_body_inputs(payload: str) -> None: + with pytest.raises(bpmn_check.BodyShapeError, match="does not merge"): + bpmn_check.body_object(_send_task(payload)) -def test_body_object_coerces_per_field_inputs_by_type() -> None: - task = _send_task( - '' - '' - '' - '' - '' - '' - '' - '' - '' - '' - ) - body = bpmn_check.body_object(task) - assert body == { - "score": 7.25, - "viewCount": 350, - "rank": 4, - "ratio": 9, - "active": True, - "archived": False, - "tags": ["x", "y"], - "title": "AllTypesTest-Smoke", - "code": "350", - "releaseDate": "2024-03-10", - } - assert type(body["score"]) is float - assert type(body["viewCount"]) is int and type(body["ratio"]) is int - assert type(body["active"]) is bool +def test_body_object_rejects_a_per_field_scalar() -> None: + task = _send_task('') + with pytest.raises(bpmn_check.BodyShapeError, match="must be an object"): + bpmn_check.body_object(task) -def test_body_object_mixes_a_blob_with_per_field_inputs() -> None: +def test_body_object_ignores_non_body_inputs() -> None: task = _send_task( '' - '' '' ) - assert bpmn_check.body_object(task) == {"title": "T", "score": 9} + assert bpmn_check.body_object(task) == {"title": "T"} def test_body_object_leaves_expressions_as_strings() -> None: task = _send_task( - '' - '' + '' ) assert bpmn_check.body_object(task) == { "score": "=vars.Score", @@ -508,6 +486,12 @@ def test_body_object_leaves_expressions_as_strings() -> None: } +def test_body_object_rejects_a_whole_body_expression() -> None: + task = _send_task('') + with pytest.raises(bpmn_check.BodyShapeError, match="expression"): + bpmn_check.body_object(task) + + def test_body_object_is_empty_without_body_inputs() -> None: task = _send_task('') assert bpmn_check.body_object(task) == {} @@ -518,11 +502,11 @@ def test_body_object_fails_a_malformed_body_blob() -> None: bad_json = _send_task( '' ) - with pytest.raises(SystemExit, match="not valid JSON"): + with pytest.raises(bpmn_check.BodyShapeError, match="not valid JSON"): bpmn_check.body_object(bad_json) not_object = _send_task( '' ) - with pytest.raises(SystemExit, match="must be an object"): + with pytest.raises(bpmn_check.BodyShapeError, match="must be an object"): bpmn_check.body_object(not_object) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py new file mode 100644 index 0000000000..679f6e69f6 --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py @@ -0,0 +1,89 @@ +"""Unit tests for the offline helpers in bpmn_live.""" + +from __future__ import annotations + +import os +import sys +import xml.etree.ElementTree as ET + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + +import bpmn_live # noqa: E402 +import pytest # noqa: E402 + +PROCESS = f""" + + + + + + + + + + + + + + + + + + + + + + + +""" + +VARIABLES = { + "Variables": [ + { + "ParentElementId": None, + "Globals": { + "input_Var_Invoice": "MCS-1", + "Var_Invoice": "MCS-1", + "output_Var_Total": 8, + }, + "Elements": [ + {"ElementId": "Start_1", "Outputs": {"invoiceNumber": "MCS-1"}}, + {"ElementId": "Task_1", "Outputs": {"response": {"rows": [{"id": "r1"}]}}}, + ], + } + ] +} + + +def test_input_echo_ids_covers_the_input_its_copy_and_the_start() -> None: + ids = bpmn_live.input_echo_ids(ET.fromstring(PROCESS)) + assert {"input_Var_Invoice", "invoiceNumber", "Var_Invoice", "Start_1"} <= ids + assert "output_Var_Total" not in ids + assert "Task_1" not in ids + + +def test_output_leaves_skips_input_echoes() -> None: + skip = bpmn_live.input_echo_ids(ET.fromstring(PROCESS)) + assert bpmn_live.output_leaves(VARIABLES, skip) == [8, "r1"] + assert "MCS-1" in bpmn_live.output_leaves(VARIABLES) + + +def _evidence(incidents) -> bpmn_live.DebugEvidence: + return bpmn_live.DebugEvidence(VARIABLES, "", incidents, incidents) + + +def test_require_clean_run_accepts_a_completed_run() -> None: + assert bpmn_live.require_clean_run({"FinalStatus": "Completed"}, _evidence([])) == "Completed" + + +@pytest.mark.parametrize( + "debug_data, incidents, message", + [ + ({"FinalStatus": "Faulted"}, [], "final status was 'Faulted'"), + ({"FinalStatus": "Completed"}, [{"Code": 102010}], "unexpected incidents"), + ({"FinalStatus": "Completed"}, None, "unknown shape"), + ], +) +def test_require_clean_run_rejects(debug_data, incidents, message) -> None: + with pytest.raises(bpmn_live.CheckFailure, match=message): + bpmn_live.require_clean_run(debug_data, _evidence(incidents)) diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml index 13f2dff3da..e9e3f3ef70 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml @@ -10,8 +10,8 @@ description: > Ported from Flow `connector_features/complex_array.yaml`; the connector activity is modeled as a bpmn:sendTask carrying the registry Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the - users array is read from its target="body" JSON (whole blob or one typed - input per field) instead of Flow's inputs.detail.bodyParameters. + users array is read from its one target="body" JSON object instead of + Flow's inputs.detail.bodyParameters. tags: - uipath-maestro-bpmn - integration diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml index c46c76cdaa..af7d704e76 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml @@ -6,10 +6,8 @@ description: > Ported from Flow `connector_features/enum.yaml`; the Gmail Send Mail connector node is modeled as a bpmn:sendTask carrying the registry Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the - request body is graded in either of two accepted shapes — one JSON - target="body" input, or one typed target="body" input per field — since - BPMN's canonical single-blob shape coexists with the CLI manifest's stale - separate-inputs guidance (registry-workflow.md §3 "Body shape"). + request body is graded as the one JSON target="body" input the runtime + consumes (registry-workflow.md §3 "Body shape"). Validate-only, matching Flow's own scope — the Flow source never called `flow debug` either, only `flow validate`. tags: diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml index 3f428821f4..438e6f0c4a 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml @@ -27,9 +27,8 @@ description: > acr_user`) instead of a Flow connector node with a node-type slug, `flow debug`'s single-call inline payload becomes an ephemeral-solution import + `bpmn debug` + `debug-instance variables-all`/`incidents` read, and - the array-output check searches the root scope's Globals plus every - element's Outputs instead of one inline debug payload's flow-output - globals. The live-check criterion timeout is raised from Flow's 720s to + the array-output check reads the declared process output's Global, falling + back to the list node's own Outputs only when every output reads back null. The live-check criterion timeout is raised from Flow's 720s to 1050s: the BPMN sequence adds ephemeral solution init/import and two separate `debug-instance` reads that flow's single inline `flow debug` call did not need (see check_generic_dynamic_node.py's budget comment) — the one @@ -37,11 +36,6 @@ description: > budget is a property of the CLI surface, not of what is graded. tags: [uipath-maestro-bpmn, e2e, "mode:operate", "lifecycle:generate", "shape:single-node", connector, uipath-servicenow-servicenow] -agent: - type: claude-code - permission_mode: acceptEdits - allowed_tools: ["Skill", "Bash", "Read", "Write", "Edit", "Glob", "Grep"] - sandbox: template_sources: - type: template_dir diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml index 2f605352ce..861c34feea 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml @@ -9,8 +9,8 @@ description: > Ported from Flow `connector_features/multiselect.yaml`; the connector activity is modeled as a bpmn:sendTask carrying the registry Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the - users multiselect is read from its target="body" JSON (whole blob or one - typed input per field) instead of Flow's inputs.detail.bodyParameters. + users multiselect is read from its one target="body" JSON object instead + of Flow's inputs.detail.bodyParameters. tags: - uipath-maestro-bpmn - integration diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml index 0dedb79542..3c51d958ab 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml @@ -12,8 +12,7 @@ description: > the HTTP-fallback node is modeled as a bpmn:sendTask carrying the registry Intsvc.ActivityExecution wrapper with connectorKey uipath-salesforce-slack targeting the emoji.list endpoint (registry-workflow.md §"Connectionless vs - connector HTTP" -- see the grader's GUESS note on the open question of - whether a connector-authenticated Intsvc.HttpExecution form also exists), + connector HTTP"), `flow debug`'s single-call inline payload becomes an ephemeral-solution import + `bpmn debug` + `debug-instance incidents` read, and the check_debug criterion's timeout is raised from Flow's 720s to 930s to cover the extra diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml index f55d8e4d1a..02203dcd94 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml @@ -1,8 +1,4 @@ -# Ported from Flow `connector_features/testmanager_crud_grounded.yaml`. -# Flow's task carries `skip: true` (pending agent-reliability confirmation). -# Per BATCH1-ADDENDUM.md: Test Manager `skip: true` is NOT carried over -- -# the codereval Test Manager connection exists, and a skipped port yields no -# signal -- this port runs so the BPMN suite measures it. +# Skipped: grades an agent-written result.json; unskip once it grades the live run. task_id: skill-bpmn-testmanager-crud-grounded description: > Data-grounded (Maestro BPMN): agent builds a process that creates a Test @@ -15,7 +11,7 @@ description: > connector node is modeled as a bpmn:sendTask carrying the registry Intsvc.ActivityExecution wrapper (create = TestCase/POST, get = TestCase/GETBYID|GET) instead of Flow's two per-operation connector node - types, and Flow's `skip: true` is not carried over (see header comment). + types. tags: - uipath-maestro-bpmn - e2e @@ -23,6 +19,7 @@ tags: - "lifecycle:generate" - "shape:multi-node" - connector +skip: true sandbox: template_sources: diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml index 259b46b71c..6a6ac10fd9 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml @@ -149,7 +149,7 @@ success_criteria: - type: run_command description: "Seeded path cases: each drives its branch and produces the expected outcome; the escalation case actually posts a Slack alert (non-empty message id)" command: "python3 $REFERENCE_DIR/_shared/check_escalation_orchestrator_paths.py" - # 7 x debug_budget(300, retries=1) = 2100 run_debug + 7x120s variables-all + # 7 x debug_budget(300) = 2100 run_debug + 7x120s variables-all # (840) + 90s solution init + 180s solution import + 60s margin = 3270; # rounded up to 3300. (Flow's own per-case run_debug math is identical -- # 2100 -- but bpmn debug needs the extra ephemeral solution diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py index 90dd603348..34345757be 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/jira_is.py @@ -11,6 +11,7 @@ from __future__ import annotations import json +import re import subprocess from typing import Any @@ -21,6 +22,22 @@ ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project +def _issue_not_found(env: dict) -> bool: + """True only for a structured HTTP 404 whose provider message says the issue + is absent, so a missing connection or activity never reads as deleted.""" + blob = json.dumps(env) + structured_404 = bool( + re.search(r"status code ['\"]?404\b", blob, re.I) + or re.search(r'"providerErrorCode"\s*:\s*404\b', blob) + or re.search(r'"statusCode"\s*:\s*"?404\b', blob) + ) + issue_specific = bool( + re.search(r"issue\s+(does\s+not\s+exist|not\s+found|no\s+longer\s+exists|is\s+not\s+found)", blob, re.I) + or re.search(r"(does\s+not\s+exist|not\s+found|no\s+longer\s+exists).{0,40}\bissue\b", blob, re.I) + ) + return structured_404 and issue_specific + + def _operation(args: tuple[str, ...]) -> str: return " ".join(args[:3]) @@ -102,8 +119,9 @@ def get_issue(conn_id: str, key: str) -> dict | None: def delete_issue(conn_id: str, key: str) -> None: - """Delete an issue by key. A 404 (already gone) is a no-op.""" - _run( + """Delete an issue by key. A 404 (already gone) is a no-op; any other + failure raises.""" + env = _run( "is", "resources", "run", "delete", CONNECTOR, "issue", "--connection-id", conn_id, "--query", f"issueId={key}", # The CLI never prompts and REFUSES an irreversible delete without this @@ -111,3 +129,6 @@ def delete_issue(conn_id: str, key: str) -> None: # teardown since 08-19 printed WARN and left its ticket in the CE project. "--yes", ) + if str(env.get("Result", "")).lower() != "failure" or _issue_not_found(env): + return + raise RuntimeError(f"delete {key} failed: {_failure_summary(env)}") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py index 91055ad017..b63751c009 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/_setup/teardown_jira.py @@ -13,7 +13,11 @@ if keys: conn = jira_is.connection_id() for key in keys: - jira_is.delete_issue(conn, key) + try: + jira_is.delete_issue(conn, key) + except Exception as e: # noqa: BLE001 — one failed key must not skip the rest + print(f"WARN: could not delete {key}: {e}") + continue print(f"OK: deleted {key}") else: print("OK: nothing to delete") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py index 9e6edb53c0..e3d218da04 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/jira_is.py @@ -11,6 +11,7 @@ from __future__ import annotations import json +import re import subprocess CONNECTOR = "uipath-atlassian-jira" @@ -20,6 +21,22 @@ ISSUETYPE_ID = "11457" # "Task" issue type, scoped to the CE project +def _issue_not_found(env: dict) -> bool: + """True only for a structured HTTP 404 whose provider message says the issue + is absent, so a missing connection or activity never reads as deleted.""" + blob = json.dumps(env) + structured_404 = bool( + re.search(r"status code ['\"]?404\b", blob, re.I) + or re.search(r'"providerErrorCode"\s*:\s*404\b', blob) + or re.search(r'"statusCode"\s*:\s*"?404\b', blob) + ) + issue_specific = bool( + re.search(r"issue\s+(does\s+not\s+exist|not\s+found|no\s+longer\s+exists|is\s+not\s+found)", blob, re.I) + or re.search(r"(does\s+not\s+exist|not\s+found|no\s+longer\s+exists).{0,40}\bissue\b", blob, re.I) + ) + return structured_404 and issue_specific + + def _run(*args: str) -> dict: out = subprocess.run( ["uip", *args, "--output", "json"], @@ -55,8 +72,9 @@ def get_issue(conn_id: str, key: str) -> dict | None: def delete_issue(conn_id: str, key: str) -> None: - """Delete an issue by key. A 404 (already gone) is a no-op.""" - _run( + """Delete an issue by key. A 404 (already gone) is a no-op; any other + failure raises.""" + env = _run( "is", "resources", "run", "delete", CONNECTOR, "issue", "--connection-id", conn_id, "--query", f"issueId={key}", # The CLI never prompts and REFUSES an irreversible delete without this @@ -64,3 +82,6 @@ def delete_issue(conn_id: str, key: str) -> None: # teardown since 08-19 printed WARN and left its ticket in the CE project. "--yes", ) + if str(env.get("Result", "")).lower() != "failure" or _issue_not_found(env): + return + raise RuntimeError(f"delete {key} failed: {json.dumps(env)[:500]}") diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py index 91055ad017..b63751c009 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/_setup/teardown_jira.py @@ -13,7 +13,11 @@ if keys: conn = jira_is.connection_id() for key in keys: - jira_is.delete_issue(conn, key) + try: + jira_is.delete_issue(conn, key) + except Exception as e: # noqa: BLE001 — one failed key must not skip the rest + print(f"WARN: could not delete {key}: {e}") + continue print(f"OK: deleted {key}") else: print("OK: nothing to delete") diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml index 195ee3f123..4a8f96e6c5 100644 --- a/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml +++ b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml @@ -13,14 +13,13 @@ description: > ("UiPath Flow" -> "UiPath Maestro BPMN process", "flow" -> "process", "a `.flow` file" -> "a `.bpmn` file"), the Script node becomes a `bpmn:scriptTask` (the `BPMN.ScriptTask` registry construct), and the - live-check criterion follows the same translation as the committed - live-simulated ports (`_shared/check_weather_bpmn_simulated.py`, - `_shared/check_jira_get_issue.py`): a JSON node-type scan + inline `flow + live-check criterion follows the same translation as + `_shared/check_jira_get_issue.py`: a JSON node-type scan + inline `flow debug` payload becomes an XML scan for a `bpmn:scriptTask` element plus the BPMN live-debug surface (ephemeral solution import, `bpmn debug`, `debug-instance variables-all`/`incidents`, per LIVE-ADDENDUM's canonical pattern). That surface's worst-case budget (90 solution init + 180 solution - import + 480 debug + 120 variables-all + 120 incidents + 60s margin = 1050s) + import + 600 debug + 120 variables-all + 120 incidents + 60s margin = 1170s) fits under Flow's own criterion timeout (1320s), so the criterion timeout is kept verbatim; `run_limits.task_timeout` is raised from Flow's 2400s to 2800s (turn_timeout 1200 + worst-case grading 1515 [180 validate + 1320 @@ -128,9 +127,9 @@ success_criteria: description: "BPMN debug runs; scriptTask executed and output is an integer in [1,6]" command: "python3 $REFERENCE_DIR/_shared/check_dice_runs_simulated.py" # 1320 funds the ephemeral-solution live-debug surface (solution init + - # import + bpmn debug + variables-all + incidents = 990s) plus margin — + # import + bpmn debug + variables-all + incidents = 1110s) plus margin — # see check_dice_runs_simulated.py's own budget comment. Kept verbatim - # from Flow, which already exceeds the 1050s this surface needs. + # from Flow, which already exceeds the 1170s this surface needs. timeout: 1320 expected_exit_code: 0 weight: 2.0 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml index dd5cc7c916..bb20fdf9d4 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml @@ -25,7 +25,7 @@ tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", conne run_limits: expected_turns: 45 max_turns: 120 - task_timeout: 1800 + task_timeout: 2490 turn_timeout: 1200 sandbox: From 6b6ce7b5a64271df29fde7429c3e57573b01e769 Mon Sep 17 00:00:00 2001 From: rockymadden Date: Thu, 24 Sep 2026 14:13:08 -0600 Subject: [PATCH 24/35] test(bpmn): treat an empty body input as an empty body Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py | 4 +++- tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py | 5 +++++ 2 files changed, 8 insertions(+), 1 deletion(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py index e2de7f8258..4c8bcd1a05 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py @@ -220,7 +220,7 @@ def body_object(element: ET.Element) -> dict: - Returns ``{}`` when there is no ``target="body"`` input. Raises + Returns ``{}`` when there is no ``target="body"`` input or it is empty. Raises :class:`BodyShapeError` for several inputs, or one whose payload is an expression or anything but a JSON object. A ``=vars.X`` value inside the object stays the string it is. @@ -236,6 +236,8 @@ def body_object(element: ET.Element) -> dict: ) raw = _input_payload(fields[0]) + if not raw: + return {} if raw.startswith("="): raise BodyShapeError(f'target="body" input is an expression, not a literal object: {raw!r}') try: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py index 5a84a2edad..982226cc7d 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py @@ -498,6 +498,11 @@ def test_body_object_is_empty_without_body_inputs() -> None: assert bpmn_check.body_fields(task) == [] +def test_body_object_is_empty_for_an_empty_body_input() -> None: + task = _send_task('') + assert bpmn_check.body_object(task) == {} + + def test_body_object_fails_a_malformed_body_blob() -> None: bad_json = _send_task( '' From dc53183f08536eee765c6aa2418dc063c77a5e72 Mon Sep 17 00:00:00 2001 From: rockymadden Date: Thu, 24 Sep 2026 14:15:11 -0600 Subject: [PATCH 25/35] docs(bpmn): drop the stale task counts from the parity ledger Co-Authored-By: Claude Opus 5.5 (1M context) --- tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md | 2 -- 1 file changed, 2 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index 5e201dfb00..5787858580 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -1,7 +1,5 @@ # Flow → BPMN eval parity map -Flow tasks: 131 · BPMN tasks: 82 - ## Porting ledger One row per task, final state. Iterations are summarised in the notes; runs are GitHub Actions `run-coder-eval.yml` ids on the alpha tenant, codex driver. Structural rows landed in PR #3426; live and field-shape rows are PR #3502; parked rows live on branch `test/bpmn-port-parked` (stacked on #3502), each with `skip: true` and its evidence in the YAML. From 851ac44b73e7ca63b64f6d3dcfd2a0c706ff2590 Mon Sep 17 00:00:00 2001 From: rockymadden Date: Thu, 24 Sep 2026 14:32:29 -0600 Subject: [PATCH 26/35] test(bpmn): close the remaining review threads on the live-tier port - Tags follow the tests/README.md vocabulary: one tier, mode and lifecycle per task, the flat connector marker, outcome-graded on live graders. - find_bpmn_file fails on ambiguity again; only the parked task needed the typed-draft preference. - Output checks skip input echoes (orchestrator, channel description); dice reads whole integers only; df smoke_error accepts the context path and plain curated names; paginated and webhook check every candidate. - Jira graders keep every key the Create node reported for teardown, and the escalation grader matches the Create op on objectName too. - Graders use bpmn_check.fail and output_leaves(elements=) instead of private copies; LIVE_OVERHEAD_SECONDS is gone. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../_porting/LIVE-ADDENDUM.md | 2 +- .../_porting/LIVE-HANDOFF.md | 6 +-- .../uipath-maestro-bpmn/_shared/bpmn_check.py | 16 ------ .../uipath-maestro-bpmn/_shared/bpmn_live.py | 22 ++++---- .../_shared/check_channel_description.py | 21 ++++---- .../_shared/check_df_smoke_error.py | 32 ++++++------ .../_shared/check_dice_runs_simulated.py | 50 +++++++++---------- .../_shared/check_escalation_jira_ticket.py | 8 +-- .../check_escalation_orchestrator_paths.py | 23 ++++----- .../_shared/check_generic_dynamic_node.py | 20 +++----- .../_shared/check_jira_create_issue.py | 30 +++++------ .../_shared/check_jira_get_issue.py | 34 ++++--------- .../check_paginated_reference_lookup.py | 37 ++++++++------ .../_shared/check_slack_weather_pipeline.py | 14 ++---- .../_shared/check_webhook_waitfor_parallel.py | 30 +++++++---- .../_shared/test_bpmn_check.py | 18 ------- .../complex_array/complex_array.yaml | 1 - .../smoke_error/smoke_error.yaml | 2 +- .../connector_features/enum/enum.yaml | 1 - .../generic_dynamic_node.yaml | 2 +- .../jdbc_databricks_query.yaml | 2 +- .../multiselect/multiselect.yaml | 1 - .../paginated_reference_lookup.yaml | 2 - .../path_params/path_params.yaml | 6 +-- .../query_params/query_params.yaml | 1 - .../searchable_joins/searchable_joins.yaml | 2 +- .../slack_http_fallback.yaml | 2 +- .../webhook_waitfor_parallel.yaml | 2 - .../jira_create_issue/jira_create_issue.yaml | 2 +- .../e2e/jira_get_issue/jira_get_issue.yaml | 2 +- .../cli_dice_roller_simulated.yaml | 2 +- .../billing_invoice_lookup.yaml | 3 +- .../slack_channel_description.yaml | 2 +- .../slack_weather_pipeline.yaml | 2 +- 34 files changed, 172 insertions(+), 228 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md index 5ceddae683..534f4324cc 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-ADDENDUM.md @@ -13,7 +13,7 @@ Read `PORTING-BRIEF.md` (Grading contract is mandatory) and `BATCH1-ADDENDUM.md` 6. `fetch_incidents(id)` then `require_clean_run(debug_data, evidence)`; a completed run with incidents is a failure. `debug_evidence(id)` does 4 and 6 together for a grader with no side effects. Known runtime facts (grade around them, do not fight them): -- Element-level `Outputs` (a script task's mapped output, a connector's `response`) are reliably readable in `variables-all`. Root **public output values** have been read back as `null` even when correctly mapped (see `debug/live_debug_e2e/check_live_debug.py` docstring). So when Flow asserted "some output equals X" (`assert_output_value` / `assert_outputs_contain` over `variables.globals` + element outputs), translate to: search the value leaves of the root scope's variables AND every element's `Outputs` in `variables-all`; do not require the value on a root public output specifically. +- Element-level `Outputs` (a script task's mapped output, a connector's `response`) are reliably readable in `variables-all`. Root **public output values** have been read back as `null` even when correctly mapped (see `debug/live_debug_e2e/check_live_debug.py` docstring). So when Flow asserted "some output equals X" (`assert_output_value` / `assert_outputs_contain` over `variables.globals` + element outputs), translate to: read the declared output first, and only when it reads back null search `output_leaves(variables, skip=input_echo_ids(process))`, narrowed with `elements=` to the nodes that produce the value when the task names them. - Variables are addressed by **id**, and the runtime may re-case ids — use `resolve_runtime_key`. - Debug instances are ephemeral; read variables-all immediately after the run. diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md index 16f6e86953..5ecbf6a7ef 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/LIVE-HANDOFF.md @@ -11,11 +11,11 @@ Methodology is in `PORTING-BRIEF.md`, `BATCH1-ADDENDUM.md`, `NORMALIZATION.md` a ## Live-grader recipe -Canonical: `_shared/check_jira_get_issue.py`. From `_shared/bpmn_live.py`: `import_exact` (ephemeral `uip solution init` + `projects import`, sha256-pinned) → `run_debug(project, inputs, log, timeout)` in the grader itself, so `test_criterion_budgets.py` can price it → `debug_evidence` (`variables-all`, `incidents`) → `require_clean_run`. Grade with `output_leaves(variables, skip=input_echo_ids(process))`: root public outputs have read back null, and an unskipped input reads as an output. A grader with side effects journals their ids before `require_clean_run`, so a faulted run still cleans up. +Canonical: `_shared/check_jira_get_issue.py`. From `_shared/bpmn_live.py`: `import_exact` (ephemeral `uip solution init` + `projects import`, sha256-pinned) → `run_debug(project, inputs, log, timeout)` in the grader itself, so `test_criterion_budgets.py` can price it → `debug_evidence` (`variables-all`, `incidents`) → `require_clean_run`. Read a declared output first; when it reads back null (root public outputs have), fall back to `output_leaves(variables, skip=input_echo_ids(process), elements=…)` over the nodes that produce it. An unskipped input reads as an output. A grader with side effects journals their ids before `require_clean_run`, so a faulted run still cleans up. post_run `_setup/cleanup_solutions.py` sweeps the ephemeral solution; a later criterion in the same task passes `resolve_project(exclude_under=[LIVE_RUN_DIR])` so that import is not read as a second project. -Budget: the guard enforces criterion `timeout` ≥ `debug_budget(...)` + 60. Size it to `debug_budget(...)` + `LIVE_OVERHEAD_SECONDS` (510) + 60, e.g. 480 + 510 + 60 = 1050 for one default debug. +Budget: the guard enforces criterion `timeout` ≥ `debug_budget(...)` + 60. Size it to cover the whole sequence: 90 (init) + 180 (import) + `debug_budget(...)` + 120 (variables-all) + 120 (incidents) + 60, e.g. 1050 for one default debug. ## Runtime and grader facts @@ -23,7 +23,7 @@ Budget: the guard enforces criterion `timeout` ≥ `debug_budget(...)` + 60. Siz - Incident 102010 "Parameter 'Folder' null" on a Slack node = missing `folderKey` binding; 102009 = missing activity parameter; 400008 on a multi-instance marker = input collection over a connector response. - Connector nodes come in two forms: curated `objectName` with path/query inputs, or generic entity-CRUD with the verb in `operation`/`method`. A request body is one JSON `target="body"` input; several do not merge at runtime, and `bpmn_check.body_object` raises `BodyShapeError` on them. - Managed HTTP is `Intsvc.HttpExecution` or `Intsvc.UnifiedHttpRequest`; connector-mode HTTP is `Intsvc.ActivityExecution`. Wait-for-event may be a `receiveTask` or an `intermediateCatchEvent`; classify by the `uipath:type` wrapper, never the BPMN tag. -- `find_bpmn_file` treats byte-identical copies as one artifact and drops drafts with no registry-typed node; `validate_bpmn.py` validates every `.bpmn` in the sandbox. +- `find_bpmn_file` treats byte-identical copies as one artifact and fails on any other ambiguity; `validate_bpmn.py` validates every `.bpmn` in the sandbox. - Actions.HITL needs a deployed Action App to pass `bpmn validate`; the curated Data Fabric query template has no sort-field parameter. ## Skill findings to report upstream diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py index 4c8bcd1a05..cd271646f8 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py @@ -59,25 +59,9 @@ def find_bpmn_file(name_hint: str | None = None) -> str: # wrapper; CI run 35538279757, testmanager_crud_grounded) are one artifact. if len({_sha256(p) for p in paths}) == 1: return paths[0] - # An abandoned draft beside the real process: `bpmn init` leaves a - # solution wrapper whose process carries no registry-typed node, while the - # deliverable does (CI run 35785806030, slack_channel_description_simulated). - # Only a candidate with at least one `` is a process - # the task could be graded on. - typed = [p for p in projects if _has_typed_node(p)] - if len(typed) == 1: - return typed[0] fail(f"multiple BPMN files found; expected one or hint match: {paths}") -def _has_typed_node(path: str | Path) -> bool: - try: - root = ET.parse(path).getroot() - except (OSError, ET.ParseError): - return False - return any(node.attrib.get("value") for node in root.iter(f"{{{NS['uipath']}}}type")) - - def _sha256(path: str | Path) -> str: import hashlib diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py index ffa320d589..26ef63f9f4 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py @@ -501,12 +501,6 @@ def run_debug( VARIABLES_ALL_TIMEOUT = 120 INCIDENTS_TIMEOUT = 120 -# Sum of the unpriced CLI steps around one run_debug: init + import + -# variables-all + incidents. A criterion timeout adds this to debug_budget(). -LIVE_OVERHEAD_SECONDS = ( - SOLUTION_INIT_TIMEOUT + SOLUTION_IMPORT_TIMEOUT + VARIABLES_ALL_TIMEOUT + INCIDENTS_TIMEOUT -) - COMPLETED_STATUSES = frozenset({"Completed", "Successful"}) @@ -631,11 +625,18 @@ def value_leaves(value: Any) -> Iterator[Any]: yield value -def output_leaves(variables_data: Any, skip: Collection[str] = ()) -> list[Any]: - """Leaves of the root Globals and every element's Outputs, minus the - globals and elements named in `skip` (see :func:`input_echo_ids`).""" +def output_leaves( + variables_data: Any, + skip: Collection[str] = (), + *, + elements: Collection[str] | None = None, +) -> list[Any]: + """Leaves of the root Globals and the elements' Outputs, minus the globals + and elements named in `skip` (see :func:`input_echo_ids`). `elements` + limits the Outputs to those element ids; None reads every element.""" skipped = {normalized_identifier(name) for name in skip} + wanted = None if elements is None else {normalized_identifier(e) for e in elements} globals_ = get_ci(root_scope(variables_data), "Globals", {}) or {} leaves: list[Any] = [] if isinstance(globals_, dict): @@ -646,7 +647,8 @@ def output_leaves(variables_data: Any, skip: Collection[str] = ()) -> list[Any]: for scope in get_ci(variables_data, "Variables", []) or []: for element in get_ci(scope, "Elements", []) or []: - if normalized_identifier(get_ci(element, "ElementId")) in skipped: + element_id = normalized_identifier(get_ci(element, "ElementId")) + if element_id in skipped or (wanted is not None and element_id not in wanted): continue leaves.extend(value_leaves(get_ci(element, "Outputs", {}))) return leaves diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py index 9b9d6478e0..335122494d 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_channel_description.py @@ -32,7 +32,9 @@ F check_channel_description.py:29 assert_outputs_contain(payload, ADDRESS_FRAGMENTS, require_all=True) → every fragment found among the root scope's variable leaves AND every element's Outputs (incl. nested connector `response`) - in `debug-instance variables-all` (LIVE-ADDENDUM: a root PUBLIC + in `debug-instance variables-all`, minus input echoes + (input_echo_ids; Flow subtracts declared inputs the same way) + (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read back null even when mapped correctly, so the search is not scoped to one declared output variable) I locate/parse .bpmn (file exists, well-formed XML, project directory resolved) @@ -63,13 +65,14 @@ # …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_jira_get_issue.py sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 -from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared.bpmn_check import fail, find_bpmn_file, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, connector_context, debug_evidence, import_exact, + input_echo_ids, output_haystack, require_clean_run, ) @@ -105,10 +108,6 @@ # what is graded). -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: """Every element carrying an Intsvc.ActivityExecution targeting connector_key. @@ -133,17 +132,17 @@ def main() -> None: bpmn_path = find_bpmn_file(NAME_HINT) raw = Path(bpmn_path).read_text(encoding="utf-8") if CONNECTOR_KEY not in raw: - _fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") + fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") print(f"OK: bpmn references {CONNECTOR_KEY}") try: root = ET.parse(bpmn_path).getroot() except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") + fail(f"{bpmn_path} is not well-formed XML: {exc}") connector_nodes = find_connector_nodes(root, CONNECTOR_KEY) if not connector_nodes: - _fail( + fail( f"bpmn does not reference a {CONNECTOR_KEY} connector node " f"({ACTIVITY_TYPE})" ) @@ -163,10 +162,10 @@ def main() -> None: evidence = debug_evidence(instance_id) require_clean_run(debug_data, evidence) - haystack = output_haystack(evidence.variables) + haystack = output_haystack(evidence.variables, input_echo_ids(root)) missing = [f for f in ADDRESS_FRAGMENTS if f.lower() not in haystack] if missing: - _fail( + fail( f"outputs missing address fragments {missing}; " f"expected all of {ADDRESS_FRAGMENTS}\noutputs: {haystack[:1000]}" ) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index c8c5a8bb12..80bcd0982b 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -11,14 +11,11 @@ ``Intsvc.ActivityExecution`` connector shell (see skills/uipath-maestro-bpmn/references/registry-workflow.md §3-4). -BPMN has no fixed home for `entityName`: the registry does not pin it to a -path param, query param, or JSON body key, and a node may also carry it only -as the generic-form `objectName` when the skill emits the generic -entity-CRUD shape instead of the curated per-operation one -- see -BATCH1-ADDENDUM.md "Where connector node values live in BPMN" and its "CI -run 35488848026" section on the two valid activity shapes. Both forms are -accepted here, exactly as `check_df_integration_create_get.py` and -`check_df_smoke_query_filter.py` already do for this connector. +BPMN has no fixed home for `entityName` (BATCH1-ADDENDUM.md "Where connector +node values live in BPMN" and its "CI run 35488848026" section on the two +valid activity shapes). A node names its entity when any one of these +equals it: the generic-form `objectName`, a `target="path"` input value, or a +`/`-separated segment of the context `path` field. Assertion map (Flow -> BPMN): F check_smoke_error.py:29-31 entity_of(node) == NonExistentEntity on a @@ -29,8 +26,9 @@ F check_smoke_error.py:40-42 `len(good_queries) < 2` -> fail -> `good_queries < 2` check I locate/parse .bpmn -> parse_bpmn() T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() - T entity as the generic objectName, or an -> mentions_entity() - exact target="path" input value + T entity as the generic objectName, an exact -> mentions_entity() + target="path" input value, or an exact + context `path` segment DROPPED topology/parallel-branch parsing (Flow's own grader does not parse it either -- see its docstring) DROPPED require_no_private_connector_values (not in Flow) DROPPED require_sequence_integrity (not in Flow; `bpmn validate` criterion covers structure) @@ -40,11 +38,11 @@ 1. BPMN file exists and is well-formed XML. 2. >=1 bpmn:sendTask carries Intsvc.ActivityExecution with connectorKey uipath-uipath-dataservice, classified as a Create (curated objectName, - or generic objectName + Create/POST verb), and mentions - NonExistentEntity somewhere in its inputs. + or generic objectName + Create/POST verb), and names + NonExistentEntity (see mentions_entity above). 3. >=2 such nodes classified as a Query (curated Query Entity Records - objectName, or generic objectName + List/GET verb), and mention - FlowCodeEvalEntity somewhere in their inputs. + objectName, or generic objectName + List/GET verb), and name + FlowCodeEvalEntity. """ from __future__ import annotations @@ -70,8 +68,8 @@ ACTIVITY_TYPE = "Intsvc.ActivityExecution" CREATE_ENTITY = "NonExistentEntity" QUERY_ENTITY = "FlowCodeEvalEntity" -CREATE_CURATED_NAMES = {"createentityrecordcurated", "createentityrecord_v3"} -QUERY_CURATED_NAMES = {"queryentityrecordscurated", "queryentityrecords_v3"} +CREATE_CURATED_NAMES = {"createentityrecord", "createentityrecordcurated", "createentityrecord_v3"} +QUERY_CURATED_NAMES = {"queryentityrecords", "queryentityrecordscurated", "queryentityrecords_v3"} _CREATE_OP_RE = re.compile(r"^create$", re.IGNORECASE) _LIST_OP_RE = re.compile(r"^list$", re.IGNORECASE) @@ -80,6 +78,8 @@ def mentions_entity(task: ET.Element, entity: str) -> bool: if is_generic_entity_object(context_value(task, "objectName"), entity): return True + if entity in context_value(task, "path").strip().split("/"): + return True return any( inp.attrib.get("target") == "path" and (inp.attrib.get("value") or inp.text or "").strip() == entity diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py index 120577bee1..9eb7a59f4f 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py @@ -41,13 +41,12 @@ -> FinalStatus in COMPLETED_STATUSES and debug-instance incidents is empty F check_dice_runs_simulated.py:24 assert_output_int_in_range(payload, 1, 6) - -> value-leaf search (never the whole JSON - payload -- see the docstring on - assert_output_int_in_range: "Extracts - integers from output values only, not from - the full debug payload") over the root - scope's Globals AND every element's Outputs - in `debug-instance variables-all` + -> first whole-integer leaf in [1, 6] (an int, + or a string that is only an integer; Flow's + digit-run regex would read `7.5` as 5) over + the root scope's Globals AND every + element's Outputs in + `debug-instance variables-all` (LIVE-ADDENDUM: a root PUBLIC OUTPUT has read back null even when mapped correctly, so the search is not scoped to one declared output @@ -82,7 +81,7 @@ # .../uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 -from _shared.bpmn_check import elements, find_bpmn_file, resolve_project # noqa: E402 +from _shared.bpmn_check import elements, fail, find_bpmn_file, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, @@ -100,8 +99,9 @@ # _shared/test_criterion_budgets.py prices a run_debug(...) call: the debug # call below passes timeout=600 (Flow's own run_debug timeout), so it # prices at bpmn_live.debug_budget(600) == 600. The surrounding CLI -# steps are not priced by that guard, so bpmn_live.LIVE_OVERHEAD_SECONDS is -# added by hand here and the criterion `timeout:` in +# steps are not priced by that guard, so their 510 s (90 init + 180 import +# + 120 variables-all + 120 incidents) is added by hand here and the +# criterion `timeout:` in # cli_dice_roller_simulated.yaml documents the arithmetic: # 90 (solution init) + 180 (solution import) + 600 (debug) # + 120 (variables-all) + 120 (incidents) = 1110 @@ -110,21 +110,19 @@ # verbatim rather than raised. -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - def find_int_in_range(leaves: list, lo: int, hi: int) -> int | None: - """First integer in [lo, hi] found in the stringified leaf values. - - Mirrors flow_check.assert_output_int_in_range's exact rule: regex over - the individual OUTPUT VALUE leaves only (never the raw variables-all JSON - blob, whose element ids, timestamps and status strings would spuriously - match a small target range like [1, 6]). + """First whole-integer leaf in [lo, hi]: an int, or a string that is only + an integer. `roll: 7.5` or a timestamp never yields a digit run. """ - haystack = "\n".join(str(v) for v in leaves) - for match in INT_RE.findall(haystack): - value = int(match) + for leaf in leaves: + if isinstance(leaf, bool): + continue + if isinstance(leaf, int): + value = leaf + elif isinstance(leaf, str) and INT_RE.fullmatch(leaf.strip()): + value = int(leaf.strip()) + else: + continue if lo <= value <= hi: return value return None @@ -140,11 +138,11 @@ def main() -> None: try: root = ET.parse(bpmn_path).getroot() except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") + fail(f"{bpmn_path} is not well-formed XML: {exc}") script_tasks = elements(root, "scriptTask") if not script_tasks: - _fail(f"{bpmn_path} has no bpmn:scriptTask element (no Script node found)") + fail(f"{bpmn_path} has no bpmn:scriptTask element (no Script node found)") print(f"OK: bpmn has a scriptTask ({len(script_tasks)} found)") project_dir = resolve_project(os.path.basename(bpmn_path)) @@ -165,7 +163,7 @@ def main() -> None: roll = find_int_in_range(leaves, ROLL_LO, ROLL_HI) if roll is None: haystack = "\n".join(str(v) for v in leaves) - _fail( + fail( f"No integer in [{ROLL_LO}, {ROLL_HI}] found in outputs\n" f"Outputs: {haystack[:1000]}" ) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py index ad19ec3df7..4186be373b 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_jira_ticket.py @@ -108,8 +108,9 @@ plus bpmn_live.CRITERION_MARGIN_SECONDS (60) = 1410 -- under Flow's original 1600s criterion timeout, so it is kept verbatim (see escalation_jira_ticket.yaml). -The confirmed key is written to `.created_keys` so post_run's `teardown_jira.py` -(copied verbatim from Flow) deletes it even if a later assertion fails. +Every key the Create-Issue node reported is written to `.created_keys` so +post_run's `teardown_jira.py` (copied verbatim from Flow) deletes it even if a +later assertion fails. """ from __future__ import annotations @@ -232,7 +233,7 @@ def resolve_contract(root: ET.Element) -> Contract: jira_create_ids = tuple( element_id for (key, path, _object_name), element_ids in connectors.items() - if key == JIRA_CONNECTOR and JIRA_CREATE_OP in path + if key == JIRA_CONNECTOR and (JIRA_CREATE_OP in path or JIRA_CREATE_OP in _object_name) for element_id in element_ids ) if not jira_create_ids: @@ -370,7 +371,6 @@ def main() -> None: for fields in [jira_is.get_issue(conn, k)] if fields is not None and correlation in str(fields.get("summary", "")) ] - _journal(owned) if not owned: _fail( f"none of {jira_keys} is a Jira issue whose summary contains " diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py index b297d2f3c5..80d191d520 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py @@ -24,7 +24,7 @@ -> assert_send_identity(): every Slack sendTask carries target=query name=send_as value=user F check_escalation_orchestrator_paths.py:55 assert_named_equals(payload, name, expected, ...) -> public_value_present(): normalized value present among root Globals leaves + non-classifier element Outputs - leaves (see NOTE below) + leaves, input echoes (input_echo_ids) excluded from both (see NOTE below) F check_escalation_orchestrator_paths.py:65 assert_slack_message_posted(payload, "slackMessageId", ...) -> assert_slack_posted(): fired Slack sendTask's own Outputs.response carries a ts-shaped id, the seeded channel, and correlationId + escalationPath in its text @@ -96,12 +96,12 @@ get_ci, import_exact, index_runtime_connectors, + input_echo_ids, + output_leaves, payload_data, q, - root_scope, run_cli, run_debug, - value_leaves, ) SLACK_KEY = "uipath-salesforce-slack" @@ -130,6 +130,7 @@ class Contract: gateway_ids: tuple[str, ...] slack_ids: tuple[str, ...] end_ids: tuple[str, ...] + echo_ids: tuple[str, ...] def resolve_contract(path: Path) -> Contract: @@ -177,6 +178,7 @@ def resolve_contract(path: Path) -> Contract: gateway_ids=gateway_ids, slack_ids=slack_ids, end_ids=end_ids, + echo_ids=tuple(sorted(input_echo_ids(process))), ) @@ -333,17 +335,12 @@ def _loose_contains(haystack: str, needle: str) -> bool: def public_value_present( variables_data, expected, *, exclude_ids: tuple[str, ...], case_sensitive: bool ) -> bool: - """Whether `expected` shows up among root Globals or a non-excluded - element's Outputs -- see the module docstring NOTE on why this is a broad - leaf search rather than one pinned root-output id.""" + """Whether `expected` shows up among root Globals or element Outputs, minus + the Globals and elements named in `exclude_ids` -- see the module docstring + NOTE on why this is a broad leaf search rather than one pinned root-output id.""" target = normalized(expected, case_fold=not case_sensitive) - candidates = list(value_leaves(get_ci(root_scope(variables_data), "Globals", {}))) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: - if get_ci(element, "ElementId") in exclude_ids: - continue - candidates.extend(value_leaves(get_ci(element, "Outputs", {}))) + candidates = output_leaves(variables_data, exclude_ids) return any(normalized(v, case_fold=not case_sensitive) == target for v in candidates) @@ -454,7 +451,7 @@ def verify_case(contract: Contract, imported_project: Path, case: dict) -> dict: if not public_value_present( variables_data, expected_value, - exclude_ids=contract.classifier_ids, + exclude_ids=contract.classifier_ids + contract.echo_ids, case_sensitive=field in CASE_SENSITIVE, ): raise CheckFailure(f"{case['name']}: no public output carries {field}={expected_value!r}") diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py index 9739ecd074..3f31972bbc 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_generic_dynamic_node.py @@ -87,7 +87,7 @@ # …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_jira_get_issue.py sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 -from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared.bpmn_check import fail, find_bpmn_file, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 BPMN_NS, @@ -136,10 +136,6 @@ # budget is a property of the CLI surface, not of what is graded). -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - def is_generic_list_node(context: dict) -> bool: """Generic (non-curated) list activity: objectName == acr_user, list op. @@ -246,15 +242,15 @@ def main() -> None: bpmn_path = find_bpmn_file(NAME_HINT) raw = Path(bpmn_path).read_text(encoding="utf-8") if CONNECTOR_KEY not in raw: - _fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") + fail(f"{bpmn_path} does not reference the {CONNECTOR_KEY} connector") if OBJECT_NAME not in raw: - _fail(f"{bpmn_path} does not reference the object {OBJECT_NAME!r}") + fail(f"{bpmn_path} does not reference the object {OBJECT_NAME!r}") print(f"OK: bpmn references {CONNECTOR_KEY} and object {OBJECT_NAME!r}") try: root = ET.parse(bpmn_path).getroot() except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") + fail(f"{bpmn_path} is not well-formed XML: {exc}") list_nodes = find_generic_list_nodes(root) if not list_nodes: @@ -265,7 +261,7 @@ def main() -> None: if connector_context(node).get("connectorKey") == CONNECTOR_KEY } ) - _fail( + fail( f"No generic ServiceNow list activity found on the {CONNECTOR_KEY} " f"connector with objectName={OBJECT_NAME!r} (expected operation " f"'list' or method 'GET'). Connector node contexts seen: {seen}" @@ -274,10 +270,10 @@ def main() -> None: process = root.find(q(BPMN_NS, "process")) if process is None: - _fail(f"{bpmn_path} has no bpmn:process") + fail(f"{bpmn_path} has no bpmn:process") outputs = declared_outputs(process) if not outputs: - _fail("process declares no uipath:output variable to surface the records") + fail("process declares no uipath:output variable to surface the records") list_node_ids = tuple(node.attrib["id"] for node in list_nodes if node.attrib.get("id")) project_dir = resolve_project(os.path.basename(bpmn_path)) @@ -295,7 +291,7 @@ def main() -> None: candidates = collect_array_candidates(evidence.variables, outputs, list_node_ids) if not candidates: - _fail( + fail( "No output variable holds an array — the connector result was not " "surfaced as a process output. Checked declared output globals " f"{outputs} and, on null readback, the list node Outputs " diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py index 096be6c503..c408d97fec 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py @@ -13,8 +13,9 @@ variables-all`/`incidents`). Exits non-zero on the first failure (``FAIL: ...``); prints ``OK: ...`` per -check. The confirmed key(s) are journaled to ``.created_keys`` so post_run -teardown (`teardown_jira.py`) deletes them even if a later step fails. +check. Every key the Create-Issue node reported is journaled to +``.created_keys`` so post_run teardown (`teardown_jira.py`) deletes them even +if a later step fails. Assertion map (Flow -> BPMN): F check_jira_create_issue.py:49 JIRA_KEY not in raw or '"nodes"' not in raw @@ -60,8 +61,8 @@ F check_jira_create_issue.py:66-77 tenant re-read via jira_is.get_issue(conn, key); first candidate whose `summary` equals the seed summary wins; confirmed key journaled for teardown - -> same logic; `.created_keys` is rewritten to the - confirmed key only; jira_is.py (task's own + -> same logic; `.created_keys` keeps every key the + Create node reported; jira_is.py (task's own `_setup/jira_is.py` copy, imported via the sandbox-mounted path since this checker lives in `_shared/`, not the task dir) @@ -98,7 +99,7 @@ # …/uipath-maestro-bpmn (for _shared), same convention as check_jira_get_issue.py sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 -from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared.bpmn_check import fail, find_bpmn_file, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, @@ -136,10 +137,6 @@ # verbatim rather than raised. -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - def _import_jira_is(): """Lazy import so an empty/pre-authoring sandbox fails cleanly on the seed.json/bpmn checks above, rather than an ImportError traceback here. @@ -203,29 +200,29 @@ def _journal(keys: list[str]) -> None: def main() -> None: seed_path = Path("seed.json") if not seed_path.is_file(): - _fail("seed.json is missing from the sandbox (pre_run seed did not run)") + fail("seed.json is missing from the sandbox (pre_run seed did not run)") seed = json.loads(seed_path.read_text(encoding="utf-8")) project = seed["project_key"] bpmn_path = find_bpmn_file(NAME_HINT) raw = Path(bpmn_path).read_text(encoding="utf-8") if JIRA_KEY not in raw: - _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") + fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") print(f"OK: bpmn references {JIRA_KEY}") try: root = ET.parse(bpmn_path).getroot() except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") + fail(f"{bpmn_path} is not well-formed XML: {exc}") create_nodes = find_create_issue_nodes(root) if not create_nodes: - _fail("bpmn does not reference a Jira Create-Issue connector node (Intsvc.ActivityExecution)") + fail("bpmn does not reference a Jira Create-Issue connector node (Intsvc.ActivityExecution)") print("OK: bpmn references a Create-Issue op") missing = [field for field in SEED_LITERAL_FIELDS if str(seed[field]) not in raw] if missing: - _fail( + fail( "bpmn does not reference the seeded " f"{', '.join(f'{field}={seed[field]!r}' for field in missing)} " "(agent must use seed.json values verbatim, not invented ones)" @@ -253,7 +250,7 @@ def main() -> None: ) if not cands: - _fail(f"no {project}- issue key in the Create-Issue node outputs {list(create_ids)}") + fail(f"no {project}- issue key in the Create-Issue node outputs {list(create_ids)}") print(f"OK: candidate keys from debug: {cands}") jira_is = _import_jira_is() @@ -261,11 +258,10 @@ def main() -> None: for key in cands: fields = jira_is.get_issue(conn, key) if fields and fields.get("summary") == seed["summary"]: - JOURNAL.write_text(key + "\n") print(f"OK: Jira issue {key} exists with the seed summary") print("PASS: all JiraCreateIssue checks passed") return - _fail( + fail( f"none of {cands} carries the seed summary {seed['summary']!r} — the " "bpmn process did not create the expected issue in Jira" ) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py index 84f9a30bba..9882ab88e3 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_get_issue.py @@ -68,20 +68,17 @@ # …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_drive_to_slack.py sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 -from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared.bpmn_check import fail, find_bpmn_file, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 CheckFailure, connector_context, debug_evidence, - element_output_records, get_ci, import_exact, input_echo_ids, - normalized_identifier, + output_leaves, require_clean_run, - root_scope, - value_leaves, ) JIRA_KEY = "uipath-atlassian-jira" @@ -107,10 +104,6 @@ # verbatim rather than raised. -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - def is_get_issue_node(node_name: str, object_name: str, method: str) -> bool: """Curated (`curated_get_issue`) OR generic (`issue` + GETBYID/GET) form.""" if GET_OP_RE.search(object_name or "") or GET_OP_RE.search(node_name or ""): @@ -160,40 +153,33 @@ def collect_output_haystack( ) -> str: """The Get-Issue node's own Outputs plus root Globals that are not input echoes, so a summary typed into an input cannot pass without the Get.""" - leaves = list(value_leaves(element_output_records(variables_data, get_ids))) - skipped = {normalized_identifier(name) for name in skip} - globals_ = get_ci(root_scope(variables_data), "Globals", {}) or {} - if isinstance(globals_, dict): - for name, value in globals_.items(): - if normalized_identifier(name) in skipped: - continue - leaves.extend(value_leaves(value)) + leaves = output_leaves(variables_data, skip, elements=get_ids) return "\n".join(str(v) for v in leaves).lower() def main() -> None: seed_path = Path("seed.json") if not seed_path.is_file(): - _fail("seed.json is missing from the sandbox (pre_run seed did not run)") + fail("seed.json is missing from the sandbox (pre_run seed did not run)") seed = json.loads(seed_path.read_text(encoding="utf-8")) issue_key = seed["issue_key"] bpmn_path = find_bpmn_file(NAME_HINT) raw = Path(bpmn_path).read_text(encoding="utf-8") if JIRA_KEY not in raw: - _fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") + fail(f"{bpmn_path} does not reference the {JIRA_KEY} connector") print(f"OK: bpmn references {JIRA_KEY}") try: root = ET.parse(bpmn_path).getroot() except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") + fail(f"{bpmn_path} is not well-formed XML: {exc}") get_issue_nodes = find_get_issue_nodes(root) if not get_issue_nodes: - _fail("bpmn does not reference a Jira Get-Issue connector node (Intsvc.ActivityExecution)") + fail("bpmn does not reference a Jira Get-Issue connector node (Intsvc.ActivityExecution)") if issue_key not in raw: - _fail(f"bpmn does not reference the seeded key {issue_key!r} (agent must read it from seed.json)") + fail(f"bpmn does not reference the seeded key {issue_key!r} (agent must read it from seed.json)") print(f"OK: bpmn references a Get-Issue op and the seeded key {issue_key}") project_dir = resolve_project(os.path.basename(bpmn_path)) @@ -212,7 +198,7 @@ def main() -> None: get_ids = tuple(node.attrib["id"] for node in get_issue_nodes if node.attrib.get("id")) ran = completed_ids(debug_data, get_ids) if not ran: - _fail( + fail( f"no Get-Issue node among {list(get_ids)} completed in the debug trace; " "the issue was not actually read" ) @@ -220,7 +206,7 @@ def main() -> None: haystack = collect_output_haystack(evidence.variables, ran, input_echo_ids(root)) if seed["summary"].lower() not in haystack: - _fail( + fail( f"Get-Issue outputs and non-input globals do not contain the seeded " f"issue summary {seed['summary']!r}\n" f"outputs: {haystack[:1000]}" diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py index a67ee38be7..034224a644 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_paginated_reference_lookup.py @@ -28,7 +28,7 @@ F criterion 5 flow_contains --flow-name SlackPaginationTest 'uipath.connector.uipath-salesforce-slack.send-message-to-channel' '"C083AN4E61E"' - → check_wired(): an Intsvc.ActivityExecution bpmn:sendTask + → check_wired(): any Intsvc.ActivityExecution bpmn:sendTask with connectorKey uipath-salesforce-slack whose objectName names the Send Message to Channel operation, and whose serialised XML contains the channel id C083AN4E61E @@ -79,13 +79,14 @@ SEND_MESSAGE_RE = re.compile(r"send[\s_-]*messages?[\s_-]*to[\s_-]*channel", re.IGNORECASE) -def find_slack_node(root: ET.Element) -> ET.Element | None: - for task in elements(root, "sendTask"): - if has_type(task, ACTIVITY_TYPE) and context_value(task, "connectorKey") == CONNECTOR_KEY: - object_name = context_value(task, "objectName") - if SEND_MESSAGE_RE.search(object_name): - return task - return None +def find_slack_nodes(root: ET.Element) -> list[ET.Element]: + return [ + task + for task in elements(root, "sendTask") + if has_type(task, ACTIVITY_TYPE) + and context_value(task, "connectorKey") == CONNECTOR_KEY + and SEND_MESSAGE_RE.search(context_value(task, "objectName")) + ] def check_exists() -> None: @@ -96,18 +97,24 @@ def check_exists() -> None: def check_wired() -> None: _path, root = parse_bpmn(NAME_HINT) - node = find_slack_node(root) - if node is None: + nodes = find_slack_nodes(root) + if not nodes: fail( f"no bpmn:sendTask carries {ACTIVITY_TYPE} with connectorKey " f"{CONNECTOR_KEY!r} and an objectName naming Send Message to Channel" ) - print(f"OK: {CONNECTOR_KEY} Send Message to Channel node present " - f"(objectName={context_value(node, 'objectName')!r})") + print(f"OK: {len(nodes)} {CONNECTOR_KEY} Send Message to Channel node(s) present") - if not has_type(node, CHANNEL_ID): - fail(f"Slack send-message node does not carry the resolved channel id {CHANNEL_ID!r}") - print(f"OK: Slack send-message node carries channel id {CHANNEL_ID!r}") + wired = [node for node in nodes if has_type(node, CHANNEL_ID)] + if not wired: + ids = [node.attrib.get("id", "?") for node in nodes] + fail( + f"none of the Slack send-message nodes {ids} carries the resolved " + f"channel id {CHANNEL_ID!r}" + ) + print(f"OK: Slack send-message node {wired[0].attrib.get('id', '?')!r} " + f"(objectName={context_value(wired[0], 'objectName')!r}) carries " + f"channel id {CHANNEL_ID!r}") DISPATCH = { diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py index e0594bbf9b..f7111bcec0 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_slack_weather_pipeline.py @@ -84,7 +84,7 @@ # …/uipath-maestro-bpmn (for _shared), same convention as _shared/check_channel_description.py sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) # noqa: E402 -from _shared.bpmn_check import find_bpmn_file, resolve_project # noqa: E402 +from _shared.bpmn_check import fail, find_bpmn_file, resolve_project # noqa: E402 from _shared import bpmn_live # noqa: E402 from _shared.bpmn_live import ( # noqa: E402 BPMN_NS, @@ -134,10 +134,6 @@ # what is graded). -def _fail(msg: str) -> None: - sys.exit(f"FAIL: {msg}") - - def find_connector_nodes(root: ET.Element, connector_key: str) -> list[ET.Element]: """Every element carrying an Intsvc.ActivityExecution targeting connector_key. @@ -224,11 +220,11 @@ def main() -> None: try: root = ET.parse(bpmn_path).getroot() except ET.ParseError as exc: - _fail(f"{bpmn_path} is not well-formed XML: {exc}") + fail(f"{bpmn_path} is not well-formed XML: {exc}") connector_nodes = find_connector_nodes(root, SLACK_CONNECTOR_KEY) if not connector_nodes: - _fail( + fail( f"bpmn does not reference a {SLACK_CONNECTOR_KEY} connector node " f"({ACTIVITY_TYPE})" ) @@ -236,7 +232,7 @@ def main() -> None: weather_nodes = find_weather_node(root) if not weather_nodes: - _fail( + fail( f"bpmn does not reference an API-capable node ({' / '.join(HTTP_TYPES)} or " f"{ACTIVITY_TYPE}) targeting one of {WEATHER_HINTS}" ) @@ -264,7 +260,7 @@ def main() -> None: hits = [v for v in ALLOWED_VERDICTS if v in haystack] if len(hits) != 1: found = "both verdicts" if len(hits) > 1 else "neither verdict" - _fail( + fail( f"{source} must contain exactly one of {list(ALLOWED_VERDICTS)}; " f"found {found}\n{source}: {haystack[:1000]}" ) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py index 8d7ab77d53..3245b4aea6 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_webhook_waitfor_parallel.py @@ -201,26 +201,36 @@ def main() -> None: ) print("OK: manual GET HttpExecution sendTask to webhook URL, no headers/query") - wait_id, get_id = attr(event_nodes[0], "id"), attr(http_nodes[0], "id") from_start = reachable(root, start_id) - if wait_id not in from_start or get_id not in from_start: + pairs = [ + (attr(wait, "id"), attr(get, "id")) + for wait in event_nodes + for get in http_nodes + if attr(wait, "id") in from_start and attr(get, "id") in from_start + ] + if not pairs: fail("the wait-for-event and HTTP-request nodes must both be reachable from the manual start") - if reaches(root, wait_id, get_id) or reaches(root, get_id, wait_id): + + parallel = [ + (wait_id, get_id) + for wait_id, get_id in pairs + if not reaches(root, wait_id, get_id) and not reaches(root, get_id, wait_id) + ] + if not parallel: fail( - f"{wait_id!r} and {get_id!r} sit in series on one branch; the wait and the GET " + f"every wait/GET pair {pairs} sits in series on one branch; the wait and the GET " "must run on parallel branches or the wait never sees the request" ) - print("OK: wait-for-event and HTTP-request run on parallel branches") + wait_id, get_id = parallel[0] + print(f"OK: wait-for-event {wait_id!r} and HTTP-request {get_id!r} run on parallel branches") end_ids = {attr(e, "id") for e in elements(root, "endEvent")} if not end_ids: fail("no end event") - for label, node in (("wait-for-event", event_nodes[0]), ("http-request", http_nodes[0])): - node_id = attr(node, "id") - reach = reachable(root, node_id) - if not (reach & end_ids): - fail(f"{label} branch does not reach an end event") + for label, node_id in (("wait-for-event", wait_id), ("http-request", get_id)): + if not (reachable(root, node_id) & end_ids): + fail(f"{label} branch {node_id!r} does not reach an end event") print("OK: both branches terminate at an end event") print( diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py index 982226cc7d..e2ed9bc14f 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_check.py @@ -407,24 +407,6 @@ def test_resolve_project_excludes_the_live_run_copy(tmp_path, monkeypatch) -> No assert resolved == tmp_path / "ProjSolution" / "Proj" -def test_find_bpmn_file_without_hint_skips_an_untyped_draft(tmp_path, monkeypatch) -> None: - """Two different .bpmn files, both beside a project.uiproj: the one with a - registry-typed node is the deliverable; the other is an abandoned draft - (CI run 35785806030). Two typed candidates stay ambiguous.""" - typed = f'' - for d, body in (("Real/Proj", typed), ("ProjSolution/Proj", "")): - (tmp_path / d).mkdir(parents=True) - (tmp_path / d / "Proj.bpmn").write_text(body, encoding="utf-8") - (tmp_path / d / "project.uiproj").write_text("{}", encoding="utf-8") - monkeypatch.chdir(tmp_path) - - assert bpmn_check.find_bpmn_file().endswith("Real/Proj/Proj.bpmn") - - (tmp_path / "ProjSolution" / "Proj" / "Proj.bpmn").write_text(typed.replace("Intsvc", "BPMN"), encoding="utf-8") - with pytest.raises(SystemExit): - bpmn_check.find_bpmn_file() - - def _send_task(payload: str) -> ET.Element: return ET.fromstring( f' Intsvc.ActivityExecution wrapper instead of Flow connector nodes, and the grader accepts either the curated per-operation objectName or the generic entity-CRUD form the connector skill may emit for the same operation. -tags: [uipath-maestro-bpmn, smoke, "mode:build", "lifecycle:generate", connector, negative, uipath-uipath-dataservice] +tags: [uipath-maestro-bpmn, smoke, "mode:build", "lifecycle:generate", connector] sandbox: template_sources: diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml index af7d704e76..8dff7b6930 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml @@ -16,7 +16,6 @@ tags: - "lifecycle:generate" - "shape:single-node" - connector - - uipath-google-gmail - "mode:build" run_limits: diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml index 438e6f0c4a..2af3d48b50 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml @@ -34,7 +34,7 @@ description: > did not need (see check_generic_dynamic_node.py's budget comment) — the one sanctioned deviation from "criteria identical" per LIVE-ADDENDUM, since the budget is a property of the CLI surface, not of what is graded. -tags: [uipath-maestro-bpmn, e2e, "mode:operate", "lifecycle:generate", "shape:single-node", connector, uipath-servicenow-servicenow] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, outcome-graded] sandbox: template_sources: diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml index 18954a7fbc..e3a44ea379 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml @@ -21,7 +21,7 @@ description: > JSON node/edge walk becomes an XML walk over the registry-driven Intsvc.ActivityExecution wrapper, and Flow's node-type-suffix disambiguation guard becomes a connectorKey + objectName/method context-field match. -tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, uipath-uipath-jdbc] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector] sandbox: template_sources: - type: template_dir diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml index 861c34feea..9ecbcc3c61 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml @@ -18,7 +18,6 @@ tags: - "shape:single-node" - connector - "mode:build" - - uipath-salesforce-slack run_limits: expected_turns: 37 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml index 68f01444dd..e80f4513dd 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml @@ -21,12 +21,10 @@ description: > tags: - uipath-maestro-bpmn - integration - - e2e - "mode:build" - "lifecycle:generate" - "shape:multi-node" - connector - - uipath-salesforce-slack - "feature:records" run_limits: diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml index f96e2783ff..42410410ff 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml @@ -7,9 +7,8 @@ description: > Ported from Flow `connector_features/path_params.yaml`; the Jira Get Issue connector node is modeled as a bpmn:sendTask carrying the registry Intsvc.ActivityExecution wrapper instead of a Flow connector node, and the - path-parameter value is graded across path/query/body inputs since BPMN has - no fixed home for it (registry-workflow.md does not pin where a path - parameter lands). Validate-only, matching Flow's own scope — the Flow + path-parameter value must appear as an exact `target="path"` input value or + inside a url/path/endpoint context field. Validate-only, matching Flow's own scope — the Flow source never called `flow debug` either, only `flow validate`. tags: - uipath-maestro-bpmn @@ -17,7 +16,6 @@ tags: - "lifecycle:generate" - "shape:multi-node" - connector - - uipath-atlassian-jira - "mode:build" run_limits: diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml index add29abcc0..2a00469271 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml @@ -17,7 +17,6 @@ tags: - "lifecycle:generate" - "shape:multi-node" - connector - - uipath-google-tasks - "mode:build" run_limits: diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml index a128640e4b..2a88f576e3 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml @@ -10,7 +10,7 @@ description: > query node is modeled as a bpmn:sendTask carrying the registry Intsvc.ActivityExecution wrapper instead of a Flow connector node, and validation runs via `bpmn validate` instead of `flow validate`. -tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", connector, uipath-salesforce, "mode:build"] +tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:multi-node", connector, "mode:build"] run_limits: expected_turns: 56 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml index 3c51d958ab..0795b96b78 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml @@ -19,7 +19,7 @@ description: > CLI steps (`solution init`/`solution projects import`, a separate `incidents` read) that BPMN's live surface needs beyond Flow's single `flow debug` call (arithmetic in the grader's docstring). -tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, uipath-salesforce-slack, "feature:http"] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, "feature:http", outcome-graded] run_limits: expected_turns: 32 diff --git a/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml b/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml index a6c4d931f9..8df95d115a 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml @@ -27,8 +27,6 @@ tags: connector, "feature:trigger", "feature:http", - wait-for-event, - uipath-http-webhook, ] sandbox: diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml index 6cd4cc983c..8a3cca1659 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml @@ -23,7 +23,7 @@ description: > variables-all`/`incidents` read, and the created-key search scans the root scope's variables plus every element's Outputs (and the raw variables-all response text) instead of one inline debug payload. -tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:single-node", connector, e2e, uipath-atlassian-jira, "mode:build"] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, outcome-graded] run_limits: expected_turns: 40 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml index 190283e44a..9501d14ba8 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml @@ -22,7 +22,7 @@ description: > `debug-instance variables-all`/`incidents` read, and grading of the fetched summary searches the root scope's variables plus every element's Outputs instead of one inline debug payload. -tags: [uipath-maestro-bpmn, integration, "lifecycle:generate", "shape:single-node", connector, e2e, uipath-atlassian-jira, "mode:build"] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector, outcome-graded] run_limits: expected_turns: 40 diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml index 4a8f96e6c5..d57b9707f2 100644 --- a/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml +++ b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml @@ -29,7 +29,7 @@ description: > assertions themselves (a scriptTask node exists; the run's outputs contain an integer in [1, 6]; the project was named as elicited) are unchanged from Flow's own grader. -tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", simulation] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", outcome-graded] sandbox: template_sources: diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml index 55f39c79db..cd67c4b98f 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml @@ -26,10 +26,11 @@ tags: - uipath-maestro-bpmn - e2e - "mode:build" - - "lifecycle:execute" + - "lifecycle:generate" - "shape:multi-node" - connector - path-to-ga + - outcome-graded sandbox: template_sources: - type: template_dir diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml index bb20fdf9d4..fb97a85649 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml @@ -20,7 +20,7 @@ description: > Flow's single `flow debug` call; `run_limits` gains `task_timeout` and `max_turns` (absent from the Flow task) so that grading window fits inside the task watchdog. -tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", connector] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", connector, outcome-graded] run_limits: expected_turns: 45 diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml index 7774572b4f..96ef3cca68 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml @@ -19,7 +19,7 @@ description: > init`/`solution projects import`, separate `variables-all`/`incidents` reads) that BPMN's live surface needs beyond Flow's single `flow debug` call. -tags: [uipath-maestro-bpmn, e2e, "lifecycle:generate", "shape:multi-node", "node:decision", connector, "feature:http"] +tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:multi-node", "node:decision", connector, "feature:http", outcome-graded] run_limits: expected_turns: 51 From 67c4ed615272cf516cb7c6c5d11eb55da458a586 Mon Sep 17 00:00:00 2001 From: rockymadden Date: Thu, 24 Sep 2026 14:35:08 -0600 Subject: [PATCH 27/35] test(bpmn): answer the second round of review threads - Jira create journals only the Create node's top-level response.key, so a nested foreign key is never deleted. - input_echo_ids follows verbatim copies through any element, and output_leaves drops the copied output names; both tolerate a malformed Variables/Elements shape. - df smoke_error accepts the entity as an exact value on any input. - Regression tests pin index_runtime_connectors' key shape and fail any grader that unpacks it short. Co-Authored-By: Claude Opus 5.5 (1M context) --- .../uipath-maestro-bpmn/_shared/bpmn_live.py | 56 +++++++++---- .../_shared/check_df_smoke_error.py | 9 +- .../_shared/check_jira_create_issue.py | 8 +- .../_shared/test_bpmn_live.py | 82 +++++++++++++++++++ 4 files changed, 135 insertions(+), 20 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py index 26ef63f9f4..7f3b3e946e 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_live.py @@ -631,9 +631,10 @@ def output_leaves( *, elements: Collection[str] | None = None, ) -> list[Any]: - """Leaves of the root Globals and the elements' Outputs, minus the globals - and elements named in `skip` (see :func:`input_echo_ids`). `elements` - limits the Outputs to those element ids; None reads every element.""" + """Leaves of the root Globals and the elements' Outputs, minus the + globals, elements and top-level output names in `skip` (see + :func:`input_echo_ids`). `elements` limits the Outputs to those element + ids; None reads every element.""" skipped = {normalized_identifier(name) for name in skip} wanted = None if elements is None else {normalized_identifier(e) for e in elements} @@ -645,12 +646,21 @@ def output_leaves( continue leaves.extend(value_leaves(value)) - for scope in get_ci(variables_data, "Variables", []) or []: - for element in get_ci(scope, "Elements", []) or []: + scopes = get_ci(variables_data, "Variables", []) + for scope in scopes if isinstance(scopes, list) else []: + elements_ = get_ci(scope, "Elements", []) + for element in elements_ if isinstance(elements_, list) else []: element_id = normalized_identifier(get_ci(element, "ElementId")) if element_id in skipped or (wanted is not None and element_id not in wanted): continue - leaves.extend(value_leaves(get_ci(element, "Outputs", {}))) + outputs = get_ci(element, "Outputs", {}) + if isinstance(outputs, dict): + outputs = { + key: value + for key, value in outputs.items() + if normalized_identifier(key) not in skipped + } + leaves.extend(value_leaves(outputs)) return leaves @@ -660,12 +670,12 @@ def output_haystack(variables_data: Any, skip: Collection[str] = ()) -> str: def input_echo_ids(process: ET.Element) -> set[str]: """Where a process input shows up unchanged in variables-all: the - `uipath:input` ids and names, the variables a start event copies them - into verbatim, and the start events themselves. + `uipath:input` ids and names, every variable and output name a mapping + copies one into verbatim (transitively), and the start events. - # on Start_1 - -> {"input_Var_Amount", "Amount", "Var_Amount", "Start_1"} + + -> {"input_Var_Amount", "Amount", "Var_Amount", "Amt", "Start_1"} """ ids: set[str] = set() @@ -673,11 +683,29 @@ def input_echo_ids(process: ET.Element) -> set[str]: for node in variables.iter(q(UIPATH_NS, "input")): ids.update(v for v in (node.attrib.get("id"), node.attrib.get("name")) if v) - copies = {f"=vars.{identifier}" for identifier in ids} for start in process.iter(q(BPMN_NS, "startEvent")): if start.attrib.get("id"): ids.add(start.attrib["id"]) - for mapping in start.iter(q(UIPATH_NS, "output")): - if (mapping.attrib.get("source") or "").strip() in copies and mapping.attrib.get("var"): - ids.add(mapping.attrib["var"]) + + copies: list[tuple[str, str, str | None]] = [] + for element in process.iter(): + extensions = element.find(q(BPMN_NS, "extensionElements")) + if extensions is None: + continue + for mapping in extensions.iter(q(UIPATH_NS, "output")): + source = (mapping.attrib.get("source") or "").strip() + var = mapping.attrib.get("var") + if source.startswith("=vars.") and var: + copies.append((source.removeprefix("=vars."), var, mapping.attrib.get("name"))) + + grew = True + while grew: + grew = False + for source, var, name in copies: + if source not in ids or var in ids: + continue + ids.add(var) + if name: + ids.add(name) + grew = True return ids diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index 80bcd0982b..6892f65de5 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -14,8 +14,8 @@ BPMN has no fixed home for `entityName` (BATCH1-ADDENDUM.md "Where connector node values live in BPMN" and its "CI run 35488848026" section on the two valid activity shapes). A node names its entity when any one of these -equals it: the generic-form `objectName`, a `target="path"` input value, or a -`/`-separated segment of the context `path` field. +equals it: the generic-form `objectName`, any input's value (path, query, +body or untargeted), or a `/`-separated segment of the context `path` field. Assertion map (Flow -> BPMN): F check_smoke_error.py:29-31 entity_of(node) == NonExistentEntity on a @@ -27,7 +27,7 @@ I locate/parse .bpmn -> parse_bpmn() T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() T entity as the generic objectName, an exact -> mentions_entity() - target="path" input value, or an exact + input value on any target, or an exact context `path` segment DROPPED topology/parallel-branch parsing (Flow's own grader does not parse it either -- see its docstring) DROPPED require_no_private_connector_values (not in Flow) @@ -81,8 +81,7 @@ def mentions_entity(task: ET.Element, entity: str) -> bool: if entity in context_value(task, "path").strip().split("/"): return True return any( - inp.attrib.get("target") == "path" - and (inp.attrib.get("value") or inp.text or "").strip() == entity + (inp.attrib.get("value") or inp.text or "").strip() == entity for inp in context_inputs(task) ) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py index c408d97fec..3dfe6d6db8 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_jira_create_issue.py @@ -105,6 +105,7 @@ CheckFailure, DebugEvidence, connector_context, + connector_response_values, element_output_records, fetch_incidents, fetch_variables, @@ -188,7 +189,12 @@ def collect_candidate_keys( variables_data: object, create_ids: tuple[str, ...], project: str ) -> list[str]: outputs = element_output_records(variables_data, create_ids) - cands = re.findall(rf"\b{re.escape(project)}-\d+\b", json.dumps(outputs, default=str)) + key_re = re.compile(rf"{re.escape(project)}-\d+") + cands = [ + key.strip() + for key in connector_response_values(outputs, "key") + if isinstance(key, str) and key_re.fullmatch(key.strip()) + ] return list(dict.fromkeys(cands)) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py index 679f6e69f6..e4cf82d6f0 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_bpmn_live.py @@ -2,9 +2,12 @@ from __future__ import annotations +import ast +import glob import os import sys import xml.etree.ElementTree as ET +from pathlib import Path sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) @@ -68,6 +71,19 @@ def test_output_leaves_skips_input_echoes() -> None: assert "MCS-1" in bpmn_live.output_leaves(VARIABLES) +def test_input_echo_ids_follows_copies_through_any_element() -> None: + process = PROCESS.replace( + "", + "" + "" + "" + "", + ) + ids = bpmn_live.input_echo_ids(ET.fromstring(process)) + assert {"Var_Work", "work"} <= ids + assert "Task_Copy" not in ids + + def _evidence(incidents) -> bpmn_live.DebugEvidence: return bpmn_live.DebugEvidence(VARIABLES, "", incidents, incidents) @@ -87,3 +103,69 @@ def test_require_clean_run_accepts_a_completed_run() -> None: def test_require_clean_run_rejects(debug_data, incidents, message) -> None: with pytest.raises(bpmn_live.CheckFailure, match=message): bpmn_live.require_clean_run(debug_data, _evidence(incidents)) + + +CONNECTOR_PROCESS = f""" + + + + + + + + + + + + + +""" + + +def test_index_runtime_connectors_keys_carry_key_path_and_object() -> None: + connectors = bpmn_live.index_runtime_connectors(ET.fromstring(CONNECTOR_PROCESS)) + assert connectors == { + ("uipath-salesforce-slack", "/send_message_to_channel_v2", "send_message_to_channel_v2"): ("Task_Send",) + } + + +def _connector_unpacks(tree: ast.AST) -> list[tuple[int, int]]: + """(line, arity) of every `for (...), ids in .items()` over a name + bound to index_runtime_connectors(...).""" + bound = { + target.id + for node in ast.walk(tree) + if isinstance(node, ast.Assign) + and isinstance(node.value, ast.Call) + and getattr(node.value.func, "id", getattr(node.value.func, "attr", None)) == "index_runtime_connectors" + for target in node.targets + if isinstance(target, ast.Name) + } + found = [] + for node in ast.walk(tree): + if not isinstance(node, (ast.For, ast.comprehension)): + continue + source = node.iter + if not ( + isinstance(source, ast.Call) + and isinstance(source.func, ast.Attribute) + and source.func.attr == "items" + and isinstance(source.func.value, ast.Name) + and source.func.value.id in bound + ): + continue + target = node.target + if isinstance(target, ast.Tuple) and isinstance(target.elts[0], ast.Tuple): + found.append((target.lineno, len(target.elts[0].elts))) + return found + + +@pytest.mark.parametrize( + "script", + sorted(glob.glob(os.path.join(os.path.dirname(os.path.abspath(__file__)), "check_*.py"))), + ids=os.path.basename, +) +def test_graders_unpack_the_full_connector_key(script: str) -> None: + tree = ast.parse(Path(script).read_text(encoding="utf-8")) + bad = [line for line, arity in _connector_unpacks(tree) if arity != 3] + assert not bad, f"{os.path.basename(script)} unpacks index_runtime_connectors keys at lines {bad}" From 3f8079fb4cc57ccc7b96c27fd6c3192a67059acd Mon Sep 17 00:00:00 2001 From: rockymadden Date: Thu, 24 Sep 2026 15:00:59 -0600 Subject: [PATCH 28/35] test(bpmn): let the orchestrator's caseKey match its correlationId input The contract sets caseKey to the correlationId input, so skipping input echoes for it failed every correct run (eval run 36056004092). Co-Authored-By: Claude Opus 5.5 (1M context) --- .../_shared/check_escalation_orchestrator_paths.py | 7 +++++-- 1 file changed, 5 insertions(+), 2 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py index 80d191d520..608d0c65b7 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_escalation_orchestrator_paths.py @@ -24,7 +24,8 @@ -> assert_send_identity(): every Slack sendTask carries target=query name=send_as value=user F check_escalation_orchestrator_paths.py:55 assert_named_equals(payload, name, expected, ...) -> public_value_present(): normalized value present among root Globals leaves + non-classifier element Outputs - leaves, input echoes (input_echo_ids) excluded from both (see NOTE below) + leaves, input echoes (input_echo_ids) excluded from both except for caseKey, which the + contract sets to the correlationId input (see NOTE below) F check_escalation_orchestrator_paths.py:65 assert_slack_message_posted(payload, "slackMessageId", ...) -> assert_slack_posted(): fired Slack sendTask's own Outputs.response carries a ts-shaped id, the seeded channel, and correlationId + escalationPath in its text @@ -116,6 +117,7 @@ BPMN_NAME = f"{PROJECT_NAME}.bpmn" CASE_SENSITIVE = {"caseKey"} # opaque id -- exact-case match +PASSTHROUGH_FIELDS = {"caseKey"} # the contract sets it to the correlationId input CLASSIFICATION_FIELDS = ("escalationPath", "severity", "engineeringNeeded", "responseMode") NAMED_OUTPUT_FIELDS = CLASSIFICATION_FIELDS + ("caseKey",) @@ -451,7 +453,8 @@ def verify_case(contract: Contract, imported_project: Path, case: dict) -> dict: if not public_value_present( variables_data, expected_value, - exclude_ids=contract.classifier_ids + contract.echo_ids, + exclude_ids=contract.classifier_ids + + (() if field in PASSTHROUGH_FIELDS else contract.echo_ids), case_sensitive=field in CASE_SENSITIVE, ): raise CheckFailure(f"{case['name']}: no public output carries {field}={expected_value!r}") From 2e9b121132af3d19f8bbf93fd909e3d920f7963d Mon Sep 17 00:00:00 2001 From: rockymadden Date: Thu, 24 Sep 2026 15:14:08 -0600 Subject: [PATCH 29/35] docs(bpmn): record the post-review eval runs in the parity ledger Co-Authored-By: Claude Opus 5.5 (1M context) --- .../_porting/parity-ledger.md | 52 ++++++++++--------- 1 file changed, 27 insertions(+), 25 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index 5787858580..67d83c200e 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -13,14 +13,14 @@ One row per task, final state. Iterations are summarised in the notes; runs are | `…/datafabric_connector/integration_create_get.yaml` | `…/integration_create_get/` | PASS it.2 (run 35489744689) | same | | `…/datafabric_connector/contractregistry_crud_filters.yaml` | `…/contractregistry_crud_filters/` | PASS it.2 (run 35489744689) | same, plus transitive output mapping | | `…/datafabric_connector/smoke_query.yaml` | `…/smoke_query/` | SKIPPED, surface gap | passed it.3 (run 35490499651) only via an invented `sortField` input; the curated Query Entity Records template has no sort-field parameter (only `isAscending`), so the "sorted by score" criterion has no carrier. Two smoke-gate runs (35674400362) confirmed. | -| `…/datafabric_connector/smoke_update.yaml` | `…/smoke_update/` | PASS (run 35499789502) | | -| `…/datafabric_connector/smoke_file_activities.yaml` | `…/smoke_file_activities/` | PASS (run 35499789502) | | -| `…/datafabric_connector/e2e_contract_intake_pipeline.yaml` | `…/e2e_contract_intake_pipeline/` | PASS (run 35499789502) | | +| `…/datafabric_connector/smoke_update.yaml` | `…/smoke_update/` | PASS (run 36056004092) | earlier PASS (run 35499789502) | +| `…/datafabric_connector/smoke_file_activities.yaml` | `…/smoke_file_activities/` | PASS (run 36056004092) | earlier PASS (run 35499789502) | +| `…/datafabric_connector/e2e_contract_intake_pipeline.yaml` | `…/e2e_contract_intake_pipeline/` | PASS (run 36056004092) | earlier PASS (run 35499789502) | | `…/datafabric_connector/trigger_lifecycle.yaml` | `…/trigger_lifecycle/` | PASS it.3 (run 35501830119) | it.1 agent omitted the entity parameter; it.2 grader required messageEventDefinition as a direct child | | `…/datafabric_connector/smoke_update_existing_flow.yaml` | `…/smoke_update_existing_flow/` | PASS (run 35500726138) | brownfield scaffold via `bpmn init` | | `connector_features/testmanager_{testcase,testset,requirement}_lifecycle`, `testmanager_{attachments,execution_results,generic_records}` | same names | PASS (runs 35488848026, 35499789502) | Flow's `skip: true` not carried over | | `connector_features/non-catalog-http-fallback/…` | `connector_features/non_catalog_http_fallback/` | PASS (run 35500726138) | grader now ActivityExecution-only (#3476) | -| `single_node/outlook_waitfor_email/…` | same | PASS (run 35500726138) | | +| `single_node/outlook_waitfor_email/…` | same | PASS (run 36056004092) | earlier PASS (run 35500726138) | | `single_node/outlook_trigger_inbox/…` | same | PASS it.2 (run 35501830119) | it.1 agent omitted `parentFolderId`; `uip is triggers` advisory stays 0 | | `e2e/devcon_expense_approval.yaml` | `e2e/devcon_expense_approval/` | SKIPPED, platform gap (0.885 it.3, run 35503094182) | every HITL assertion passes; `validate` needs a deployed Action App for Actions.HITL | | `interactive/customer_escalation_simulated/…` | same | PASS 0.94 (run 35501830119) | advisory name check missed, as in Flow | @@ -31,30 +31,32 @@ One row per task, final state. Iterations are summarised in the notes; runs are ### Live and field-shape (PR #3502) +Final is the latest run after the review fixes; each row's earlier result is kept in its notes. + | Flow task | BPMN port | Final | Notes | |---|---|---|---| -| `e2e/jira_get_issue/…` | same | PASS (run 35501830119) | live pilot: ephemeral solution + `bpmn debug` + `variables-all` | -| `e2e/jira_create_issue/…` | same | PASS (run 35503094182) | | -| `e2e/escalation_jira_ticket/…` | same | PASS (run 35503094182) | | -| `e2e/escalation_orchestrator_paths/…` | same | PASS (run 35503094182) | 7 debug runs | -| `e2e/escalation_slack_alert/…` | same | PASS it.2 (run 35524004307) | it.1 agent omitted the Slack `folderKey` binding (102010) | -| `multi_node/slack_channel_description/…` | same | PASS it.2 (run 35525387843) | it.1 agent omitted the channel parameter | -| `connector_features/datafabric_connector/smoke_error.yaml` | `…/smoke_error/` | PASS (run 35538279757) | structural | -| `connector_features/generic_dynamic_node/…` | same | PASS (run 35538279757) | | -| `connector_features/jdbc_databricks_query/…` | same | PASS (run 35538279757) | structural | -| `connector_features/slack-http-fallback/…` | `connector_features/slack_http_fallback/` | PASS (run 35783045540) | it.1 grader wanted `emoji.list`; the connector's generic resource is `emoji_list_GET` | -| `connector_trigger/webhook_waitfor_parallel.yaml` | same | PASS (run 35783045540) | it.1 grader classified the wait by BPMN tag; agent emits `intermediateCatchEvent` + `Intsvc.WaitForEvent` and `Intsvc.UnifiedHttpRequest` | +| `e2e/jira_get_issue/…` | same | PASS (run 36056004092) | earlier PASS (run 35501830119); live pilot: ephemeral solution + `bpmn debug` + `variables-all` | +| `e2e/jira_create_issue/…` | same | PASS (run 36056004092) | earlier PASS (run 35503094182) | +| `e2e/escalation_jira_ticket/…` | same | PASS (run 36056004092) | earlier PASS (run 35503094182) | +| `e2e/escalation_orchestrator_paths/…` | same | PASS (run 36058708886) | earlier PASS (run 35503094182); 7 debug runs | +| `e2e/escalation_slack_alert/…` | same | FAIL (runs 36053143338, 36056004092) | after the review fixes the Slack step faulted at runtime in both runs (102010 `folderKey` in the first); earlier PASS it.2 (run 35524004307); it.1 agent omitted the Slack `folderKey` binding (102010) | +| `multi_node/slack_channel_description/…` | same | PASS (run 36056004092) | earlier PASS it.2 (run 35525387843); it.1 agent omitted the channel parameter | +| `connector_features/datafabric_connector/smoke_error.yaml` | `…/smoke_error/` | PASS (run 36056004092) | earlier PASS (run 35538279757); structural | +| `connector_features/generic_dynamic_node/…` | same | PASS (run 36056004092) | earlier PASS (run 35538279757) | +| `connector_features/jdbc_databricks_query/…` | same | PASS (run 36056004092) | earlier PASS (run 35538279757); structural | +| `connector_features/slack-http-fallback/…` | `connector_features/slack_http_fallback/` | PASS (run 36056004092) | earlier PASS (run 35783045540); it.1 grader wanted `emoji.list`; the connector's generic resource is `emoji_list_GET` | +| `connector_trigger/webhook_waitfor_parallel.yaml` | same | PASS (run 36053143338) | run 36056004092 failed on an agent-written `` (invalid tag); earlier PASS (run 35783045540); it.1 grader classified the wait by BPMN tag; agent emits `intermediateCatchEvent` + `Intsvc.WaitForEvent` and `Intsvc.UnifiedHttpRequest` | | `connector_features/testmanager_crud_grounded/…` | same | SKIPPED, as in Flow | grades an agent-written `result.json`; unskip once it grades the live run | -| `interactive/cli_dice_roller_simulated/…` | same | PASS (run 35783045540) | | -| `multi_node/billing_invoice_lookup/…` | same | PASS (run 35785806030) | it.1 grader read its own ephemeral live solution as a second project; run 35783045540 was a platform 504 | -| `multi_node/slack_weather_pipeline/…` | same | PASS it.3 (run 35785806030) | it.1 channel not found, it.2 wrong Slack connection bound (401); Flow passes 6/12 nightlies | -| `connector_features/enum.yaml` | `connector_features/enum/` | PASS (run 35789221753) | | -| `connector_features/query_params.yaml` | `connector_features/query_params/` | PASS (run 35789221753) | | -| `connector_features/multiselect.yaml` | `connector_features/multiselect/` | PASS (run 35789221753) | | -| `connector_features/searchable_joins.yaml` | `connector_features/searchable_joins/` | PASS (run 35789221753) | | -| `connector_features/complex_array.yaml` | `connector_features/complex_array/` | PASS 0.875 (run 35789221753) | only the advisory resolved-user-id check missed; Flow fully passes 3/12 | -| `connector_features/path_params.yaml` | `connector_features/path_params/` | PASS it.2 (run 35790934047) | it.1 agent left the issue key as an unbound variable | -| `connector_features/paginated_reference_lookup.yaml` | `connector_features/paginated_reference_lookup/` | PASS it.3 (run 35791969905) | it.1 channel by name, no pagination; it.2 channel id with its last character dropped | +| `interactive/cli_dice_roller_simulated/…` | same | PASS (run 36056004092) | earlier PASS (run 35783045540) | +| `multi_node/billing_invoice_lookup/…` | same | PASS (run 36056004092) | earlier PASS (run 35785806030); it.1 grader read its own ephemeral live solution as a second project; run 35783045540 was a platform 504 | +| `multi_node/slack_weather_pipeline/…` | same | PASS (run 36056004092) | earlier PASS it.3 (run 35785806030); it.1 channel not found, it.2 wrong Slack connection bound (401); Flow passes 6/12 nightlies | +| `connector_features/enum.yaml` | `connector_features/enum/` | PASS (run 36056004092) | earlier PASS (run 35789221753) | +| `connector_features/query_params.yaml` | `connector_features/query_params/` | PASS (run 36056004092) | earlier PASS (run 35789221753) | +| `connector_features/multiselect.yaml` | `connector_features/multiselect/` | PASS (run 36056004092) | earlier PASS (run 35789221753) | +| `connector_features/searchable_joins.yaml` | `connector_features/searchable_joins/` | PASS (run 36056004092) | earlier PASS (run 35789221753) | +| `connector_features/complex_array.yaml` | `connector_features/complex_array/` | PASS 0.875 (run 36056004092) | earlier PASS 0.875 (run 35789221753); only the advisory resolved-user-id check missed; Flow fully passes 3/12 | +| `connector_features/path_params.yaml` | `connector_features/path_params/` | PASS (run 36056004092) | earlier PASS it.2 (run 35790934047); it.1 agent left the issue key as an unbound variable | +| `connector_features/paginated_reference_lookup.yaml` | `connector_features/paginated_reference_lookup/` | FAIL (runs 36053143338, 36056004092) | after the review fixes the agent never resolved the channel id in either run; earlier PASS it.3 (run 35791969905); it.1 channel by name, no pagination; it.2 channel id with its last character dropped | ### Parked (branch `test/bpmn-port-parked`, all `skip: true`) From 4febd24198287c8d45560093347cfb41c8173a96 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Thu, 24 Sep 2026 23:20:38 -0700 Subject: [PATCH 30/35] test(bpmn): accept a whole-valued float roll; record why escalation_slack_alert faulted Jint serialises numbers as doubles, so a correct dice roll can read back as 5.0 from variables-all; find_int_in_range now takes it (7.5, 0, 7 and booleans still fail), with a unit test. The ledger row for escalation_slack_alert now carries the per-run incident evidence: no folderKey input in run 36053143338, missing required send_as in run 36056004092. The grader judged both correctly. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_porting/parity-ledger.md | 2 +- .../_shared/check_dice_runs_simulated.py | 8 +++-- .../_shared/test_check_dice_runs_simulated.py | 32 +++++++++++++++++++ 3 files changed, 39 insertions(+), 3 deletions(-) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/test_check_dice_runs_simulated.py diff --git a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md index 67d83c200e..a1c7394b5c 100644 --- a/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md +++ b/tests/tasks/uipath-maestro-bpmn/_porting/parity-ledger.md @@ -39,7 +39,7 @@ Final is the latest run after the review fixes; each row's earlier result is kep | `e2e/jira_create_issue/…` | same | PASS (run 36056004092) | earlier PASS (run 35503094182) | | `e2e/escalation_jira_ticket/…` | same | PASS (run 36056004092) | earlier PASS (run 35503094182) | | `e2e/escalation_orchestrator_paths/…` | same | PASS (run 36058708886) | earlier PASS (run 35503094182); 7 debug runs | -| `e2e/escalation_slack_alert/…` | same | FAIL (runs 36053143338, 36056004092) | after the review fixes the Slack step faulted at runtime in both runs (102010 `folderKey` in the first); earlier PASS it.2 (run 35524004307); it.1 agent omitted the Slack `folderKey` binding (102010) | +| `e2e/escalation_slack_alert/…` | same | FAIL (runs 36053143338, 36056004092); grader verified | both runs faulted on the agent's Slack send node, not the grader: run 36053143338 has NO `folderKey` input on the node (incident 102010 "Value cannot be null (Parameter 'Folder')"); run 36056004092 has the folderKey binding but omits the required `send_as` parameter (incident 102003, IS 400 "Value for required parameter 'send_as' not found"). The grader reads `final status Faulted` off variables-all/incidents, as it did when it.2 passed (run 35524004307). Skill finding: Slack send-message node authoring (folderKey binding, required `send_as`). | | `multi_node/slack_channel_description/…` | same | PASS (run 36056004092) | earlier PASS it.2 (run 35525387843); it.1 agent omitted the channel parameter | | `connector_features/datafabric_connector/smoke_error.yaml` | `…/smoke_error/` | PASS (run 36056004092) | earlier PASS (run 35538279757); structural | | `connector_features/generic_dynamic_node/…` | same | PASS (run 36056004092) | earlier PASS (run 35538279757) | diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py index 9eb7a59f4f..e65fefe117 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_dice_runs_simulated.py @@ -111,14 +111,18 @@ def find_int_in_range(leaves: list, lo: int, hi: int) -> int | None: - """First whole-integer leaf in [lo, hi]: an int, or a string that is only - an integer. `roll: 7.5` or a timestamp never yields a digit run. + """First whole-integer leaf in [lo, hi]: an int, a whole-valued float + (the script task runs on Jint, whose numbers are doubles, so a correct + roll can read back as `5.0`), or a string that is only an integer. + `roll: 7.5` or a timestamp never qualifies. """ for leaf in leaves: if isinstance(leaf, bool): continue if isinstance(leaf, int): value = leaf + elif isinstance(leaf, float) and leaf.is_integer(): + value = int(leaf) elif isinstance(leaf, str) and INT_RE.fullmatch(leaf.strip()): value = int(leaf.strip()) else: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_check_dice_runs_simulated.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_dice_runs_simulated.py new file mode 100644 index 0000000000..3e9f440b5a --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_dice_runs_simulated.py @@ -0,0 +1,32 @@ +"""Unit tests for check_dice_runs_simulated's roll-value rule.""" + +from __future__ import annotations + +import importlib.util +import os +import sys +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + + +def _load(): + path = Path(__file__).parent / "check_dice_runs_simulated.py" + spec = importlib.util.spec_from_file_location("_check_dice_runs_simulated", path) + module = importlib.util.module_from_spec(spec) + spec.loader.exec_module(module) + return module + + +def test_find_int_in_range_accepts_a_whole_valued_float() -> None: + """Jint serialises numbers as doubles, so a correct roll can read back as + 5.0; 7.5, 0, 7 and booleans still fail, and int / digit-string pass.""" + grader = _load() + f = grader.find_int_in_range + assert f([5.0], 1, 6) == 5 + assert f([5], 1, 6) == 5 + assert f(["5"], 1, 6) == 5 + assert f([7.5], 1, 6) is None + assert f([0, 7, 6.5], 1, 6) is None + assert f([True], 1, 6) is None + assert f([True, 6.0], 1, 6) == 6 From 7dc6bd6e1017f7cb3971e69aac25d036d656d328 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Fri, 25 Sep 2026 01:35:45 -0700 Subject: [PATCH 31/35] test(bpmn): stop staging the skill directory into the live-port sandboxes Main's task-driver gate (2026-09-23) forbids a template_sources entry for skills/uipath-maestro-bpmn: the skill arrives through agent.plugins, which is what selects the generation under test, and a cwd copy hands a preview-arm run the shipped v1 guidance. The 22 ported tasks drop the entry; sandbox blocks that only carried it are removed. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../connector_features/complex_array/complex_array.yaml | 2 -- .../datafabric_connector/smoke_error/smoke_error.yaml | 2 -- .../uipath-maestro-bpmn/connector_features/enum/enum.yaml | 2 -- .../generic_dynamic_node/generic_dynamic_node.yaml | 2 -- .../jdbc_databricks_query/jdbc_databricks_query.yaml | 2 -- .../connector_features/multiselect/multiselect.yaml | 2 -- .../paginated_reference_lookup/paginated_reference_lookup.yaml | 2 -- .../connector_features/path_params/path_params.yaml | 2 -- .../connector_features/query_params/query_params.yaml | 2 -- .../connector_features/searchable_joins/searchable_joins.yaml | 2 -- .../slack_http_fallback/slack_http_fallback.yaml | 2 -- .../testmanager_crud_grounded/testmanager_crud_grounded.yaml | 2 -- .../webhook_waitfor_parallel/webhook_waitfor_parallel.yaml | 2 -- .../e2e/escalation_jira_ticket/escalation_jira_ticket.yaml | 2 -- .../escalation_orchestrator_paths.yaml | 2 -- .../e2e/escalation_slack_alert/escalation_slack_alert.yaml | 2 -- .../e2e/jira_create_issue/jira_create_issue.yaml | 2 -- .../uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml | 2 -- .../cli_dice_roller_simulated/cli_dice_roller_simulated.yaml | 2 -- .../billing_invoice_lookup/billing_invoice_lookup.yaml | 2 -- .../slack_channel_description/slack_channel_description.yaml | 2 -- .../slack_weather_pipeline/slack_weather_pipeline.yaml | 3 --- 22 files changed, 45 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml index 3e1f346085..e3bbb52ae9 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/complex_array/complex_array.yaml @@ -28,8 +28,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn reference: directory: ../.. diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml index 1529493a29..752f1ced86 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/datafabric_connector/smoke_error/smoke_error.yaml @@ -15,8 +15,6 @@ tags: [uipath-maestro-bpmn, smoke, "mode:build", "lifecycle:generate", connector sandbox: template_sources: - - type: template_dir - path: ../../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml index 8dff7b6930..c64469fa91 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/enum/enum.yaml @@ -25,8 +25,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn reference: directory: ../.. diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml index 2af3d48b50..ed29e4ee9f 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/generic_dynamic_node/generic_dynamic_node.yaml @@ -38,8 +38,6 @@ tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:sing sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml index e3a44ea379..10040d6192 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/jdbc_databricks_query/jdbc_databricks_query.yaml @@ -24,8 +24,6 @@ description: > tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:single-node", connector] sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml index 9ecbcc3c61..32872a2790 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/multiselect/multiselect.yaml @@ -27,8 +27,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn reference: directory: ../.. diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml index e80f4513dd..c836062e91 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/paginated_reference_lookup/paginated_reference_lookup.yaml @@ -35,8 +35,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn reference: directory: ../.. diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml index 42410410ff..bcd360473c 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/path_params/path_params.yaml @@ -26,8 +26,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn reference: directory: ../.. diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml index 2a00469271..7c3f0f418a 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/query_params/query_params.yaml @@ -27,8 +27,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn reference: directory: ../.. diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml index 2a88f576e3..81fbd4be7a 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/searchable_joins/searchable_joins.yaml @@ -20,8 +20,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn reference: directory: ../.. diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml index 0795b96b78..2fe610f0cf 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/slack_http_fallback/slack_http_fallback.yaml @@ -29,8 +29,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml index 02203dcd94..1282174ea1 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_features/testmanager_crud_grounded/testmanager_crud_grounded.yaml @@ -23,8 +23,6 @@ skip: true sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: _setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml b/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml index 8df95d115a..50dbd78784 100644 --- a/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml +++ b/tests/tasks/uipath-maestro-bpmn/connector_trigger/webhook_waitfor_parallel/webhook_waitfor_parallel.yaml @@ -31,8 +31,6 @@ tags: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn run_limits: expected_turns: 35 diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml index 83999d87a0..efac1abc11 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_jira_ticket/escalation_jira_ticket.yaml @@ -39,8 +39,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml index 6a6ac10fd9..b79bcebb03 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_orchestrator_paths/escalation_orchestrator_paths.yaml @@ -28,8 +28,6 @@ tags: [uipath-maestro-bpmn, e2e, mode:build, lifecycle:generate, shape:multi-nod sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml index 37b26b1507..23fab06ba8 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/escalation_slack_alert/escalation_slack_alert.yaml @@ -33,8 +33,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml index 8a3cca1659..036366f466 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_create_issue/jira_create_issue.yaml @@ -33,8 +33,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml index 9501d14ba8..f9dbe48a94 100644 --- a/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml +++ b/tests/tasks/uipath-maestro-bpmn/e2e/jira_get_issue/jira_get_issue.yaml @@ -32,8 +32,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml index d57b9707f2..d89e6c5c1f 100644 --- a/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml +++ b/tests/tasks/uipath-maestro-bpmn/interactive/cli_dice_roller_simulated/cli_dice_roller_simulated.yaml @@ -33,8 +33,6 @@ tags: [uipath-maestro-bpmn, e2e, "mode:build", "lifecycle:generate", "shape:mult sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml index cd67c4b98f..a2d315f389 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/billing_invoice_lookup/billing_invoice_lookup.yaml @@ -33,8 +33,6 @@ tags: - outcome-graded sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml index fb97a85649..5af563840d 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_channel_description/slack_channel_description.yaml @@ -30,8 +30,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup diff --git a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml index 96ef3cca68..73acef939b 100644 --- a/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml +++ b/tests/tasks/uipath-maestro-bpmn/multi_node/slack_weather_pipeline/slack_weather_pipeline.yaml @@ -28,8 +28,6 @@ run_limits: sandbox: template_sources: - - type: template_dir - path: ../../../../../skills/uipath-maestro-bpmn - type: template_dir path: ../../_setup mount_point: _setup @@ -37,7 +35,6 @@ sandbox: reference: directory: ../.. - initial_prompt: | Build the process inside a solution of the same name. From a009d2b4aafb88326bbba959517c4bc23ec569ca Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Sun, 27 Sep 2026 17:23:28 -0600 Subject: [PATCH 32/35] test(bpmn): grade smoke_error over every .bpmn, as Flow's grader globs every .flow The agent split the error path into its own file beside the main process (smoke run 36357685718), and the single-file lookup called that ambiguity. Flow's check_smoke_error.py iterates every .flow and passes when any file satisfies the shape; the port now does the same, printing each file's reason before failing. Replays green on the new artifact and on the batch-10 one. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_df_smoke_error.py | 70 ++++++++++++++----- 1 file changed, 52 insertions(+), 18 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index 6892f65de5..17676125d2 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -24,7 +24,9 @@ `.query-entity-records` nodes -> is_query_node() + mentions_entity() F check_smoke_error.py:37-39 `not error_creates` -> fail -> `error_creates < 1` check F check_smoke_error.py:40-42 `len(good_queries) < 2` -> fail -> `good_queries < 2` check - I locate/parse .bpmn -> parse_bpmn() + F check_smoke_error.py:21 `for path in glob("**/*.flow")`: the first -> candidate_files(): every .bpmn under + file satisfying the shape passes, else fail the sandbox, first satisfying file passes + I parse each .bpmn -> ET.parse() T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() T entity as the generic objectName, an exact -> mentions_entity() input value on any target, or an exact @@ -50,6 +52,7 @@ import os import re import sys +from pathlib import Path import xml.etree.ElementTree as ET sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) @@ -61,7 +64,6 @@ elements, fail, has_typed_uipath_extension, - parse_bpmn, ) CONNECTOR_KEY = "uipath-uipath-dataservice" @@ -119,33 +121,65 @@ def is_query_node(task: ET.Element, object_name: str, entity: str) -> bool: return bool(_LIST_OP_RE.match(operation)) or method == "GET" -def main() -> None: - path, root = parse_bpmn() +SKIP_PARTS = {"node_modules", ".npm-prefix", ".venv"} + + +def candidate_files() -> list[Path]: + """Every .bpmn under the sandbox, as Flow's grader globbed every .flow. + + The agent may split the error path into its own file beside the main + process (smoke run 36357685718 left DataFabricSmokeError.bpmn and + DataFabricSmokeError_error.bpmn in one project); Flow passes when ANY + file satisfies the shape, so the port does the same. + """ + return sorted( + p for p in Path.cwd().rglob("*.bpmn") if not (SKIP_PARTS & set(p.parts)) + ) + +def shape_of(root: ET.Element) -> tuple[int, int]: error_creates = 0 good_queries = 0 - for task in connector_nodes(root): object_name = context_value(task, "objectName") if is_create_node(task, object_name, CREATE_ENTITY) and mentions_entity(task, CREATE_ENTITY): error_creates += 1 if is_query_node(task, object_name, QUERY_ENTITY) and mentions_entity(task, QUERY_ENTITY): good_queries += 1 + return error_creates, good_queries - if error_creates < 1: - fail(f"no Create Entity Record node targeting {CREATE_ENTITY!r}") - print(f"OK: {error_creates} Create Entity Record node(s) on {CREATE_ENTITY}") - - if good_queries < 2: - fail( - f"expected >=2 Query Entity Records node(s) on {QUERY_ENTITY!r}, " - f"found {good_queries}" - ) - print(f"OK: {good_queries} Query Entity Records node(s) on {QUERY_ENTITY}") - print( - f"OK: {path} -- create on {CREATE_ENTITY}, {good_queries} query on {QUERY_ENTITY}" - ) +def main() -> None: + files = candidate_files() + if not files: + fail("no BPMN file found") + + for path in files: + try: + root = ET.parse(path).getroot() + except ET.ParseError as exc: + print(f"FAIL: {path} is not well-formed XML: {exc}", file=sys.stderr) + continue + error_creates, good_queries = shape_of(root) + if error_creates < 1: + print( + f"FAIL: {path} -- no Create Entity Record node targeting {CREATE_ENTITY!r}", + file=sys.stderr, + ) + continue + if good_queries < 2: + print( + f"FAIL: {path} -- expected >=2 Query Entity Records node(s) on " + f"{QUERY_ENTITY!r}, found {good_queries}", + file=sys.stderr, + ) + continue + print(f"OK: {error_creates} Create Entity Record node(s) on {CREATE_ENTITY}") + print(f"OK: {good_queries} Query Entity Records node(s) on {QUERY_ENTITY}") + print(f"OK: {path} -- create on {CREATE_ENTITY}, {good_queries} query on {QUERY_ENTITY}") + return + + fail("no .bpmn satisfies the error-path shape") if __name__ == "__main__": From dc664129390a69fdb884f83252613902306232ee Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Mon, 28 Sep 2026 09:25:37 -0600 Subject: [PATCH 33/35] test(bpmn): grade smoke_error only on the project's entry-point files The any-file scan passed run 36357685718 on DataFabricSmokeError_error.bpmn, a file that is not an entry point: entry-points.json named only the empty scaffold, so the process that runs carries nothing. A Flow project is one file; a BPMN project holds several and runs only its entry points. The scan is now scoped to files beside a project.uiproj when any project exists (a scratch copy never outranks the project; a bare .bpmn with no project still counts, as find_bpmn_file does) and then to the entry points named by that project's entry-points.json, falling back to every project file before refresh has written it. bpmn_check.entry_point_files holds the parsing. Seven unit tests cover the run's layout, the passing layout, a malformed sibling, the pre-refresh fallback, a scratch copy, a bare file and the empty sandbox. Replay on run 36357685718 now fails naming the non-entry file. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../uipath-maestro-bpmn/_shared/bpmn_check.py | 27 +++++ .../_shared/check_df_smoke_error.py | 62 ++++++++-- .../_shared/test_check_df_smoke_error.py | 111 ++++++++++++++++++ 3 files changed, 187 insertions(+), 13 deletions(-) create mode 100644 tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py index cd271646f8..b3a7d822d4 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/bpmn_check.py @@ -37,6 +37,33 @@ def _project_files(paths: Iterable[_PathLike]) -> list[_PathLike]: return [p for p in paths if (Path(p).parent / "project.uiproj").is_file()] +def entry_point_files(project_dir: Path) -> list[Path] | None: + """The ``.bpmn`` files ``entry-points.json`` names in ``project_dir``. + + ``filePath`` is written as ``/content/.bpmn#``; + the ``/content/`` prefix and the ``#…`` suffix are stripped. Returns + ``None`` when the project has no ``entry-points.json`` yet (``refresh`` + has not run), so a caller can fall back to every project file. A BPMN + project can hold several ``.bpmn`` files and only the entry points run: + smoke run 36357685718 left the graded shape in a non-entry file beside an + empty scaffold that was the only entry point. + """ + manifest = Path(project_dir) / "entry-points.json" + if not manifest.is_file(): + return None + try: + data = json.loads(manifest.read_text(encoding="utf-8")) + except (OSError, ValueError): + return [] + out: list[Path] = [] + for ep in data.get("entryPoints") or []: + file_path = str(ep.get("filePath") or "").split("#", 1)[0] + name = Path(file_path.removeprefix("/content/")).name + if name: + out.append(Path(project_dir) / name) + return out + + def find_bpmn_file(name_hint: str | None = None) -> str: paths = sorted(glob.glob("**/*.bpmn", recursive=True)) if not paths: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index 17676125d2..a0993ae2e2 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -24,8 +24,10 @@ `.query-entity-records` nodes -> is_query_node() + mentions_entity() F check_smoke_error.py:37-39 `not error_creates` -> fail -> `error_creates < 1` check F check_smoke_error.py:40-42 `len(good_queries) < 2` -> fail -> `good_queries < 2` check - F check_smoke_error.py:21 `for path in glob("**/*.flow")`: the first -> candidate_files(): every .bpmn under - file satisfying the shape passes, else fail the sandbox, first satisfying file passes + F check_smoke_error.py:21 `for path in glob("**/*.flow")`: the first -> candidate_files(): every project .bpmn + file satisfying the shape passes, else fail that is an entry point (a Flow project is + one file; a BPMN project runs only its + entry-points.json files) I parse each .bpmn -> ET.parse() T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() T entity as the generic objectName, an exact -> mentions_entity() @@ -58,6 +60,8 @@ sys.path.insert(0, os.path.dirname(os.path.dirname(os.path.abspath(__file__)))) from _shared.bpmn_check import ( # noqa: E402 + _project_files, + entry_point_files, NS, context_inputs, context_value, @@ -124,17 +128,37 @@ def is_query_node(task: ET.Element, object_name: str, entity: str) -> bool: SKIP_PARTS = {"node_modules", ".npm-prefix", ".venv"} -def candidate_files() -> list[Path]: - """Every .bpmn under the sandbox, as Flow's grader globbed every .flow. - - The agent may split the error path into its own file beside the main - process (smoke run 36357685718 left DataFabricSmokeError.bpmn and - DataFabricSmokeError_error.bpmn in one project); Flow passes when ANY - file satisfies the shape, so the port does the same. +def candidate_files() -> tuple[list[Path], list[Path]]: + """(graded, skipped): the ``.bpmn`` files this grader may pass on, and + the project files it deliberately ignores. + + Flow's grader globbed every ``.flow`` because a Flow project is one file. + A BPMN project holds several, and ``entry-points.json`` says which run, so + the scan is scoped twice: only files beside a ``project.uiproj`` when any + project exists (``bpmn_check._project_files``, so a scratch copy under + ``tmp/`` never outranks the project; a bare .bpmn with no project at all + still counts, as in find_bpmn_file), then only the entry points of that project when its + ``entry-points.json`` exists (before ``refresh`` writes it, every project + file counts). Smoke run 36357685718 left the graded shape in + ``DataFabricSmokeError_error.bpmn`` beside an empty scaffold that was the + project's only entry point; that project must fail. """ - return sorted( + every = sorted( p for p in Path.cwd().rglob("*.bpmn") if not (SKIP_PARTS & set(p.parts)) ) + # As find_bpmn_file: the project guard disambiguates when a project + # exists; a bare .bpmn with no project.uiproj anywhere (batch-10 run + # 35538279757 passed that way) is still the artifact. + in_project = _project_files(every) or every + graded: list[Path] = [] + skipped: list[Path] = [] + for path in in_project: + entry = entry_point_files(path.parent) + if entry is None or path in entry: + graded.append(path) + else: + skipped.append(path) + return graded, skipped def shape_of(root: ET.Element) -> tuple[int, int]: @@ -150,8 +174,8 @@ def shape_of(root: ET.Element) -> tuple[int, int]: def main() -> None: - files = candidate_files() - if not files: + files, skipped = candidate_files() + if not files and not skipped: fail("no BPMN file found") for path in files: @@ -179,7 +203,19 @@ def main() -> None: print(f"OK: {path} -- create on {CREATE_ENTITY}, {good_queries} query on {QUERY_ENTITY}") return - fail("no .bpmn satisfies the error-path shape") + for path in skipped: + try: + error_creates, good_queries = shape_of(ET.parse(path).getroot()) + except ET.ParseError: + continue + if error_creates >= 1 and good_queries >= 2: + print( + f"FAIL: the error-path shape is in {path}, which is not an entry point " + f"of its project (entry-points.json); the process that runs does not " + f"carry it", + file=sys.stderr, + ) + fail("no entry-point .bpmn satisfies the error-path shape") if __name__ == "__main__": diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py new file mode 100644 index 0000000000..3b26c0b35a --- /dev/null +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py @@ -0,0 +1,111 @@ +"""Unit tests for check_df_smoke_error's multi-file, entry-point-scoped scan.""" + +from __future__ import annotations + +import importlib.util +import json +import os +import subprocess +import sys +from pathlib import Path + +sys.path.insert(0, os.path.dirname(os.path.abspath(__file__))) + +GRADER = Path(__file__).parent / "check_df_smoke_error.py" +NS = ( + 'xmlns:bpmn="http://www.omg.org/spec/BPMN/20100524/MODEL" ' + 'xmlns:uipath="http://uipath.org/schema/bpmn"' +) + + +def _node(object_name: str, entity: str, method: str) -> str: + return ( + f'' + '' + '' + f'' + f'' + f'' + "" + ) + + +SHAPE = ( + f'' + + _node("CreateEntityRecordCurated", "NonExistentEntity", "POST") + + _node("QueryEntityRecordsCurated", "FlowCodeEvalEntity", "POST") + + _node("QueryEntityRecordsCurated", "FlowCodeEvalEntity", "POST") + + "" +) +EMPTY = ( + f'' + '' +) + + +def _project(root: Path, files: dict[str, str], entry: list[str] | None) -> Path: + root.mkdir(parents=True, exist_ok=True) + (root / "project.uiproj").write_text("{}", encoding="utf-8") + for name, body in files.items(): + (root / name).write_text(body, encoding="utf-8") + if entry is not None: + (root / "entry-points.json").write_text( + json.dumps({"entryPoints": [{"filePath": f"/content/{n}#Event_start"} for n in entry]}), + encoding="utf-8", + ) + return root + + +def _run(cwd: Path) -> tuple[int, str]: + proc = subprocess.run( + [sys.executable, str(GRADER)], cwd=cwd, capture_output=True, text=True + ) + return proc.returncode, proc.stdout + proc.stderr + + +def test_shape_in_the_entry_point_file_passes(tmp_path): + _project(tmp_path / "P", {"P.bpmn": SHAPE, "P_draft.bpmn": EMPTY}, ["P.bpmn"]) + rc, out = _run(tmp_path) + assert rc == 0, out + assert "P.bpmn -- create on NonExistentEntity" in out + + +def test_shape_in_a_non_entry_file_beside_an_empty_entry_fails(tmp_path): + """The run 36357685718 layout: the process that runs carries nothing.""" + _project(tmp_path / "P", {"P.bpmn": EMPTY, "P_error.bpmn": SHAPE}, ["P.bpmn"]) + rc, out = _run(tmp_path) + assert rc == 1, out + assert "P_error.bpmn, which is not an entry point" in out + + +def test_malformed_file_beside_a_valid_entry_point_is_skipped(tmp_path): + _project(tmp_path / "P", {"P.bpmn": SHAPE, "P_bad.bpmn": " Date: Mon, 28 Sep 2026 09:42:11 -0600 Subject: [PATCH 34/35] test(bpmn): let smoke_error find the entity inside the request body Smoke run 36443527602's curated query nodes carried the entity only as a body field ({"entityName": "FlowCodeEvalEntity", ...}); mentions_entity read flat inputs and the path but not the body. It now also matches a string leaf of body_object(), which covers both registry body forms. Unit test added; replay on that run passes and the non-entry-point layout still fails. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_df_smoke_error.py | 23 +++++++++++++--- .../_shared/test_check_df_smoke_error.py | 26 ++++++++++++++++--- 2 files changed, 42 insertions(+), 7 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index a0993ae2e2..8ddfad06c9 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -31,8 +31,9 @@ I parse each .bpmn -> ET.parse() T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() T entity as the generic objectName, an exact -> mentions_entity() - input value on any target, or an exact - context `path` segment + input value on any target, an exact + context `path` segment, or a string leaf + of the request body (either body form) DROPPED topology/parallel-branch parsing (Flow's own grader does not parse it either -- see its docstring) DROPPED require_no_private_connector_values (not in Flow) DROPPED require_sequence_integrity (not in Flow; `bpmn validate` criterion covers structure) @@ -61,6 +62,7 @@ from _shared.bpmn_check import ( # noqa: E402 _project_files, + body_object, entry_point_files, NS, context_inputs, @@ -86,10 +88,23 @@ def mentions_entity(task: ET.Element, entity: str) -> bool: return True if entity in context_value(task, "path").strip().split("/"): return True - return any( + if any( (inp.attrib.get("value") or inp.text or "").strip() == entity for inp in context_inputs(task) - ) + ): + return True + # The entity may be a field of the request body rather than a flat input + # (smoke run 36443527602: body {"entityName": "FlowCodeEvalEntity", ...}); + # body_object reads both registry body forms. + return entity in _string_leaves(body_object(task)) + + +def _string_leaves(value: object) -> list[str]: + if isinstance(value, dict): + return [leaf for v in value.values() for leaf in _string_leaves(v)] + if isinstance(value, list): + return [leaf for v in value for leaf in _string_leaves(v)] + return [value.strip()] if isinstance(value, str) else [] def is_generic_entity_object(object_name: str, entity: str) -> bool: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py index 3b26c0b35a..043413e717 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py @@ -18,15 +18,20 @@ ) -def _node(object_name: str, entity: str, method: str) -> str: +def _node(object_name: str, entity: str, method: str, body_entity: bool = False) -> str: + entity_input = ( + f'' + if body_entity + else f'' + ) return ( f'' '' '' f'' f'' - f'' - "" + + entity_input + + "" ) @@ -109,3 +114,18 @@ def test_a_bare_bpmn_with_no_project_anywhere_is_graded(tmp_path): (tmp_path / "P.bpmn").write_text(SHAPE, encoding="utf-8") rc, out = _run(tmp_path) assert rc == 0, out + + +def test_entity_named_only_in_the_body_json_counts(tmp_path): + """Smoke run 36443527602: curated query nodes carried the entity as a + body field, not a flat input.""" + shape = ( + f'' + + _node("CreateEntityRecordCurated", "NonExistentEntity", "POST", body_entity=True) + + _node("QueryEntityRecordsCurated", "FlowCodeEvalEntity", "POST", body_entity=True) + + _node("QueryEntityRecordsCurated", "FlowCodeEvalEntity", "POST", body_entity=True) + + "" + ) + (tmp_path / "P.bpmn").write_text(shape, encoding="utf-8") + rc, out = _run(tmp_path) + assert rc == 0, out From 992449fbba8a404053e25e9f7b4cb12c3a83e6d3 Mon Sep 17 00:00:00 2001 From: Monir Imamverdi Date: Mon, 28 Sep 2026 09:51:26 -0600 Subject: [PATCH 35/35] test(bpmn): match the smoke_error entity only under entity-naming body keys A Create on the real entity whose title happens to equal the missing entity's name is not the error-path Create; the body match now reads only entityName / entity keys at any depth instead of every string leaf. Negative test added. Co-Authored-By: Claude Fable 5.1 Claude-Session: https://claude.ai/code/session_0162cmh6J5y8h6rZtzwyT37S --- .../_shared/check_df_smoke_error.py | 30 +++++++++++++------ .../_shared/test_check_df_smoke_error.py | 26 ++++++++++++++++ 2 files changed, 47 insertions(+), 9 deletions(-) diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py index 8ddfad06c9..9fb8b58c74 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/check_df_smoke_error.py @@ -32,8 +32,8 @@ T curated|generic entity-CRUD classification -> is_create_node()/is_query_node() T entity as the generic objectName, an exact -> mentions_entity() input value on any target, an exact - context `path` segment, or a string leaf - of the request body (either body form) + context `path` segment, or the value of an + entity-naming body key (either body form) DROPPED topology/parallel-branch parsing (Flow's own grader does not parse it either -- see its docstring) DROPPED require_no_private_connector_values (not in Flow) DROPPED require_sequence_integrity (not in Flow; `bpmn validate` criterion covers structure) @@ -95,16 +95,28 @@ def mentions_entity(task: ET.Element, entity: str) -> bool: return True # The entity may be a field of the request body rather than a flat input # (smoke run 36443527602: body {"entityName": "FlowCodeEvalEntity", ...}); - # body_object reads both registry body forms. - return entity in _string_leaves(body_object(task)) + # body_object reads both registry body forms. Only entity-naming keys + # count: a record whose `title` happens to equal an entity name is not a + # node on that entity (review nit on this grader). + return entity in _entity_values(body_object(task)) -def _string_leaves(value: object) -> list[str]: +_ENTITY_KEYS = {"entityname", "entity"} + + +def _entity_values(value: object) -> list[str]: + """String values under an entity-naming key, at any depth of the body.""" + out: list[str] = [] if isinstance(value, dict): - return [leaf for v in value.values() for leaf in _string_leaves(v)] - if isinstance(value, list): - return [leaf for v in value for leaf in _string_leaves(v)] - return [value.strip()] if isinstance(value, str) else [] + for key, v in value.items(): + if str(key).lower() in _ENTITY_KEYS and isinstance(v, str): + out.append(v.strip()) + else: + out.extend(_entity_values(v)) + elif isinstance(value, list): + for v in value: + out.extend(_entity_values(v)) + return out def is_generic_entity_object(object_name: str, entity: str) -> bool: diff --git a/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py index 043413e717..490639ffdb 100644 --- a/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py +++ b/tests/tasks/uipath-maestro-bpmn/_shared/test_check_df_smoke_error.py @@ -129,3 +129,29 @@ def test_entity_named_only_in_the_body_json_counts(tmp_path): (tmp_path / "P.bpmn").write_text(shape, encoding="utf-8") rc, out = _run(tmp_path) assert rc == 0, out + + +def test_entity_name_in_a_non_entity_body_field_does_not_count(tmp_path): + """A Create on the real entity whose title merely equals the missing + entity's name is not the error-path Create (review nit).""" + decoy = ( + '' + '' + '' + '' + '' + '' + "" + ) + shape = ( + f'' + + decoy + + _node("QueryEntityRecordsCurated", "FlowCodeEvalEntity", "POST", body_entity=True) + + _node("QueryEntityRecordsCurated", "FlowCodeEvalEntity", "POST", body_entity=True) + + "" + ) + (tmp_path / "P.bpmn").write_text(shape, encoding="utf-8") + rc, out = _run(tmp_path) + assert rc == 1, out + assert "no Create Entity Record node targeting 'NonExistentEntity'" in out