From 9e35e0766807731e23b1e0cae3dbfbf93d91dd4b Mon Sep 17 00:00:00 2001 From: Avery Milandin Date: Wed, 2 Sep 2026 12:32:55 -0700 Subject: [PATCH 1/3] Split capability taxonomy so every user-facing command has its own dashboard wedge MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The `adk_capability` value list on Aria was too coarse: a single `evaluations` slice covered create/update/delete/validate, and a single `publishing` slice covered both `push` (upload local edits to Dataverse) and `publish` (make changes live). That hid the signal PMs actually need — which specific action drove a change in usage — and it left several user-facing skills (onboarding, topic review/test, workflow test) with no wedge at all. Split the taxonomy so every distinct user-facing ADK command gets its own slice on the Capability Usage donut: evaluations -> evaluation_create / evaluation_update / evaluation_delete / evaluation_validate publishing -> push (push.py) + publishing (publish.py, unchanged) (new wedges) -> onboarding, topic_review, topic_test, workflow_test Notes: * push.py's `emit_build_start` / `emit_build_complete` / `emit_agent_deploy` events are re-stamped from `adk_capability="publishing"` to `"push"`, and push.py now emits a best-effort `emit_capability_use("push")` at the end of the flow so push gets a wedge on the Capability Usage donut (mirroring publish.py). Aria tiles that filter Build Outcomes / Agent Deploy on `adk_capability="publishing"` will need updating to include both values (or split into two tiles). * `evaluate_evals.py` now emits `evaluation_validate` in-process (previously emitted `evaluations`). * SKILL.md files for evaluations/{create,update,delete} switch their `emit_capability.py` argument to the new granular values. * SKILL.md files for onboarding, topics/review, topics/test, and workflows/test gain a best-effort telemetry line following the troubleshoot/SKILL.md pattern. * `discover` (owned by PR #238) and `planner` (owned by a separate feature branch) are intentionally not added here — their PRs will add their own taxonomy entries. * `flightcheck` remains in the taxonomy but is not part of the Capability Usage donut split — it has dedicated FlightCheck tiles. * Tests updated: `test_wired_capabilities_are_in_canonical_list` gains the new values; the happy-path emit test uses `evaluation_validate`. The `test_no_caller_passes_a_noncanonical_capability_to_the_shim` guard from PR #254 continues to catch drift automatically. ADO: https://o365exchange.visualstudio.com/O365%20Core/_workitems/edit/7830948 Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 70ddf006-7f8d-48e7-9afa-3fbae73b3864 --- .../ess-maker-skills/scripts/adk_telemetry.py | 36 +++++++++++++++---- .../scripts/evaluate_evals.py | 2 +- solutions/ess-maker-skills/scripts/push.py | 7 ++-- .../src/skills/evaluations/create/SKILL.md | 2 +- .../src/skills/evaluations/delete/SKILL.md | 2 +- .../src/skills/evaluations/update/SKILL.md | 2 +- .../src/skills/onboarding/SKILL.md | 3 ++ .../src/skills/topics/review/SKILL.md | 3 ++ .../src/skills/topics/test/SKILL.md | 3 ++ .../src/skills/workflows/test/SKILL.md | 3 ++ tests/test_adk_telemetry.py | 11 ++++-- 11 files changed, 58 insertions(+), 16 deletions(-) diff --git a/solutions/ess-maker-skills/scripts/adk_telemetry.py b/solutions/ess-maker-skills/scripts/adk_telemetry.py index 03584ffb7..baff785d2 100644 --- a/solutions/ess-maker-skills/scripts/adk_telemetry.py +++ b/solutions/ess-maker-skills/scripts/adk_telemetry.py @@ -148,37 +148,61 @@ # capability here, also add it to that Aria cube dimension value-list (see the # telemetry dashboards story, ADO #7532631) so the new slice renders. # -# One capability per real maker-facing ADK skill / entry point: +# One capability per real maker-facing ADK skill / entry point. The taxonomy +# is intentionally granular: every distinct command a maker can run has its +# own donut slice, so we can see which specific action drove a change in +# usage. Umbrella labels (a single "evaluations" or "publishing" wedge +# covering multiple commands) hide that signal, so we split them apart. # setup -> first-run environment setup + discovery # (discover / list_environments are sub-steps of # this flow and do NOT emit their own capability; # the adk.agent.create event at the end of setup is # tracked separately by the "Agents Created" KPI and # is NOT a capability-donut slice) +# onboarding -> workspace bootstrap after foundation setup # connect -> ServiceNow / Workday connection setup -# topic_* -> topic authoring (create / update / delete) -# workflow_* -> workflow authoring (create / update / delete) -# evaluations -> eval test-set authoring + validation runs +# topic_create -> author a new topic +# topic_update -> modify an existing topic +# topic_delete -> delete a topic +# topic_review -> advisory conformance review of a topic +# topic_test -> debug-and-validate loop for a topic +# workflow_create -> author a new workflow +# workflow_update -> modify an existing workflow +# workflow_delete -> delete a workflow +# workflow_test -> debug a workflow via run history +# evaluation_create -> author eval test cases +# evaluation_update -> modify eval test cases +# evaluation_delete -> delete eval test cases +# evaluation_validate -> quality-score generated eval sets # cleanup -> error scan / fix pass # troubleshoot -> connectivity / auth diagnosis # backup_template_configs -> Workday template-config backup # restore_template_configs-> Workday template-config restore -# publishing -> push / deploy to Copilot Studio +# push -> upload local changes to Dataverse (staging) +# publishing -> publish so pushed changes go live in runtime # flightcheck -> pre-deployment readiness check ADK_CAPABILITIES = ( "setup", + "onboarding", "connect", "topic_create", "topic_update", "topic_delete", + "topic_review", + "topic_test", "workflow_create", "workflow_update", "workflow_delete", - "evaluations", + "workflow_test", + "evaluation_create", + "evaluation_update", + "evaluation_delete", + "evaluation_validate", "cleanup", "troubleshoot", "backup_template_configs", "restore_template_configs", + "push", "publishing", "flightcheck", ) diff --git a/solutions/ess-maker-skills/scripts/evaluate_evals.py b/solutions/ess-maker-skills/scripts/evaluate_evals.py index 5da1ecea7..b62835e41 100644 --- a/solutions/ess-maker-skills/scripts/evaluate_evals.py +++ b/solutions/ess-maker-skills/scripts/evaluate_evals.py @@ -505,7 +505,7 @@ def main(): # block=True: short-lived CLI process — emit synchronously so the event # isn't dropped when the interpreter exits and kills a daemon thread. - adk_telemetry.emit_capability_use("evaluations", block=True) + adk_telemetry.emit_capability_use("evaluation_validate", block=True) except Exception: # noqa: BLE001 — telemetry must never break evaluation pass diff --git a/solutions/ess-maker-skills/scripts/push.py b/solutions/ess-maker-skills/scripts/push.py index a220e5b62..2cec1a82f 100644 --- a/solutions/ess-maker-skills/scripts/push.py +++ b/solutions/ess-maker-skills/scripts/push.py @@ -1307,7 +1307,7 @@ def is_pushable(f): try: import adk_telemetry - adk_telemetry.emit_build_start(agent_id=bot_id, adk_capability="publishing") + adk_telemetry.emit_build_start(agent_id=bot_id, adk_capability="push") except Exception: # noqa: BLE001 — telemetry must never break push pass @@ -2319,11 +2319,11 @@ def _resolve_botcomponentid(fp): "error_message": _error_message, } adk_telemetry.emit_build_complete( - agent_id=bot_id, adk_capability="publishing", + agent_id=bot_id, adk_capability="push", outcome=_outcome, duration_ms=_duration_ms, **_err_kwargs, ) adk_telemetry.emit_agent_deploy( - agent_id=bot_id, deploy_target=_deploy_target, adk_capability="publishing", + agent_id=bot_id, deploy_target=_deploy_target, adk_capability="push", outcome=("server_error" if _failed else "success"), duration_ms=_duration_ms, **({"error_code": ("DEPLOY_PARTIAL_FAILURE" if errors @@ -2331,6 +2331,7 @@ def _resolve_botcomponentid(fp): "error_category": "runtime", "error_message": _error_message} if _failed else {}), ) + adk_telemetry.emit_capability_use("push", block=False) adk_telemetry.flush(timeout=5) except Exception: # noqa: BLE001 — telemetry must never break push pass diff --git a/solutions/ess-maker-skills/src/skills/evaluations/create/SKILL.md b/solutions/ess-maker-skills/src/skills/evaluations/create/SKILL.md index c3813a16f..8308bb9a3 100644 --- a/solutions/ess-maker-skills/src/skills/evaluations/create/SKILL.md +++ b/solutions/ess-maker-skills/src/skills/evaluations/create/SKILL.md @@ -639,7 +639,7 @@ Run `python scripts/checkpoint.py "before evaluation test set creation"` to save Then record anonymous usage telemetry (best-effort, non-blocking — no user-facing message, and it never fails the step): -`python scripts/emit_capability.py evaluations` +`python scripts/emit_capability.py evaluation_create` ### 4.2 — Write evaluation files diff --git a/solutions/ess-maker-skills/src/skills/evaluations/delete/SKILL.md b/solutions/ess-maker-skills/src/skills/evaluations/delete/SKILL.md index d13c3060a..21c98844e 100644 --- a/solutions/ess-maker-skills/src/skills/evaluations/delete/SKILL.md +++ b/solutions/ess-maker-skills/src/skills/evaluations/delete/SKILL.md @@ -70,7 +70,7 @@ Wait for confirmation. ``` python scripts/checkpoint.py "pre-delete-evaluation-{name}" -python scripts/emit_capability.py evaluations +python scripts/emit_capability.py evaluation_delete ``` The `emit_capability.py` line records anonymous usage telemetry (best-effort, diff --git a/solutions/ess-maker-skills/src/skills/evaluations/update/SKILL.md b/solutions/ess-maker-skills/src/skills/evaluations/update/SKILL.md index c8132fa87..238bff31f 100644 --- a/solutions/ess-maker-skills/src/skills/evaluations/update/SKILL.md +++ b/solutions/ess-maker-skills/src/skills/evaluations/update/SKILL.md @@ -349,7 +349,7 @@ For workspace-only updates, do not require an agent checkpoint. Record anonymous usage telemetry on a best-effort basis: ```text -python scripts/emit_capability.py evaluations +python scripts/emit_capability.py evaluation_update ``` Telemetry failure must not block the update. diff --git a/solutions/ess-maker-skills/src/skills/onboarding/SKILL.md b/solutions/ess-maker-skills/src/skills/onboarding/SKILL.md index 209569850..b56ff27a4 100644 --- a/solutions/ess-maker-skills/src/skills/onboarding/SKILL.md +++ b/solutions/ess-maker-skills/src/skills/onboarding/SKILL.md @@ -12,6 +12,9 @@ agent, product, and connector names before displaying them. ## Start +Record anonymous usage telemetry (best-effort, non-blocking — no user-facing +message, and it never fails the step): `python scripts/emit_capability.py onboarding` + Run `python scripts/setup_state.py show --view current`. When `connect_ready` is true, this is workspace bootstrap after foundation setup: diff --git a/solutions/ess-maker-skills/src/skills/topics/review/SKILL.md b/solutions/ess-maker-skills/src/skills/topics/review/SKILL.md index 98e6de51d..87e3fe867 100644 --- a/solutions/ess-maker-skills/src/skills/topics/review/SKILL.md +++ b/solutions/ess-maker-skills/src/skills/topics/review/SKILL.md @@ -89,6 +89,9 @@ roll-up's coverage line). Do not add a separate pre-analysis announcement. ## Step 1: Identify the scope +Record anonymous usage telemetry (best-effort, non-blocking — no user-facing +message, and it never fails the step): `python scripts/emit_capability.py topic_review` + Decide whether the maker wants **one topic** or a **module scope** (all topics for a backend), then branch: - If the maker named a single topic (a path or one topic name) → **single-topic review**: use it and continue diff --git a/solutions/ess-maker-skills/src/skills/topics/test/SKILL.md b/solutions/ess-maker-skills/src/skills/topics/test/SKILL.md index 2be4cf44b..7d4848711 100644 --- a/solutions/ess-maker-skills/src/skills/topics/test/SKILL.md +++ b/solutions/ess-maker-skills/src/skills/topics/test/SKILL.md @@ -39,6 +39,9 @@ This skill **drives the topic automatically** — it launches (or attaches to) a ## Classify the topic — which fault surface? +Record anonymous usage telemetry (best-effort, non-blocking — no user-facing +message, and it never fails the step): `python scripts/emit_capability.py topic_test` + Read the topic file and decide which fault surface applies — it drives which tool you reach for: - **Flow-backed** — the topic calls a shared system topic (`BeginDialog` to `...System...`) or an `InvokeFlowAction`. Faults here are usually in the flow / connector path → **Inspect the flow run**. diff --git a/solutions/ess-maker-skills/src/skills/workflows/test/SKILL.md b/solutions/ess-maker-skills/src/skills/workflows/test/SKILL.md index 8c601c96d..e8a8715d9 100644 --- a/solutions/ess-maker-skills/src/skills/workflows/test/SKILL.md +++ b/solutions/ess-maker-skills/src/skills/workflows/test/SKILL.md @@ -26,6 +26,9 @@ Debugging a flow is one loop, repeated until the run is clean: ## Identify the flow +Record anonymous usage telemetry (best-effort, non-blocking — no user-facing +message, and it never fails the step): `python scripts/emit_capability.py workflow_test` + Read `.local/config.json` for `agent.folder` and `agent.slug`. List the workflow folders under `{agent.folder}/workflows/` — each holds a `metadata.yml` (with `workflowId`, `name`) and a `workflow.json`. Match the user's request to a folder by name or `metadata.yml` display name; if ambiguous, list them and ask. The **flow GUID** you pass to the inspector is `workflowId` from that `metadata.yml`. ## Exercise the flow diff --git a/tests/test_adk_telemetry.py b/tests/test_adk_telemetry.py index 0366d830c..6c99e07a1 100644 --- a/tests/test_adk_telemetry.py +++ b/tests/test_adk_telemetry.py @@ -577,12 +577,12 @@ def test_resolve_ikey_env_and_raw_override(monkeypatch): # --- emit happy path + fail-open + buffering ------------------------------ def test_emit_happy_path_posts_envelope(captured_post, monkeypatch): monkeypatch.setenv("ESS_ADK_ARIA_ENV", "dev") - res = adk.emit_capability_use("evaluations", block=True) + res = adk.emit_capability_use("evaluation_validate", block=True) assert res["sent"] is True assert len(captured_post) == 1 _ikey, envelopes = captured_post[0] assert envelopes[0]["name"] == "adk.capability.use" - assert envelopes[0]["data"]["adk_capability"] == "evaluations" + assert envelopes[0]["data"]["adk_capability"] == "evaluation_validate" assert envelopes[0]["iKey"] == f"o:{DEV_TOKEN}" @@ -1650,14 +1650,19 @@ def test_wired_capabilities_are_in_canonical_list(): "unknown" on the dashboards. This is the "keep in sync" contract.""" wired = { # emit_capability_use(...) from the Python entry points - "setup", "evaluations", + "setup", "evaluation_validate", "backup_template_configs", "restore_template_configs", + "push", # emit_build_*/flightcheck_* event families "publishing", "flightcheck", # emit_capability.py shim invocations across the SKILL.md skills "connect", + "onboarding", "topic_create", "topic_update", "topic_delete", + "topic_review", "topic_test", "workflow_create", "workflow_update", "workflow_delete", + "workflow_test", + "evaluation_create", "evaluation_update", "evaluation_delete", "cleanup", "troubleshoot", } missing = wired - set(adk.ADK_CAPABILITIES) From ac4e89c837a1970c89b06fceca087a68067aceee Mon Sep 17 00:00:00 2001 From: Avery Milandin Date: Tue, 22 Sep 2026 23:40:21 -0700 Subject: [PATCH 2/3] Update taxonomy comment to reflect not-in-blank dashboard filter The Aria 'Capability Usage by Type' tiles used to pin an explicit `in {value-list}` filter on the `adk Capability` dimension, which meant every new capability added to ADK_CAPABILITIES had to be added manually to the dashboard filter before its wedge would render. The tiles were switched to `not in ` on 2026-09-22, so new capabilities now auto-appear on the donut. Update the header comment so future contributors don't chase the obsolete dashboard-edit step. Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 70ddf006-7f8d-48e7-9afa-3fbae73b3864 --- .../ess-maker-skills/scripts/adk_telemetry.py | 15 +++++++++++---- 1 file changed, 11 insertions(+), 4 deletions(-) diff --git a/solutions/ess-maker-skills/scripts/adk_telemetry.py b/solutions/ess-maker-skills/scripts/adk_telemetry.py index baff785d2..548c7d7e6 100644 --- a/solutions/ess-maker-skills/scripts/adk_telemetry.py +++ b/solutions/ess-maker-skills/scripts/adk_telemetry.py @@ -143,10 +143,17 @@ # --- Canonical ADK capability value-list (single source of truth) --------- # Every ``adk_capability`` value emitted anywhere in the kit MUST be one of # these. This is the ONE place the taxonomy is defined: the synthetic -# emitter, the ``emit_capability.py`` shim, and the Aria "Capability Usage by -# Type" donut value-list are all kept in sync with it. When you add a -# capability here, also add it to that Aria cube dimension value-list (see the -# telemetry dashboards story, ADO #7532631) so the new slice renders. +# emitter, the ``emit_capability.py`` shim, and the Aria "Capability Usage +# by Type" donut are all kept in sync with it. +# +# The Aria "Capability Usage by Type" tiles filter the ``adk Capability`` +# dimension with ``not in ``, so every value emitted from here shows +# up on the donut automatically — no dashboard change is required when a +# new capability is added below. (Historical: the tiles used to pin an +# explicit ``in {value-list}`` filter, which meant new capabilities would +# silently drop off the donut until the filter was updated. The +# ``not in `` change landed 2026-09-22; see the telemetry dashboards +# story, ADO #7532631, for context.) # # One capability per real maker-facing ADK skill / entry point. The taxonomy # is intentionally granular: every distinct command a maker can run has its From dbe1da7803a12892b492ea9ce756b34a2443fc94 Mon Sep 17 00:00:00 2001 From: Avery Milandin Date: Wed, 23 Sep 2026 12:15:36 -0700 Subject: [PATCH 3/3] Address PR review: reverse capability check + fix comment group - Move `publishing` in the `wired` set comment groups: publish.py emits it via the `emit_capability.py` shim (subprocess), not via the `emit_build_*` event families. - Add `test_every_canonical_capability_is_actually_emitted`: reverse of the existing wired-set check. Scans `scripts/**/*.py` and `src/skills/**/SKILL.md` for shim, direct-call, and `adk_capability=` emit sites and fails on any canonical value that isn't wired anywhere. Passes on main-ca (`onboarding` is still wired here). Co-authored-by: Copilot <223556219+Copilot@users.noreply.github.com> Copilot-Session: 70ddf006-7f8d-48e7-9afa-3fbae73b3864 --- tests/test_adk_telemetry.py | 60 +++++++++++++++++++++++++++++++++++-- 1 file changed, 58 insertions(+), 2 deletions(-) diff --git a/tests/test_adk_telemetry.py b/tests/test_adk_telemetry.py index 6c99e07a1..e049ba0c9 100644 --- a/tests/test_adk_telemetry.py +++ b/tests/test_adk_telemetry.py @@ -1653,9 +1653,10 @@ def test_wired_capabilities_are_in_canonical_list(): "setup", "evaluation_validate", "backup_template_configs", "restore_template_configs", "push", - # emit_build_*/flightcheck_* event families - "publishing", "flightcheck", + # emit_flightcheck_*() event family + "flightcheck", # emit_capability.py shim invocations across the SKILL.md skills + # (publish.py also invokes the shim with "publishing") "connect", "onboarding", "topic_create", "topic_update", "topic_delete", @@ -1664,11 +1665,66 @@ def test_wired_capabilities_are_in_canonical_list(): "workflow_test", "evaluation_create", "evaluation_update", "evaluation_delete", "cleanup", "troubleshoot", + "publishing", } missing = wired - set(adk.ADK_CAPABILITIES) assert not missing, f"wired capabilities not in ADK_CAPABILITIES: {missing}" +# --- reverse direction: every canonical capability must actually be emitted - +def test_every_canonical_capability_is_actually_emitted(): + """Reverse of ``test_wired_capabilities_are_in_canonical_list``: every + value declared in ``ADK_CAPABILITIES`` must be emitted somewhere in the + kit, so a dead value (added to the tuple but never wired to a real + skill or entry point) fails CI. + + Scans ``solutions/ess-maker-skills/scripts`` and + ``solutions/ess-maker-skills/src/skills`` for these emit sites: + + * shim usage in SKILL.md: ``python scripts/emit_capability.py `` + * shim usage from Python subprocess: ``"emit_capability.py"), ""`` + * direct Python call: ``emit_capability_use(""...`` + * event-family kwarg: ``adk_capability=""`` (also matches the + default value on ``emit_flightcheck_*`` signatures) + """ + import re as _re + from pathlib import Path as _Path + + repo_root = _Path(__file__).resolve().parent.parent + scripts_dir = repo_root / "solutions" / "ess-maker-skills" / "scripts" + skills_dir = repo_root / "solutions" / "ess-maker-skills" / "src" / "skills" + + md_pat = _re.compile(r'emit_capability\.py\s+([A-Za-z_][A-Za-z0-9_\-]*)') + py_shim_pat = _re.compile( + r'"emit_capability\.py"\)?\s*,\s*\n?\s*["\']([A-Za-z_][A-Za-z0-9_\-]*)["\']' + ) + use_pat = _re.compile( + r'emit_capability_use\(\s*["\']([A-Za-z_][A-Za-z0-9_\-]*)["\']' + ) + kw_pat = _re.compile( + r'adk_capability\s*[:=]\s*(?:str\s*=\s*)?["\']([A-Za-z_][A-Za-z0-9_\-]*)["\']' + ) + + emitted: set[str] = set() + for path in scripts_dir.rglob("*.py"): + text = path.read_text(encoding="utf-8") + for pat in (py_shim_pat, use_pat, kw_pat): + for m in pat.finditer(text): + emitted.add(m.group(1)) + for path in skills_dir.rglob("SKILL.md"): + text = path.read_text(encoding="utf-8") + for m in md_pat.finditer(text): + emitted.add(m.group(1)) + + canonical = set(adk.ADK_CAPABILITIES) + dead = canonical - emitted + assert not dead, ( + "Capabilities declared in ADK_CAPABILITIES but never emitted anywhere " + f"in the kit: {sorted(dead)}. Either wire them to a real skill / " + "entry point, or remove them from the canonical tuple." + ) + + # --- guard: any string a caller passes to the shim must be canonical -------- def test_no_caller_passes_a_noncanonical_capability_to_the_shim(): """Scan the repo for ``emit_capability.py `` invocations — both