Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
51 changes: 41 additions & 10 deletions solutions/ess-maker-skills/scripts/adk_telemetry.py
Original file line number Diff line number Diff line change
Expand Up @@ -143,42 +143,73 @@
# --- Canonical ADK capability value-list (single source of truth) ---------
# Every ``adk_capability`` value emitted anywhere in the kit MUST be one of
# these. This is the ONE place the taxonomy is defined: the synthetic
# emitter, the ``emit_capability.py`` shim, and the Aria "Capability Usage by
# Type" donut value-list are all kept in sync with it. When you add a
# capability here, also add it to that Aria cube dimension value-list (see the
# telemetry dashboards story, ADO #7532631) so the new slice renders.
# emitter, the ``emit_capability.py`` shim, and the Aria "Capability Usage
# by Type" donut are all kept in sync with it.
#
# One capability per real maker-facing ADK skill / entry point:
# The Aria "Capability Usage by Type" tiles filter the ``adk Capability``
# dimension with ``not in <blank>``, so every value emitted from here shows
# up on the donut automatically — no dashboard change is required when a
# new capability is added below. (Historical: the tiles used to pin an
# explicit ``in {value-list}`` filter, which meant new capabilities would
# silently drop off the donut until the filter was updated. The
# ``not in <blank>`` change landed 2026-09-22; see the telemetry dashboards
# story, ADO #7532631, for context.)
#
# One capability per real maker-facing ADK skill / entry point. The taxonomy
# is intentionally granular: every distinct command a maker can run has its
# own donut slice, so we can see which specific action drove a change in
# usage. Umbrella labels (a single "evaluations" or "publishing" wedge
# covering multiple commands) hide that signal, so we split them apart.
# setup -> first-run environment setup + discovery
# (discover / list_environments are sub-steps of
# this flow and do NOT emit their own capability;
# the adk.agent.create event at the end of setup is
# tracked separately by the "Agents Created" KPI and
# is NOT a capability-donut slice)
# onboarding -> workspace bootstrap after foundation setup
# connect -> ServiceNow / Workday connection setup
# topic_* -> topic authoring (create / update / delete)
# workflow_* -> workflow authoring (create / update / delete)
# evaluations -> eval test-set authoring + validation runs
# topic_create -> author a new topic
# topic_update -> modify an existing topic
# topic_delete -> delete a topic
# topic_review -> advisory conformance review of a topic
# topic_test -> debug-and-validate loop for a topic
# workflow_create -> author a new workflow
# workflow_update -> modify an existing workflow
# workflow_delete -> delete a workflow
# workflow_test -> debug a workflow via run history
# evaluation_create -> author eval test cases
# evaluation_update -> modify eval test cases
# evaluation_delete -> delete eval test cases
# evaluation_validate -> quality-score generated eval sets
# cleanup -> error scan / fix pass
# troubleshoot -> connectivity / auth diagnosis
# backup_template_configs -> Workday template-config backup
# restore_template_configs-> Workday template-config restore
# publishing -> push / deploy to Copilot Studio
# push -> upload local changes to Dataverse (staging)
# publishing -> publish so pushed changes go live in runtime
# flightcheck -> pre-deployment readiness check
ADK_CAPABILITIES = (
"setup",
"onboarding",
"connect",
"topic_create",
"topic_update",
"topic_delete",
"topic_review",
"topic_test",
"workflow_create",
"workflow_update",
"workflow_delete",
"evaluations",
"workflow_test",
"evaluation_create",
"evaluation_update",
"evaluation_delete",
"evaluation_validate",
"cleanup",
"troubleshoot",
"backup_template_configs",
"restore_template_configs",
"push",
"publishing",
"flightcheck",
)
Expand Down
2 changes: 1 addition & 1 deletion solutions/ess-maker-skills/scripts/evaluate_evals.py
Original file line number Diff line number Diff line change
Expand Up @@ -505,7 +505,7 @@ def main():

# block=True: short-lived CLI process — emit synchronously so the event
# isn't dropped when the interpreter exits and kills a daemon thread.
adk_telemetry.emit_capability_use("evaluations", block=True)
adk_telemetry.emit_capability_use("evaluation_validate", block=True)
except Exception: # noqa: BLE001 — telemetry must never break evaluation
pass

Expand Down
7 changes: 4 additions & 3 deletions solutions/ess-maker-skills/scripts/push.py
Original file line number Diff line number Diff line change
Expand Up @@ -1307,7 +1307,7 @@ def is_pushable(f):
try:
import adk_telemetry

adk_telemetry.emit_build_start(agent_id=bot_id, adk_capability="publishing")
adk_telemetry.emit_build_start(agent_id=bot_id, adk_capability="push")
except Exception: # noqa: BLE001 — telemetry must never break push
pass

Expand Down Expand Up @@ -2319,18 +2319,19 @@ def _resolve_botcomponentid(fp):
"error_message": _error_message,
}
adk_telemetry.emit_build_complete(
agent_id=bot_id, adk_capability="publishing",
agent_id=bot_id, adk_capability="push",
outcome=_outcome, duration_ms=_duration_ms, **_err_kwargs,
)
adk_telemetry.emit_agent_deploy(
agent_id=bot_id, deploy_target=_deploy_target, adk_capability="publishing",
agent_id=bot_id, deploy_target=_deploy_target, adk_capability="push",
outcome=("server_error" if _failed else "success"),
duration_ms=_duration_ms,
**({"error_code": ("DEPLOY_PARTIAL_FAILURE" if errors
else "REGISTRATION_INCOMPLETE"),
"error_category": "runtime",
"error_message": _error_message} if _failed else {}),
)
adk_telemetry.emit_capability_use("push", block=False)
adk_telemetry.flush(timeout=5)
except Exception: # noqa: BLE001 — telemetry must never break push
pass
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -639,7 +639,7 @@ Run `python scripts/checkpoint.py "before evaluation test set creation"` to save

Then record anonymous usage telemetry (best-effort, non-blocking — no
user-facing message, and it never fails the step):
`python scripts/emit_capability.py evaluations`
`python scripts/emit_capability.py evaluation_create`

### 4.2 — Write evaluation files

Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -70,7 +70,7 @@ Wait for confirmation.

```
python scripts/checkpoint.py "pre-delete-evaluation-{name}"
python scripts/emit_capability.py evaluations
python scripts/emit_capability.py evaluation_delete
```

The `emit_capability.py` line records anonymous usage telemetry (best-effort,
Expand Down
Original file line number Diff line number Diff line change
Expand Up @@ -349,7 +349,7 @@ For workspace-only updates, do not require an agent checkpoint.
Record anonymous usage telemetry on a best-effort basis:

```text
python scripts/emit_capability.py evaluations
python scripts/emit_capability.py evaluation_update
```

Telemetry failure must not block the update.
Expand Down
3 changes: 3 additions & 0 deletions solutions/ess-maker-skills/src/skills/onboarding/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -12,6 +12,9 @@ agent, product, and connector names before displaying them.

## Start

Record anonymous usage telemetry (best-effort, non-blocking — no user-facing
message, and it never fails the step): `python scripts/emit_capability.py onboarding`

Run `python scripts/setup_state.py show --view current`. When `connect_ready` is
true, this is workspace bootstrap after foundation setup:

Expand Down
3 changes: 3 additions & 0 deletions solutions/ess-maker-skills/src/skills/topics/review/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -89,6 +89,9 @@ roll-up's coverage line). Do not add a separate pre-analysis announcement.

## Step 1: Identify the scope

Record anonymous usage telemetry (best-effort, non-blocking — no user-facing
message, and it never fails the step): `python scripts/emit_capability.py topic_review`

Decide whether the maker wants **one topic** or a **module scope** (all topics for a backend), then branch:

- If the maker named a single topic (a path or one topic name) → **single-topic review**: use it and continue
Expand Down
3 changes: 3 additions & 0 deletions solutions/ess-maker-skills/src/skills/topics/test/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -39,6 +39,9 @@ This skill **drives the topic automatically** — it launches (or attaches to) a

## Classify the topic — which fault surface?

Record anonymous usage telemetry (best-effort, non-blocking — no user-facing
message, and it never fails the step): `python scripts/emit_capability.py topic_test`

Read the topic file and decide which fault surface applies — it drives which tool you reach for:

- **Flow-backed** — the topic calls a shared system topic (`BeginDialog` to `...System...`) or an `InvokeFlowAction`. Faults here are usually in the flow / connector path → **Inspect the flow run**.
Expand Down
3 changes: 3 additions & 0 deletions solutions/ess-maker-skills/src/skills/workflows/test/SKILL.md
Original file line number Diff line number Diff line change
Expand Up @@ -26,6 +26,9 @@ Debugging a flow is one loop, repeated until the run is clean:

## Identify the flow

Record anonymous usage telemetry (best-effort, non-blocking — no user-facing
message, and it never fails the step): `python scripts/emit_capability.py workflow_test`

Read `.local/config.json` for `agent.folder` and `agent.slug`. List the workflow folders under `{agent.folder}/workflows/` — each holds a `metadata.yml` (with `workflowId`, `name`) and a `workflow.json`. Match the user's request to a folder by name or `metadata.yml` display name; if ambiguous, list them and ask. The **flow GUID** you pass to the inspector is `workflowId` from that `metadata.yml`.

## Exercise the flow
Expand Down
71 changes: 66 additions & 5 deletions tests/test_adk_telemetry.py
Original file line number Diff line number Diff line change
Expand Up @@ -577,12 +577,12 @@ def test_resolve_ikey_env_and_raw_override(monkeypatch):
# --- emit happy path + fail-open + buffering ------------------------------
def test_emit_happy_path_posts_envelope(captured_post, monkeypatch):
monkeypatch.setenv("ESS_ADK_ARIA_ENV", "dev")
res = adk.emit_capability_use("evaluations", block=True)
res = adk.emit_capability_use("evaluation_validate", block=True)
assert res["sent"] is True
assert len(captured_post) == 1
_ikey, envelopes = captured_post[0]
assert envelopes[0]["name"] == "adk.capability.use"
assert envelopes[0]["data"]["adk_capability"] == "evaluations"
assert envelopes[0]["data"]["adk_capability"] == "evaluation_validate"
assert envelopes[0]["iKey"] == f"o:{DEV_TOKEN}"


Expand Down Expand Up @@ -1650,20 +1650,81 @@ def test_wired_capabilities_are_in_canonical_list():
"unknown" on the dashboards. This is the "keep in sync" contract."""
wired = {
# emit_capability_use(...) from the Python entry points
"setup", "evaluations",
"setup", "evaluation_validate",
"backup_template_configs", "restore_template_configs",
# emit_build_*/flightcheck_* event families
"publishing", "flightcheck",
"push",
# emit_flightcheck_*() event family
"flightcheck",
# emit_capability.py shim invocations across the SKILL.md skills
# (publish.py also invokes the shim with "publishing")
"connect",
"onboarding",
"topic_create", "topic_update", "topic_delete",
"topic_review", "topic_test",
"workflow_create", "workflow_update", "workflow_delete",
"workflow_test",
"evaluation_create", "evaluation_update", "evaluation_delete",
"cleanup", "troubleshoot",
"publishing",
}
missing = wired - set(adk.ADK_CAPABILITIES)
assert not missing, f"wired capabilities not in ADK_CAPABILITIES: {missing}"


# --- reverse direction: every canonical capability must actually be emitted -
def test_every_canonical_capability_is_actually_emitted():
"""Reverse of ``test_wired_capabilities_are_in_canonical_list``: every
value declared in ``ADK_CAPABILITIES`` must be emitted somewhere in the
kit, so a dead value (added to the tuple but never wired to a real
skill or entry point) fails CI.

Scans ``solutions/ess-maker-skills/scripts`` and
``solutions/ess-maker-skills/src/skills`` for these emit sites:

* shim usage in SKILL.md: ``python scripts/emit_capability.py <cap>``
* shim usage from Python subprocess: ``"emit_capability.py"), "<cap>"``
* direct Python call: ``emit_capability_use("<cap>"...``
* event-family kwarg: ``adk_capability="<cap>"`` (also matches the
default value on ``emit_flightcheck_*`` signatures)
"""
import re as _re
from pathlib import Path as _Path

repo_root = _Path(__file__).resolve().parent.parent
scripts_dir = repo_root / "solutions" / "ess-maker-skills" / "scripts"
skills_dir = repo_root / "solutions" / "ess-maker-skills" / "src" / "skills"

md_pat = _re.compile(r'emit_capability\.py\s+([A-Za-z_][A-Za-z0-9_\-]*)')
py_shim_pat = _re.compile(
r'"emit_capability\.py"\)?\s*,\s*\n?\s*["\']([A-Za-z_][A-Za-z0-9_\-]*)["\']'
)
use_pat = _re.compile(
r'emit_capability_use\(\s*["\']([A-Za-z_][A-Za-z0-9_\-]*)["\']'
)
kw_pat = _re.compile(
r'adk_capability\s*[:=]\s*(?:str\s*=\s*)?["\']([A-Za-z_][A-Za-z0-9_\-]*)["\']'
)

emitted: set[str] = set()
for path in scripts_dir.rglob("*.py"):
text = path.read_text(encoding="utf-8")
for pat in (py_shim_pat, use_pat, kw_pat):
for m in pat.finditer(text):
emitted.add(m.group(1))
for path in skills_dir.rglob("SKILL.md"):
text = path.read_text(encoding="utf-8")
for m in md_pat.finditer(text):
emitted.add(m.group(1))

canonical = set(adk.ADK_CAPABILITIES)
dead = canonical - emitted
assert not dead, (
"Capabilities declared in ADK_CAPABILITIES but never emitted anywhere "
f"in the kit: {sorted(dead)}. Either wire them to a real skill / "
"entry point, or remove them from the canonical tuple."
)


# --- guard: any string a caller passes to the shim must be canonical --------
def test_no_caller_passes_a_noncanonical_capability_to_the_shim():
"""Scan the repo for ``emit_capability.py <cap>`` invocations — both
Expand Down
Loading