From dbd4de1a89058ccca8aa49d3eff1ab5afe6c9d1c Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Sat, 22 Aug 2026 05:17:41 +0000 Subject: [PATCH 1/2] Initial plan From 51251e1d696cef110b8cc0903db6f8a715b3ea25 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Sat, 22 Aug 2026 05:30:55 +0000 Subject: [PATCH 2/2] Fix skipped workflow metrics classification Co-authored-by: pelikhan <4175913+pelikhan@users.noreply.github.com> --- .../agent-performance-analyzer.lock.yml | 4 +- .../workflows/agent-performance-analyzer.md | 7 +++ .github/workflows/metrics-collector.lock.yml | 2 +- .github/workflows/metrics-collector.md | 62 +++++++++++++++---- .../workflow-health-manager.lock.yml | 4 +- .github/workflows/workflow-health-manager.md | 18 +++--- 6 files changed, 74 insertions(+), 23 deletions(-) diff --git a/.github/workflows/agent-performance-analyzer.lock.yml b/.github/workflows/agent-performance-analyzer.lock.yml index 81b9657ff71..f2b62afb93c 100644 --- a/.github/workflows/agent-performance-analyzer.lock.yml +++ b/.github/workflows/agent-performance-analyzer.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"87110e7c96fd24eee920a67c3a72c1bf1450e62e788f948848ea5937daac158d","body_hash":"4495b7b8ac66500dd3724fef344313a8290166c31a12a66a75fccfd436e4b5f1","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"87110e7c96fd24eee920a67c3a72c1bf1450e62e788f948848ea5937daac158d","body_hash":"2e1df0801582987fdd3e068fd57e000ffac459e9747375b768dcbefe8e5f39fe","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} # gh-aw-manifest: {"version":1,"secrets":["GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-go","sha":"b7ad1dad31e06c5925ef5d2fc7ad053ef454303e","version":"v7.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"docker/build-push-action","sha":"53b7df96c91f9c12dcc8a07bcb9ccacbed38856a","version":"v7.3.0"},{"repo":"docker/setup-buildx-action","sha":"37fe631027851001ddb9b187196cc803df7f5f0e","version":"v4.3.0"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.4","digest":"sha256:8f18587981eff7e6291784200a88a7191d23bfd8f5723db848c640d6b4e88e46","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.4@sha256:8f18587981eff7e6291784200a88a7191d23bfd8f5723db848c640d6b4e88e46"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.4","digest":"sha256:64e668297d1b9d83ee102626e104c054ac2cf5cfd7225bf6e144f86962229882","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.4@sha256:64e668297d1b9d83ee102626e104c054ac2cf5cfd7225bf6e144f86962229882"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.4","digest":"sha256:7ad9113203642c5b12303b97e221fc6b5dd0fb0655c09a2eba9d5c6fa3201a3b","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.4@sha256:7ad9113203642c5b12303b97e221fc6b5dd0fb0655c09a2eba9d5c6fa3201a3b"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.4","digest":"sha256:35953d0beac18f642aa0bc98bb726a85288539b1fc630c16e94da33300058258","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.4@sha256:35953d0beac18f642aa0bc98bb726a85288539b1fc630c16e94da33300058258"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.10","digest":"sha256:08bb5fa417aed94b40a14e2b7b3ae457531a5f22b143a32fe58317139d9b8f42","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.10@sha256:08bb5fa417aed94b40a14e2b7b3ae457531a5f22b143a32fe58317139d9b8f42"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.10.0","digest":"sha256:097512ddf58af80a620c177ae9cad93448f9a2a55c70ee8fde5cec6714522a8c","pinned_image":"ghcr.io/github/github-mcp-server:v1.10.0@sha256:097512ddf58af80a620c177ae9cad93448f9a2a55c70ee8fde5cec6714522a8c"}],"mcp_servers":[{"name":"agenticworkflows","tools":["*"]},{"name":"safeoutputs","tools":["add_comment","create_discussion","create_issue","missing_data","missing_tool","noop"]}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # @@ -284,7 +284,7 @@ jobs: GH_AW_EXPERIMENT_SPEC: '{"prompt_compression":{"variants":["verbose","caveman"],"description":"Test whether extreme prompt compression preserves output quality for meta-orchestrator workflows","hypothesis":"H0: no change in effective_tokens. H1: caveman reduces tokens by ≥20% while maintaining quality ≥90%","metric":"effective_tokens","secondary_metrics":["run_duration_seconds","issues_created","discussion_engagement_score","assessment_completeness_score"],"guardrail_metrics":[{"name":"run_success_rate","threshold":"\u003e=0.90"},{"name":"output_quality_score","threshold":"\u003e=0.70"}],"min_samples":14,"weight":[50,50],"issue":33280,"start_date":"2026-05-20","analysis_type":"mann_whitney","tags":["cost_optimization","prompt_engineering","meta_orchestrator"],"notify":{"issue":33280}}}' GH_AW_EXPERIMENT_STATE_FILE: /tmp/gh-aw/experiments/state.jsonl GH_AW_EXPERIMENT_STATE_DIR: /tmp/gh-aw/experiments - GH_AW_HARNESS_VERSION: 87110e7c96fd24eee920a67c3a72c1bf1450e62e788f948848ea5937daac158d:4495b7b8ac66500dd3724fef344313a8290166c31a12a66a75fccfd436e4b5f1 + GH_AW_HARNESS_VERSION: 87110e7c96fd24eee920a67c3a72c1bf1450e62e788f948848ea5937daac158d:2e1df0801582987fdd3e068fd57e000ffac459e9747375b768dcbefe8e5f39fe with: script: | const path = require('path'); diff --git a/.github/workflows/agent-performance-analyzer.md b/.github/workflows/agent-performance-analyzer.md index a2dad7ebef5..728a0b50ea3 100644 --- a/.github/workflows/agent-performance-analyzer.md +++ b/.github/workflows/agent-performance-analyzer.md @@ -110,6 +110,10 @@ Treat `copilot-swe-agent` as a built-in team member in attribution/engagement fi **CI vs. agentic workflow distinction:** Workflows such as `CWI`, `CGO`, `CI`, `CJS`, and `CPI` are plain CI workflows — not agentic workflows. An `action_required` conclusion on a CI workflow means GitHub is waiting for a maintainer to approve a pull-request workflow run (a GitHub Actions permission gate), **not** an agentic activation-refused. Do not count CI-workflow `action_required` runs as agentic AR. Report them separately under "CI approval-pending" and note that the fix is to approve the Copilot-bot's workflow runs at the org level or in the PR, not an agent-side change. +Use `workflow_runs.executed` and its success rate for performance scoring. Do not score workflows +with zero executed runs as failures; report high `skipped` or `action_required` counts separately as +trigger or approval gating. + ### Phase 1: Data Collection (10m) 1. Load shared metrics/memory files (read each listed path and parse its contents; treat missing files as absent). 2. Gather recent agent outputs (issues/PRs/discussions/comments + metadata). @@ -407,6 +411,9 @@ The Metrics Collector workflow runs daily and stores performance metrics in a st - Extract agent decisions and actions - Capture error messages and warnings - Record resource usage metrics + - Use `workflow_runs.executed` as the denominator for success and effectiveness rates. Do not + score workflows with zero executed runs as failures; report high `skipped` or `action_required` + counts separately as trigger or approval gating. - **CI vs. agentic distinction:** Workflows `CWI`, `CGO`, `CI`, `CJS`, and `CPI` are plain CI workflows, not agentic workflows. An `action_required` conclusion on a CI workflow means GitHub is waiting for a maintainer to approve a pull-request workflow run (a GitHub Actions permission gate). Do **not** count CI-workflow `action_required` as agentic activation-refused (AR). Track these separately as "CI approval-pending" and recommend approving the Copilot-bot's workflow runs at the org level rather than treating them as agent failures. 4. **Build agent profiles:** diff --git a/.github/workflows/metrics-collector.lock.yml b/.github/workflows/metrics-collector.lock.yml index 211927f3c0b..dbf14125e0b 100644 --- a/.github/workflows/metrics-collector.lock.yml +++ b/.github/workflows/metrics-collector.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"8737a7a53486ef07eebd3c0e04461ef7211cfdf38779eeca10564c7422555d7c","body_hash":"5fb8ebf83f3bb7e74d10bc9f9ee2a324a07b6e300aa7e54b6e5f6cc699ea3fd9","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"8737a7a53486ef07eebd3c0e04461ef7211cfdf38779eeca10564c7422555d7c","body_hash":"bfead148217b9dbd7ad1c1ef8b9ffa74bb28cb7aa371ab7f6f52c1346f103353","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-go","sha":"b7ad1dad31e06c5925ef5d2fc7ad053ef454303e","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"docker/build-push-action","sha":"53b7df96c91f9c12dcc8a07bcb9ccacbed38856a","version":"v7.3.0"},{"repo":"docker/setup-buildx-action","sha":"37fe631027851001ddb9b187196cc803df7f5f0e","version":"v4.3.0"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.4","digest":"sha256:8f18587981eff7e6291784200a88a7191d23bfd8f5723db848c640d6b4e88e46","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.4@sha256:8f18587981eff7e6291784200a88a7191d23bfd8f5723db848c640d6b4e88e46"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.4","digest":"sha256:64e668297d1b9d83ee102626e104c054ac2cf5cfd7225bf6e144f86962229882","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.4@sha256:64e668297d1b9d83ee102626e104c054ac2cf5cfd7225bf6e144f86962229882"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.4","digest":"sha256:7ad9113203642c5b12303b97e221fc6b5dd0fb0655c09a2eba9d5c6fa3201a3b","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.4@sha256:7ad9113203642c5b12303b97e221fc6b5dd0fb0655c09a2eba9d5c6fa3201a3b"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.4","digest":"sha256:35953d0beac18f642aa0bc98bb726a85288539b1fc630c16e94da33300058258","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.4@sha256:35953d0beac18f642aa0bc98bb726a85288539b1fc630c16e94da33300058258"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.10","digest":"sha256:08bb5fa417aed94b40a14e2b7b3ae457531a5f22b143a32fe58317139d9b8f42","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.10@sha256:08bb5fa417aed94b40a14e2b7b3ae457531a5f22b143a32fe58317139d9b8f42"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.10.0","digest":"sha256:097512ddf58af80a620c177ae9cad93448f9a2a55c70ee8fde5cec6714522a8c","pinned_image":"ghcr.io/github/github-mcp-server:v1.10.0@sha256:097512ddf58af80a620c177ae9cad93448f9a2a55c70ee8fde5cec6714522a8c"}],"mcp_servers":[{"name":"agenticworkflows","tools":["*"]},{"name":"safeoutputs","tools":["create_issue","missing_data","missing_tool","noop"]}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/metrics-collector.md b/.github/workflows/metrics-collector.md index ead1a400503..f6ac4d86eb2 100644 --- a/.github/workflows/metrics-collector.md +++ b/.github/workflows/metrics-collector.md @@ -77,17 +77,20 @@ If you see any `.md` files at the root (e.g. `agent-performance-latest.md`, `sha ``` Parameters (first call): - start_date: "-1d" (last 24 hours) - - count: 80 - - timeout: 3 + - count: 20 + - timeout: 1 - Include all workflows (no workflow_name filter) ``` - **Pagination loop (required)**: the `logs` tool returns a `continuation` field when it stops early (timeout or count limit). While a `continuation` field is present in the returned data, issue another `logs` call using the parameters it provides (notably `before_run_id`, plus the - original `start_date`) with `count: 80` and `timeout: 3`, and accumulate the runs from every + original `start_date`) with `count: 20` and `timeout: 1`, and accumulate the runs from every batch. Stop only when there is no `continuation` field **and** the oldest collected run is at or before the 24h window start. Do not stop after a fixed number of batches if the oldest collected run is still newer than the window start; that produces a partial ~10h snapshot. +- If a bounded `logs` call fails with the MCP gateway's 60-second context deadline instead of + returning a continuation, switch immediately to the GitHub API fallback. Retrying progressively + smaller counts has shown the same deadline failure and only consumes the collection budget. - If the logs tool continues returning `continuation` after the oldest collected run reaches the 24h window start, stop paginating and ignore the remaining older cursor because the requested window is complete. @@ -97,7 +100,10 @@ If you see any `.md` files at the root (e.g. `agent-performance-latest.md`, `sha - Total runs in last 24 hours - Successful runs (conclusion: "success") - Failed runs (conclusion: "failure", "cancelled", "timed_out") - - Calculate success rate: `successful / total` + - Skipped runs (conclusion: "skipped") + - Approval-gated runs (conclusion: "action_required") + - Executed runs: `successful + failed` + - Calculate success rate: `successful / executed`; use `null` when `executed` is 0 - Token usage and costs (if available in logs) - Execution duration statistics @@ -137,6 +143,17 @@ item shape, aggregate the counts per workflow, and set `"safe_outputs_source": cannot determine outcome status, set those items to `"pending"` rather than omitting the `safe_output_outcomes` breakdown. The same fallback applies to `engagement` fields. +**Workflow Runs Fallback (GitHub API)**: + +When the logs-based path is truncated or unavailable, paginate `list_workflow_runs` across the full +collection window and classify each run by its `conclusion`. Count `success` as `successful`; +`failure`, `cancelled`, and `timed_out` as `failed`; and count `skipped` and `action_required` +separately. Set `executed` to `successful + failed` and calculate `success_rate` as +`successful / executed`; when `executed` is 0, set `success_rate` to `null`. Never treat skipped or +approval-gated runs as failures. Include all counts in every workflow's `workflow_runs` object, +including zero values, and set `"data_source": "github_api_fallback"` plus a `collection_note` +describing why fallback was used. + **Additional Metrics via GitHub API**: - Use GitHub MCP server (default toolset) to supplement with: - Engagement metrics: reactions on issues created by workflows @@ -190,6 +207,9 @@ Create a JSON object following this schema: "total": 7, "successful": 6, "failed": 1, + "skipped": 0, + "action_required": 0, + "executed": 7, "success_rate": 0.857, "avg_duration_seconds": 180, "total_tokens": 45000, @@ -295,7 +315,8 @@ find /tmp/gh-aw/repo-memory/default/metrics/daily/ -name "*.json" -mtime +30 -de - Sum of all safe outputs (issues + PRs + comments + discussions) across all workflows **Overall Success Rate**: -- Calculate: `(sum of successful runs across all workflows) / (sum of total runs across all workflows)` +- Calculate: `(sum of successful runs across all workflows) / (sum of executed runs across all workflows)` +- Set it to `null` when the ecosystem has no executed runs **Total Resource Usage**: - Sum total tokens used across all workflows @@ -307,11 +328,12 @@ find /tmp/gh-aw/repo-memory/default/metrics/daily/ -name "*.json" -mtime +30 -de **Primary data source**: Use the agentic-workflows tool for all workflow run metrics: 1. Start with `status` tool to get workflow inventory -2. Use `logs` tool with `start_date: "-1d"`, `count: 80`, and `timeout: 3`, then follow the +2. Use `logs` tool with `start_date: "-1d"`, `count: 20`, and `timeout: 1`, then follow the `continuation` field (using its `before_run_id`) until the oldest collected run reaches the - 24h window start -3. Extract metrics from the accumulated log data (success/failure, tokens, costs, safe outputs, - typed safe-output counts, and outcome breakdowns) + 24h window start. On a 60-second context-deadline error, switch directly to the GitHub API + fallback instead of retrying smaller counts. +3. Extract metrics from the accumulated log data (success/failure/skipped/action-required, tokens, + costs, safe outputs, typed safe-output counts, and outcome breakdowns) **Secondary data source**: Use GitHub MCP server for engagement metrics only: - Reactions on issues/PRs created by workflows @@ -328,7 +350,7 @@ find /tmp/gh-aw/repo-memory/default/metrics/daily/ -name "*.json" -mtime +30 -de covered the window and found no outputs. - If token/cost data is unavailable, omit or set to null - Always include workflows in the metrics even if they have no activity (helps detect stalled workflows) -- **If the agentic-workflows `logs` tool is unavailable**, collect what you can from the GitHub API directly (workflow runs via `list_workflow_runs`) and set `"data_source": "github_api_fallback"` in the JSON +- **If the agentic-workflows `logs` tool is unavailable**, collect what you can from the GitHub API directly using the Workflow Runs Fallback rules above and set `"data_source": "github_api_fallback"` in the JSON - Set `"collection_status": "complete"` only when workflow run counts and safe-output breakdowns cover the full 24h window through logs data and/or the GitHub API fallback. Set `"collection_status": "partial"` only when both logs pagination and fallback collection fail to @@ -456,7 +478,25 @@ STORED_DATE=$(jq -r '.timestamp' /tmp/gh-aw/repo-memory/default/metrics/latest.j TODAY=$(date +%Y-%m-%d) echo "Stored date: $STORED_DATE | Today: $TODAY" -# Step 5: List all metrics files that will be committed +# Step 5: Validate workflow-run classification and success-rate semantics +jq -e ' + all(.workflows[]?.workflow_runs; + has("skipped") and + has("action_required") and + has("executed") and + .executed == ((.successful // 0) + (.failed // 0)) and + (if .executed == 0 then + .success_rate == null + else + ((.success_rate - (.successful / .executed)) | fabs) < 0.001 + end) + ) +' /tmp/gh-aw/repo-memory/default/metrics/latest.json >/dev/null || { + echo "ERROR: workflow run classifications or success rates are inconsistent" + exit 1 +} + +# Step 6: List all metrics files that will be committed echo "Files to be committed:" find /tmp/gh-aw/repo-memory/default/metrics -type f | sort ``` diff --git a/.github/workflows/workflow-health-manager.lock.yml b/.github/workflows/workflow-health-manager.lock.yml index 062be0f0286..6b55a289cee 100644 --- a/.github/workflows/workflow-health-manager.lock.yml +++ b/.github/workflows/workflow-health-manager.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"f55cc98d67c89d82f56fa7df3f0340f89a57eb5c85f86677833d7902fa0ff3d4","body_hash":"287bc2f63f60f52d2ad6c239cd461dd9d93db1cc4f283497c2e842c589d972ed","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c3c7b91b9c186144a3f68b15fcdd6f59c2b9652774347c3894369e0e68003a12","body_hash":"e05a6ed1645d79e7585c744dae6f32b76509382ce72586cd9c1ebbb3a21015d0","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.4","digest":"sha256:8f18587981eff7e6291784200a88a7191d23bfd8f5723db848c640d6b4e88e46","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.4@sha256:8f18587981eff7e6291784200a88a7191d23bfd8f5723db848c640d6b4e88e46"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.4","digest":"sha256:64e668297d1b9d83ee102626e104c054ac2cf5cfd7225bf6e144f86962229882","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.4@sha256:64e668297d1b9d83ee102626e104c054ac2cf5cfd7225bf6e144f86962229882"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.4","digest":"sha256:7ad9113203642c5b12303b97e221fc6b5dd0fb0655c09a2eba9d5c6fa3201a3b","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.4@sha256:7ad9113203642c5b12303b97e221fc6b5dd0fb0655c09a2eba9d5c6fa3201a3b"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.4","digest":"sha256:35953d0beac18f642aa0bc98bb726a85288539b1fc630c16e94da33300058258","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.4@sha256:35953d0beac18f642aa0bc98bb726a85288539b1fc630c16e94da33300058258"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.10","digest":"sha256:08bb5fa417aed94b40a14e2b7b3ae457531a5f22b143a32fe58317139d9b8f42","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.10@sha256:08bb5fa417aed94b40a14e2b7b3ae457531a5f22b143a32fe58317139d9b8f42"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.10.0","digest":"sha256:097512ddf58af80a620c177ae9cad93448f9a2a55c70ee8fde5cec6714522a8c","pinned_image":"ghcr.io/github/github-mcp-server:v1.10.0@sha256:097512ddf58af80a620c177ae9cad93448f9a2a55c70ee8fde5cec6714522a8c"}],"mcp_servers":[{"name":"safeoutputs","tools":["add_comment","create_issue","missing_data","missing_tool","noop","update_issue"]}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # @@ -553,7 +553,7 @@ jobs: GH_AW_SKILL_DIR: ".github/skills" run: bash "${RUNNER_TEMP}/gh-aw/actions/restore_inline_skills.sh" - name: Load Metrics - run: "set -euo pipefail\nmkdir -p /tmp/gh-aw/agent\nMETRICS_FILE=\"/tmp/gh-aw/repo-memory/default/metrics/latest.json\"\nif [ -f \"$METRICS_FILE\" ]; then\n jq '[.workflow_runs | to_entries[]\n | select(.value.success_rate < 0.8)]\n | sort_by(.value.success_rate) | .[0:20]' \\\n \"$METRICS_FILE\" > /tmp/gh-aw/agent/failing-workflows.json 2>/dev/null \\\n || echo '[]' > /tmp/gh-aw/agent/failing-workflows.json\nelse\n echo '[]' > /tmp/gh-aw/agent/failing-workflows.json\nfi\necho \"Metrics loaded: $(jq 'length' /tmp/gh-aw/agent/failing-workflows.json) failing workflows (<80% success)\"" + run: "set -euo pipefail\nmkdir -p /tmp/gh-aw/agent\nMETRICS_FILE=\"/tmp/gh-aw/repo-memory/default/metrics/latest.json\"\nif [ -f \"$METRICS_FILE\" ]; then\n jq '[.workflows // {} | to_entries[]\n | (.value.workflow_runs.executed //\n ((.value.workflow_runs.successful // 0) + (.value.workflow_runs.failed // 0))) as $executed\n | select($executed > 0 and (.value.workflow_runs.success_rate // 0) < 0.8)]\n | sort_by(.value.workflow_runs.success_rate) | .[0:20]' \\\n \"$METRICS_FILE\" > /tmp/gh-aw/agent/failing-workflows.json 2>/dev/null \\\n || echo '[]' > /tmp/gh-aw/agent/failing-workflows.json\nelse\n echo '[]' > /tmp/gh-aw/agent/failing-workflows.json\nfi\necho \"Metrics loaded: $(jq 'length' /tmp/gh-aw/agent/failing-workflows.json) failing workflows (<80% success)\"" - name: Download container images run: bash "${RUNNER_TEMP}/gh-aw/actions/download_docker_images.sh" ghcr.io/github/gh-aw-firewall/agent:0.28.4@sha256:8f18587981eff7e6291784200a88a7191d23bfd8f5723db848c640d6b4e88e46 ghcr.io/github/gh-aw-firewall/api-proxy:0.28.4@sha256:64e668297d1b9d83ee102626e104c054ac2cf5cfd7225bf6e144f86962229882 ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.4@sha256:7ad9113203642c5b12303b97e221fc6b5dd0fb0655c09a2eba9d5c6fa3201a3b ghcr.io/github/gh-aw-firewall/squid:0.28.4@sha256:35953d0beac18f642aa0bc98bb726a85288539b1fc630c16e94da33300058258 ghcr.io/github/gh-aw-mcpg:v0.4.10@sha256:08bb5fa417aed94b40a14e2b7b3ae457531a5f22b143a32fe58317139d9b8f42 ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196 ghcr.io/github/github-mcp-server:v1.10.0@sha256:097512ddf58af80a620c177ae9cad93448f9a2a55c70ee8fde5cec6714522a8c diff --git a/.github/workflows/workflow-health-manager.md b/.github/workflows/workflow-health-manager.md index 04ddff0bb50..b565c7b6395 100644 --- a/.github/workflows/workflow-health-manager.md +++ b/.github/workflows/workflow-health-manager.md @@ -56,9 +56,11 @@ pre-agent-steps: mkdir -p /tmp/gh-aw/agent METRICS_FILE="/tmp/gh-aw/repo-memory/default/metrics/latest.json" if [ -f "$METRICS_FILE" ]; then - jq '[.workflow_runs | to_entries[] - | select(.value.success_rate < 0.8)] - | sort_by(.value.success_rate) | .[0:20]' \ + jq '[.workflows // {} | to_entries[] + | (.value.workflow_runs.executed // + ((.value.workflow_runs.successful // 0) + (.value.workflow_runs.failed // 0))) as $executed + | select($executed > 0 and (.value.workflow_runs.success_rate // 0) < 0.8)] + | sort_by(.value.workflow_runs.success_rate) | .[0:20]' \ "$METRICS_FILE" > /tmp/gh-aw/agent/failing-workflows.json 2>/dev/null \ || echo '[]' > /tmp/gh-aw/agent/failing-workflows.json else @@ -112,8 +114,10 @@ As a meta-orchestrator for workflow health, you oversee the operational health o **Monitor workflow execution:** - Load shared metrics from: `/tmp/gh-aw/repo-memory/default/metrics/latest.json` - Use workflow_runs data for each workflow: - - Total runs, successful runs, failed runs - - Success rate (already calculated) + - Total, executed, successful, failed, skipped, and action-required runs + - Success rate (already calculated from executed runs only) +- Exclude workflows with zero executed runs from failure-rate scoring. Treat high skipped counts as + trigger-gating behavior and high action-required counts as approval gating, not execution failures. - Query recent workflow runs (past 7 days) for detailed error analysis - Track success/failure rates from metrics data - Identify workflows with: @@ -175,7 +179,7 @@ As a meta-orchestrator for workflow health, you oversee the operational health o - Identify workflows with declining quality - Calculate workflow reliability score (0-100): - Compilation success: +20 points - - Recent runs successful (from metrics): +30 points + - Recent executed runs successful (from metrics): +30 points - No timeout issues: +20 points - Proper error handling: +15 points - Up-to-date documentation: +15 points @@ -274,7 +278,7 @@ Pre-computed data is available in `/tmp/gh-aw/agent/` and is the authoritative s ### Phase 2: Health Assessment (7 minutes) 4. **Use pre-loaded metrics:** - - `/tmp/gh-aw/agent/failing-workflows.json` contains the top-20 workflows with <80% success rate (pre-filtered from `metrics/latest.json`). Use this as your starting point. + - `/tmp/gh-aw/agent/failing-workflows.json` contains the top-20 workflows with executed runs and <80% success rate (pre-filtered from `metrics/latest.json`). Use this as your starting point. - For trend analysis, load daily metrics directly: `/tmp/gh-aw/repo-memory/default/metrics/daily/*.json` 5. **Query workflow runs:**