From 9bd0a4cb72b40b0d2126b3f0c371bcbe1b716e52 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:04:06 +0000 Subject: [PATCH 1/4] Initial plan From afee656c915ce7d7df0c59d6b4114b4374dce191 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Wed, 22 Jul 2026 10:32:45 +0000 Subject: [PATCH 2/4] Exclude intentional credit-guardrail tests from prod-main health rollups - Add intentional-failure: true to daily-credit-limit-test.md and daily-max-ai-credits-test.md - Add intentional-failure field to main_workflow_schema.json - Add IsIntentionalFailure() helper in pkg/workflow/resolve.go - Add IntentionalFailure field to RunData and IntentionalFailureRuns to LogsSummary - Exclude intentional-failure workflows from CalculateHealthSummary rollup counts - Update audit-workflows.md and deep-report.md to exclude these workflows from success-rate rollups - Recompile both affected workflows - Add unit tests for IsIntentionalFailure and CalculateHealthSummaryExcludesIntentionalFailure Co-authored-by: pelikhan <4175913+pelikhan@users.noreply.github.com> --- .github/workflows/audit-workflows.lock.yml | 2 +- .github/workflows/audit-workflows.md | 6 ++ .../daily-credit-limit-test.lock.yml | 2 +- .github/workflows/daily-credit-limit-test.md | 1 + .../daily-max-ai-credits-test.lock.yml | 2 +- .../workflows/daily-max-ai-credits-test.md | 1 + .github/workflows/deep-report.lock.yml | 2 +- .github/workflows/deep-report.md | 6 ++ pkg/cli/health_command.go | 12 +++- pkg/cli/health_metrics.go | 11 +++- pkg/cli/health_metrics_test.go | 24 ++++++++ pkg/cli/logs_report.go | 31 ++++++++-- pkg/parser/schemas/main_workflow_schema.json | 6 ++ pkg/workflow/resolve.go | 47 ++++++++++++++++ pkg/workflow/resolve_test.go | 56 +++++++++++++++++++ 15 files changed, 198 insertions(+), 11 deletions(-) diff --git a/.github/workflows/audit-workflows.lock.yml b/.github/workflows/audit-workflows.lock.yml index e5d4944d5ef..73d5c14951a 100644 --- a/.github/workflows/audit-workflows.lock.yml +++ b/.github/workflows/audit-workflows.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c7ff24d3d0dcf2b3fe92a4cb86ea4542501e862cbcdc0c6617b7dd0211a3d018","body_hash":"a0dca994e3a45f38d44e16d7503debe72c878129c052eda1bcd72330570940a1","strict":true,"agent_id":"claude","engine_versions":{"claude":"2.1.216"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c7ff24d3d0dcf2b3fe92a4cb86ea4542501e862cbcdc0c6617b7dd0211a3d018","body_hash":"62e3e27e5cb1e0b0dca7ae1275994ecafdd2ac73c2f2d21cac7630aa15c8d180","strict":true,"agent_id":"claude","engine_versions":{"claude":"2.1.216"}} # gh-aw-manifest: {"version":1,"secrets":["ANTHROPIC_API_KEY","COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0","version":"v7.0.0"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-go","sha":"b7ad1dad31e06c5925ef5d2fc7ad053ef454303e","version":"v7.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/setup-python","sha":"ece7cb06caefa5fff74198d8649806c4678c61a1","version":"v6.3.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"docker/build-push-action","sha":"53b7df96c91f9c12dcc8a07bcb9ccacbed38856a","version":"v7.3.0"},{"repo":"docker/setup-buildx-action","sha":"bb05f3f5519dd87d3ba754cc423b652a5edd6d2c","version":"v4.2.0"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37","digest":"sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37@sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37","digest":"sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37@sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37","digest":"sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37@sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.3","digest":"sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.3@sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b","pinned_image":"ghcr.io/github/gh-aw-node@sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b"},{"image":"ghcr.io/github/github-mcp-server:v1.6.0","digest":"sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3","pinned_image":"ghcr.io/github/github-mcp-server:v1.6.0@sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/audit-workflows.md b/.github/workflows/audit-workflows.md index 5ae2195b439..81fdadf7c22 100644 --- a/.github/workflows/audit-workflows.md +++ b/.github/workflows/audit-workflows.md @@ -105,6 +105,12 @@ Output is saved to: /tmp/gh-aw/aw-mcp/logs **IMPORTANT**: Do NOT infer engine type by scanning `.lock.yml` files. Lock files contain the word `copilot` in allowed-domains lists and workflow source paths regardless of which engine the workflow uses, causing false positives. +**Success Rate Rollups โ€” Exclude Intentional-Failure Workflows**: When computing the fleet-wide or prod-main success rate, **exclude** runs where `intentional_failure` is `true`. These workflows (e.g. `Daily Credit Limit Test`, `Daily Max AI Credits Test`) are credit-guardrail stress tests that are *designed* to fail; including them would depress the real-regression baseline. The `logs` tool marks them in `runs[].intentional_failure` and counts them in `summary.intentional_failure_runs`. Always report the adjusted rate alongside the raw rate, e.g. `"92.7% raw (94.2% excl. intentional failures)"`. + +**Intentional-failure workflows that MUST be excluded from all success-rate and health rollups**: +- `Daily Credit Limit Test` (`daily-credit-limit-test`) โ€” trips the `max-daily-ai-credits` guardrail by design +- `Daily Max AI Credits Test` (`daily-max-ai-credits-test`) โ€” trips the `max-ai-credits` per-run firewall by design + {{#if experiments.audit_decomposition == 'phased_sub_agents'}} **Analyze** in explicit phases: 1. **Collection phase**: summarize missing tools, hard failures, and token/runtime outliers. diff --git a/.github/workflows/daily-credit-limit-test.lock.yml b/.github/workflows/daily-credit-limit-test.lock.yml index 587931d9666..ffab498220d 100644 --- a/.github/workflows/daily-credit-limit-test.lock.yml +++ b/.github/workflows/daily-credit-limit-test.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"82ff5ef9e0f59a34aa6ae53d46068091315846875e18239b1a5a3990db475c2c","body_hash":"2c1e2a980de325aa80a2d5495f13e7f0a1de2c50cf26762c0467af1f73f5dc0d","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"6b956db3a94dd3e6463564759f899983dae7d72c3ddd42532add7148e26ada5d","body_hash":"2c1e2a980de325aa80a2d5495f13e7f0a1de2c50cf26762c0467af1f73f5dc0d","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0","version":"v7.0.0"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37","digest":"sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37@sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37","digest":"sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37@sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37","digest":"sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37@sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.3","digest":"sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.3@sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b","pinned_image":"ghcr.io/github/gh-aw-node@sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b"},{"image":"ghcr.io/github/github-mcp-server:v1.6.0","digest":"sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3","pinned_image":"ghcr.io/github/github-mcp-server:v1.6.0@sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/daily-credit-limit-test.md b/.github/workflows/daily-credit-limit-test.md index 8a08dc60357..5c49c44c9e5 100644 --- a/.github/workflows/daily-credit-limit-test.md +++ b/.github/workflows/daily-credit-limit-test.md @@ -2,6 +2,7 @@ private: true emoji: "๐Ÿงช" description: "โš ๏ธ INTENTIONALLY BROKEN โ€” Tests that max-daily-ai-credits: 1 is enforced by the activation guardrail and that a limit-exceeded message is posted when the daily budget is consumed." +intentional-failure: true on: schedule: every 12 hours workflow_dispatch: diff --git a/.github/workflows/daily-max-ai-credits-test.lock.yml b/.github/workflows/daily-max-ai-credits-test.lock.yml index 7b94745e0e2..8dfd68faef2 100644 --- a/.github/workflows/daily-max-ai-credits-test.lock.yml +++ b/.github/workflows/daily-max-ai-credits-test.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"ed0e285cdd12745fc6b94c5417b25b419f38ca95808a09af99235d2a9a3cf302","body_hash":"d0cbd35665fdbbca4938280c94f3e789f67499488e8e9d803676324167d120a6","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"b2196f0f8898b51a6cf8da885b2a7941067faa40485bf103fd77258ac2befbda","body_hash":"d0cbd35665fdbbca4938280c94f3e789f67499488e8e9d803676324167d120a6","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/checkout","sha":"9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0","version":"v7.0.0"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37","digest":"sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37@sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37","digest":"sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37@sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37","digest":"sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37@sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.3","digest":"sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.3@sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b","pinned_image":"ghcr.io/github/gh-aw-node@sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b"},{"image":"ghcr.io/github/github-mcp-server:v1.6.0","digest":"sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3","pinned_image":"ghcr.io/github/github-mcp-server:v1.6.0@sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/daily-max-ai-credits-test.md b/.github/workflows/daily-max-ai-credits-test.md index e25da763fcb..118bc2d3588 100644 --- a/.github/workflows/daily-max-ai-credits-test.md +++ b/.github/workflows/daily-max-ai-credits-test.md @@ -2,6 +2,7 @@ private: true emoji: "๐Ÿงช" description: "โš ๏ธ INTENTIONALLY FAILS โ€” Tests that max-ai-credits: 1 is enforced by the AWF firewall and that the per-run budget guardrail cuts off the agent." +intentional-failure: true on: schedule: daily around 10:30 workflow_dispatch: diff --git a/.github/workflows/deep-report.lock.yml b/.github/workflows/deep-report.lock.yml index e299837ea93..496a2453980 100644 --- a/.github/workflows/deep-report.lock.yml +++ b/.github/workflows/deep-report.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"2b251c27f6d5e535332d97eb3e28f0151d3df0cf5b1e03b3a3dce5581d8b034f","body_hash":"8f8f453268abdc2aac93a9c79b85e4859753b40809dab98f60dcb40bc32aa2a9","strict":true,"agent_id":"claude","engine_versions":{"claude":"2.1.216"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"2b251c27f6d5e535332d97eb3e28f0151d3df0cf5b1e03b3a3dce5581d8b034f","body_hash":"5e56aff6f60c9e412c77e981a34d69e00842833a5637994bff85b58d42c460a0","strict":true,"agent_id":"claude","engine_versions":{"claude":"2.1.216"}} # gh-aw-manifest: {"version":1,"secrets":["ANTHROPIC_API_KEY","COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0","version":"v7.0.0"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-go","sha":"b7ad1dad31e06c5925ef5d2fc7ad053ef454303e","version":"v7.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"},{"repo":"docker/build-push-action","sha":"53b7df96c91f9c12dcc8a07bcb9ccacbed38856a","version":"v7.3.0"},{"repo":"docker/setup-buildx-action","sha":"bb05f3f5519dd87d3ba754cc423b652a5edd6d2c","version":"v4.2.0"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37","digest":"sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37@sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37","digest":"sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37@sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.27.37","digest":"sha256:1d5300d9b08e1c4f2ad1830860656a0656383a83280058f17e805a7c3ecda203","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.27.37@sha256:1d5300d9b08e1c4f2ad1830860656a0656383a83280058f17e805a7c3ecda203"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37","digest":"sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37@sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.3","digest":"sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.3@sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b","pinned_image":"ghcr.io/github/gh-aw-node@sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b"},{"image":"ghcr.io/github/github-mcp-server:v1.6.0","digest":"sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3","pinned_image":"ghcr.io/github/github-mcp-server:v1.6.0@sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3"},{"image":"node:lts-alpine","digest":"sha256:2bdb65ed1dab192432bc31c95f94155ca5ad7fc1392fb7eb7526ab682fa5bf14","pinned_image":"node:lts-alpine@sha256:2bdb65ed1dab192432bc31c95f94155ca5ad7fc1392fb7eb7526ab682fa5bf14"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/deep-report.md b/.github/workflows/deep-report.md index f1f488f5226..33d235ea202 100644 --- a/.github/workflows/deep-report.md +++ b/.github/workflows/deep-report.md @@ -180,6 +180,12 @@ Use the gh-aw `logs` tool to: - Execution time trends - Firewall activity (if enabled) +**Success Rate Rollups โ€” Exclude Intentional-Failure Workflows**: When computing fleet-wide or prod-main success rates, **exclude** runs where `intentional_failure` is `true`. These are credit-guardrail stress tests designed to fail; including them depresses the real-regression baseline. The `logs` tool marks them via `runs[].intentional_failure` and `summary.intentional_failure_runs`. Always report the adjusted rate alongside the raw rate, e.g. `"92.7% raw (94.2% excl. intentional failures)"`. + +Intentional-failure workflows (always exclude from success-rate rollups): +- `Daily Credit Limit Test` โ€” `max-daily-ai-credits` guardrail test, expected to fail +- `Daily Max AI Credits Test` โ€” `max-ai-credits` per-run firewall test, expected to fail + ### Step 2.5: Analyze Repository Issues Use the `issues-analyst` sub-agent to analyze `/tmp/gh-aw/agent/weekly-issues-data/issues.json` and produce a structured issues summary. diff --git a/pkg/cli/health_command.go b/pkg/cli/health_command.go index 9e04c1ab201..8ea52058cb6 100644 --- a/pkg/cli/health_command.go +++ b/pkg/cli/health_command.go @@ -234,10 +234,18 @@ func displayHealthSummary(runs []WorkflowRun, config HealthConfig) error { // Group runs by workflow groupedRuns := GroupRunsByWorkflow(runs) - // Calculate health for each workflow + // Calculate health for each workflow, marking intentional-failure workflows so they + // are excluded from the fleet-health / prod-main success-rate rollup in CalculateHealthSummary. workflowHealths := make([]WorkflowHealth, 0, len(groupedRuns)) for workflowName, workflowRuns := range groupedRuns { health := CalculateWorkflowHealth(workflowName, workflowRuns, config.Threshold) + // Derive the workflow path from the first available run and check frontmatter. + for _, r := range workflowRuns { + if r.WorkflowPath != "" { + health.IntentionalFailure = workflow.IsIntentionalFailure(r.WorkflowPath) + break + } + } workflowHealths = append(workflowHealths, health) } @@ -246,7 +254,7 @@ func displayHealthSummary(runs []WorkflowRun, config HealthConfig) error { return cmp.Compare(a.SuccessRate, b.SuccessRate) }) - // Calculate summary + // Calculate summary (intentional-failure workflows excluded from rollup counts) summary := CalculateHealthSummary(workflowHealths, fmt.Sprintf("Last %d Days", config.Days), config.Threshold) // Output results diff --git a/pkg/cli/health_metrics.go b/pkg/cli/health_metrics.go index c3c15d8125c..8358eb1abbc 100644 --- a/pkg/cli/health_metrics.go +++ b/pkg/cli/health_metrics.go @@ -27,6 +27,9 @@ type WorkflowHealth struct { AvgTokens int `json:"avg_tokens" console:"-"` DisplayTokens string `json:"-" console:"header:Avg Tokens"` BelowThresh bool `json:"below_threshold" console:"-"` + // IntentionalFailure is true when the workflow is tagged with intentional-failure: true. + // These workflows are excluded from fleet-health and prod-main success-rate rollups. + IntentionalFailure bool `json:"intentional_failure,omitempty" console:"-"` } // HealthSummary represents aggregated health metrics across all workflows @@ -175,7 +178,9 @@ func calculateSuccessRate(runs []WorkflowRun) float64 { return safePercent(successCount, len(runs)) } -// CalculateHealthSummary calculates aggregated health metrics across all workflows +// CalculateHealthSummary calculates aggregated health metrics across all workflows. +// Workflows tagged with IntentionalFailure: true are excluded from the healthy/below-threshold +// counts so they do not depress fleet-health and prod-main success-rate baselines. func CalculateHealthSummary(workflowHealths []WorkflowHealth, period string, threshold float64) HealthSummary { healthMetricsLog.Printf("Calculating health summary: workflows=%d, period=%s", len(workflowHealths), period) @@ -183,6 +188,10 @@ func CalculateHealthSummary(workflowHealths []WorkflowHealth, period string, thr belowThresholdCount := 0 for _, wh := range workflowHealths { + if wh.IntentionalFailure { + // Exclude from rollup counts โ€” these workflows are expected to fail. + continue + } if wh.SuccessRate >= threshold { healthyCount++ } diff --git a/pkg/cli/health_metrics_test.go b/pkg/cli/health_metrics_test.go index 870a93f5b3c..d9d359f1c94 100644 --- a/pkg/cli/health_metrics_test.go +++ b/pkg/cli/health_metrics_test.go @@ -191,6 +191,30 @@ func TestCalculateHealthSummary(t *testing.T) { assert.Len(t, summary.Workflows, 3, "Workflows array should have 3 entries") } +func TestCalculateHealthSummaryExcludesIntentionalFailure(t *testing.T) { + // Intentional-failure workflows (e.g. credit-guardrail stress tests) must be excluded + // from fleet-health rollup counts so they do not depress the real-regression baseline. + workflowHealths := []WorkflowHealth{ + {WorkflowName: "normal-a", SuccessRate: 90.0, BelowThresh: false}, + {WorkflowName: "normal-b", SuccessRate: 75.0, BelowThresh: true}, + // These are tagged intentional-failure: should not affect HealthyWorkflows / BelowThreshold. + {WorkflowName: "daily-credit-limit-test", SuccessRate: 0.0, BelowThresh: true, IntentionalFailure: true}, + {WorkflowName: "daily-max-ai-credits-test", SuccessRate: 0.0, BelowThresh: true, IntentionalFailure: true}, + } + + summary := CalculateHealthSummary(workflowHealths, "Last 7 Days", 80.0) + + // TotalWorkflows still counts all four (the table shows them). + assert.Equal(t, 4, summary.TotalWorkflows, "TotalWorkflows should include intentional-failure entries") + + // Rollup counts must exclude the two intentional-failure workflows. + assert.Equal(t, 1, summary.HealthyWorkflows, "HealthyWorkflows should exclude intentional-failure workflows") + assert.Equal(t, 1, summary.BelowThreshold, "BelowThreshold should exclude intentional-failure workflows") + + // The full workflows slice is preserved so callers can render the table. + assert.Len(t, summary.Workflows, 4, "Workflows slice should contain all entries") +} + func TestTrendDirectionString(t *testing.T) { tests := []struct { name string diff --git a/pkg/cli/logs_report.go b/pkg/cli/logs_report.go index cd35d40b5c0..d13636656d9 100644 --- a/pkg/cli/logs_report.go +++ b/pkg/cli/logs_report.go @@ -13,6 +13,7 @@ import ( "github.com/github/gh-aw/pkg/console" "github.com/github/gh-aw/pkg/logger" "github.com/github/gh-aw/pkg/timeutil" + "github.com/github/gh-aw/pkg/workflow" ) var reportLog = logger.New("cli:logs_report") @@ -83,6 +84,13 @@ type LogsSummary struct { // regardless of which engine the workflow actually uses. EngineCounts map[string]int `json:"engine_counts,omitempty" console:"-"` + // IntentionalFailureRuns is the count of runs belonging to workflows tagged with + // intentional-failure: true. These runs are intentionally expected to fail (e.g. + // credit-guardrail stress tests) and should be excluded from prod-main / fleet-health + // success-rate rollups. Agents should subtract this count from TotalRuns before computing + // fleet-level success rates so that deliberate failures do not depress the baseline. + IntentionalFailureRuns int `json:"intentional_failure_runs,omitempty" console:"-"` + // Outcome metrics (populated when outcome evaluation is enabled) OutcomeAccepted int `json:"outcome_accepted,omitempty" console:"-"` OutcomeRejected int `json:"outcome_rejected,omitempty" console:"-"` @@ -95,10 +103,15 @@ type LogsSummary struct { // RunData contains information about a single workflow run type RunData struct { - RunID int64 `json:"run_id" console:"header:Run ID"` - Number int `json:"number" console:"-"` - WorkflowName string `json:"workflow_name" console:"header:Workflow"` - WorkflowPath string `json:"workflow_path" console:"-"` + RunID int64 `json:"run_id" console:"header:Run ID"` + Number int `json:"number" console:"-"` + WorkflowName string `json:"workflow_name" console:"header:Workflow"` + WorkflowPath string `json:"workflow_path" console:"-"` + // IntentionalFailure is true when the workflow is tagged with intentional-failure: true + // in its frontmatter (e.g. credit-guardrail stress tests that are expected to fail). + // Agents and dashboards MUST exclude these runs from prod-main and fleet-health + // success-rate rollups to avoid depressing the real-regression baseline. + IntentionalFailure bool `json:"intentional_failure,omitempty" console:"-"` Agent string `json:"agent,omitempty" console:"header:Agent,omitempty"` Engine string `json:"engine,omitempty" console:"-"` EngineID string `json:"engine_id,omitempty" console:"-"` @@ -181,6 +194,7 @@ func buildLogsData(processedRuns []ProcessedRun, outputDir string, continuation // lock file contents, which contain "copilot" in allowed-domains and source paths // regardless of which engine the workflow uses. engineCounts := make(map[string]int) + var intentionalFailureRuns int // Build runs data // Initialize as empty slice to ensure JSON marshals to [] instead of null @@ -329,6 +343,12 @@ func buildLogsData(processedRuns []ProcessedRun, outputDir string, continuation runData.WorkflowPath = inferWorkflowPathFromDisplayName(awInfo.WorkflowName) } } + // Mark runs from workflows tagged intentional-failure: true so that + // agents and dashboards can exclude them from fleet-health success-rate rollups. + runData.IntentionalFailure = workflow.IsIntentionalFailure(runData.WorkflowPath) + if runData.IntentionalFailure { + intentionalFailureRuns++ + } if run.Duration > 0 { runData.Duration = timeutil.FormatDuration(run.Duration) } @@ -370,6 +390,9 @@ func buildLogsData(processedRuns []ProcessedRun, outputDir string, continuation if len(engineCounts) > 0 { summary.EngineCounts = engineCounts } + if intentionalFailureRuns > 0 { + summary.IntentionalFailureRuns = intentionalFailureRuns + } episodes, edges := buildEpisodeData(runs, processedRuns) for _, episode := range episodes { diff --git a/pkg/parser/schemas/main_workflow_schema.json b/pkg/parser/schemas/main_workflow_schema.json index 16cf0c3bd37..70ffb70ee7b 100644 --- a/pkg/parser/schemas/main_workflow_schema.json +++ b/pkg/parser/schemas/main_workflow_schema.json @@ -11352,6 +11352,12 @@ "description": "Mark the workflow as private, preventing it from being added to other repositories via 'gh aw add'. A workflow with private: true is not meant to be shared outside its repository.", "examples": [true, false] }, + "intentional-failure": { + "type": "boolean", + "default": false, + "description": "Mark the workflow as one that is intentionally expected to fail (e.g., guardrail stress-tests). Workflows tagged with intentional-failure: true are excluded from fleet-health and prod-main success-rate rollups so they do not depress real-regression baselines.", + "examples": [true, false] + }, "check-for-updates": { "type": "boolean", "default": true, diff --git a/pkg/workflow/resolve.go b/pkg/workflow/resolve.go index 29f32668cfa..fe3c886e515 100644 --- a/pkg/workflow/resolve.go +++ b/pkg/workflow/resolve.go @@ -10,6 +10,7 @@ import ( "github.com/github/gh-aw/pkg/console" "github.com/github/gh-aw/pkg/constants" "github.com/github/gh-aw/pkg/logger" + "github.com/github/gh-aw/pkg/parser" "github.com/github/gh-aw/pkg/stringutil" "github.com/goccy/go-yaml" ) @@ -273,3 +274,49 @@ func GetAllWorkflows() ([]WorkflowNameMatch, error) { return workflows, nil } + +// IsIntentionalFailure reports whether the workflow identified by workflowPath is tagged +// with intentional-failure: true in its frontmatter. workflowPath may be: +// - a .lock.yml path (e.g. ".github/workflows/daily-credit-limit-test.lock.yml") +// - a .md path (e.g. ".github/workflows/daily-credit-limit-test.md") +// - a bare workflow ID (e.g. "daily-credit-limit-test") +// +// Returns false whenever the file cannot be read or parsed (fail-open: unknown workflows +// are not excluded from health rollups). +func IsIntentionalFailure(workflowPath string) bool { + if workflowPath == "" { + return false + } + + // Derive the markdown file path. + var mdPath string + switch { + case strings.HasSuffix(workflowPath, ".lock.yml"): + mdPath = strings.TrimSuffix(workflowPath, ".lock.yml") + ".md" + case strings.HasSuffix(workflowPath, ".md"): + mdPath = workflowPath + default: + // Treat as a bare workflow ID. + normalizedName := stringutil.NormalizeWorkflowName(workflowPath) + mdPath = filepath.Join(constants.GetWorkflowDir(), normalizedName+".md") + } + + content, err := os.ReadFile(mdPath) + if err != nil { + // File not available locally (e.g. running against a remote repo). + return false + } + + result, err := parser.ExtractFrontmatterFromContent(string(content)) + if err != nil || result == nil { + return false + } + + val, ok := result.Frontmatter["intentional-failure"] + if !ok { + return false + } + + b, ok := val.(bool) + return ok && b +} diff --git a/pkg/workflow/resolve_test.go b/pkg/workflow/resolve_test.go index 7e5455cddad..d2faf2e4c0a 100644 --- a/pkg/workflow/resolve_test.go +++ b/pkg/workflow/resolve_test.go @@ -522,3 +522,59 @@ func TestGetWorkflowLockFileName(t *testing.T) { }) } } + +func TestIsIntentionalFailure(t *testing.T) { + workflowsDir := setupWorkflowDir(t) + + // Create a workflow marked with intentional-failure: true + intentionalMD := filepath.Join(workflowsDir, "daily-credit-limit-test.md") + require.NoError(t, os.WriteFile(intentionalMD, []byte("---\nintentional-failure: true\nprivate: true\n---\n\nSome content"), 0644)) + + // Create a normal workflow (no intentional-failure) + normalMD := filepath.Join(workflowsDir, "normal-workflow.md") + require.NoError(t, os.WriteFile(normalMD, []byte("---\nprivate: false\n---\n\nSome content"), 0644)) + + tests := []struct { + name string + workflowPath string + want bool + }{ + { + name: "intentional-failure via .lock.yml path", + workflowPath: ".github/workflows/daily-credit-limit-test.lock.yml", + want: true, + }, + { + name: "intentional-failure via .md path", + workflowPath: ".github/workflows/daily-credit-limit-test.md", + want: true, + }, + { + name: "intentional-failure via bare workflow ID", + workflowPath: "daily-credit-limit-test", + want: true, + }, + { + name: "normal workflow returns false", + workflowPath: ".github/workflows/normal-workflow.lock.yml", + want: false, + }, + { + name: "non-existent workflow returns false (fail-open)", + workflowPath: ".github/workflows/does-not-exist.lock.yml", + want: false, + }, + { + name: "empty path returns false", + workflowPath: "", + want: false, + }, + } + + for _, tt := range tests { + t.Run(tt.name, func(t *testing.T) { + got := IsIntentionalFailure(tt.workflowPath) + assert.Equal(t, tt.want, got, "IsIntentionalFailure(%q) = %v, want %v", tt.workflowPath, got, tt.want) + }) + } +} From f6ac389a184bb765f16fb50c7b19c2ea9f330f5d Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Wed, 22 Jul 2026 11:18:35 +0000 Subject: [PATCH 3/4] Move intentional-failure to features.intentional-failure in frontmatter Co-authored-by: pelikhan <4175913+pelikhan@users.noreply.github.com> --- .../daily-credit-limit-test.lock.yml | 4 ++-- .github/workflows/daily-credit-limit-test.md | 6 +++--- .../daily-max-ai-credits-test.lock.yml | 4 ++-- .../workflows/daily-max-ai-credits-test.md | 6 +++--- pkg/parser/schemas/main_workflow_schema.json | 19 ++++++++++++------- pkg/workflow/resolve.go | 10 +++++++++- pkg/workflow/resolve_test.go | 2 +- 7 files changed, 32 insertions(+), 19 deletions(-) diff --git a/.github/workflows/daily-credit-limit-test.lock.yml b/.github/workflows/daily-credit-limit-test.lock.yml index ffab498220d..756b1ef675d 100644 --- a/.github/workflows/daily-credit-limit-test.lock.yml +++ b/.github/workflows/daily-credit-limit-test.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"6b956db3a94dd3e6463564759f899983dae7d72c3ddd42532add7148e26ada5d","body_hash":"2c1e2a980de325aa80a2d5495f13e7f0a1de2c50cf26762c0467af1f73f5dc0d","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"3fa6c04bb32237fbfe05339ed5ae4aee09b3bf32676087f5905dba5067bbc82e","body_hash":"2c1e2a980de325aa80a2d5495f13e7f0a1de2c50cf26762c0467af1f73f5dc0d","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0","version":"v7.0.0"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37","digest":"sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37@sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37","digest":"sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37@sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37","digest":"sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37@sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.3","digest":"sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.3@sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b","pinned_image":"ghcr.io/github/gh-aw-node@sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b"},{"image":"ghcr.io/github/github-mcp-server:v1.6.0","digest":"sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3","pinned_image":"ghcr.io/github/github-mcp-server:v1.6.0@sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # @@ -132,7 +132,7 @@ jobs: GH_AW_INFO_FIREWALL_TYPE: "squid" GH_AW_INFO_FRONTMATTER_EMOJI: "๐Ÿงช" GH_AW_COMPILED_STRICT: "true" - GH_AW_INFO_FEATURES: '{"gh-aw-detection":true}' + GH_AW_INFO_FEATURES: '{"gh-aw-detection":true,"intentional-failure":true}' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 with: script: | diff --git a/.github/workflows/daily-credit-limit-test.md b/.github/workflows/daily-credit-limit-test.md index 5c49c44c9e5..bf894f54b45 100644 --- a/.github/workflows/daily-credit-limit-test.md +++ b/.github/workflows/daily-credit-limit-test.md @@ -2,7 +2,9 @@ private: true emoji: "๐Ÿงช" description: "โš ๏ธ INTENTIONALLY BROKEN โ€” Tests that max-daily-ai-credits: 1 is enforced by the activation guardrail and that a limit-exceeded message is posted when the daily budget is consumed." -intentional-failure: true +features: + intentional-failure: true + gh-aw-detection: true on: schedule: every 12 hours workflow_dispatch: @@ -29,8 +31,6 @@ safe-outputs: run-started: "๐Ÿงช [{workflow_name}]({run_url}) โ€” credit limit test running (intentionally broken, limit: 1 AI credit/day)." run-success: "โš ๏ธ [{workflow_name}]({run_url}) completed without hitting the daily limit of 1 AI credit โ€” verify that credit accounting is working." run-failure: "๐Ÿšซ [{workflow_name}]({run_url}) {status} โ€” expected: the daily AI credit limit of 1 was reached and this run was blocked." -features: - gh-aw-detection: true --- ### Daily Credit Limit Test (Intentionally Broken) diff --git a/.github/workflows/daily-max-ai-credits-test.lock.yml b/.github/workflows/daily-max-ai-credits-test.lock.yml index 8dfd68faef2..bd5acbf8d92 100644 --- a/.github/workflows/daily-max-ai-credits-test.lock.yml +++ b/.github/workflows/daily-max-ai-credits-test.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"b2196f0f8898b51a6cf8da885b2a7941067faa40485bf103fd77258ac2befbda","body_hash":"d0cbd35665fdbbca4938280c94f3e789f67499488e8e9d803676324167d120a6","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"6e1a9f597b80403c8d2f755995a0036cb59c8484f0b5b8583321c816f929e95c","body_hash":"d0cbd35665fdbbca4938280c94f3e789f67499488e8e9d803676324167d120a6","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.73"}} # gh-aw-manifest: {"version":1,"secrets":["COPILOT_GITHUB_TOKEN","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GITHUB_TOKEN"],"actions":[{"repo":"actions/checkout","sha":"9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0","version":"v7.0.0"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37","digest":"sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.27.37@sha256:0d35e8682845f183c1c634699a8e8a6cbe2c271b867031410df74533243c5f67"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37","digest":"sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.27.37@sha256:fc2970aadaeae05993e76697d29f03dc8bfb9248ff87a8f3d8b0975485a4b317"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37","digest":"sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.27.37@sha256:5abc51995e5901c5d1daeefc957301ee409980e2e607391ec22c06cb2513327b"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.3","digest":"sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.3@sha256:3c744710ea275cd5ee65db92a1099e0d980754bd9fafda9ce67704c67004dc83"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b","pinned_image":"ghcr.io/github/gh-aw-node@sha256:529d02eb970b1161aa25c593a9c3df57fdfad5a8add328cb3b6eccef66f3183b"},{"image":"ghcr.io/github/github-mcp-server:v1.6.0","digest":"sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3","pinned_image":"ghcr.io/github/github-mcp-server:v1.6.0@sha256:2b0c48b070f61e9d3969269ead600f62d00fb237b60ac849ef3d166ee7de9ad3"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # @@ -124,7 +124,7 @@ jobs: GH_AW_INFO_FIREWALL_TYPE: "squid" GH_AW_INFO_FRONTMATTER_EMOJI: "๐Ÿงช" GH_AW_COMPILED_STRICT: "true" - GH_AW_INFO_FEATURES: '{"gh-aw-detection":true}' + GH_AW_INFO_FEATURES: '{"gh-aw-detection":true,"intentional-failure":true}' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 with: script: | diff --git a/.github/workflows/daily-max-ai-credits-test.md b/.github/workflows/daily-max-ai-credits-test.md index 118bc2d3588..6d5f141765c 100644 --- a/.github/workflows/daily-max-ai-credits-test.md +++ b/.github/workflows/daily-max-ai-credits-test.md @@ -2,7 +2,9 @@ private: true emoji: "๐Ÿงช" description: "โš ๏ธ INTENTIONALLY FAILS โ€” Tests that max-ai-credits: 1 is enforced by the AWF firewall and that the per-run budget guardrail cuts off the agent." -intentional-failure: true +features: + intentional-failure: true + gh-aw-detection: true on: schedule: daily around 10:30 workflow_dispatch: @@ -32,8 +34,6 @@ safe-outputs: run-started: "๐Ÿงช [{workflow_name}]({run_url}) โ€” per-run AI credit limit test running (intentionally fails, limit: 1 AI credit/run)." run-success: "โš ๏ธ [{workflow_name}]({run_url}) completed without hitting the per-run limit of 1 AI credit โ€” verify that max-ai-credits enforcement is working." run-failure: "๐Ÿšซ [{workflow_name}]({run_url}) {status} โ€” expected: the per-run AI credit limit of 1 was reached and the AWF firewall cut off the agent." -features: - gh-aw-detection: true --- ### Daily Max AI Credits Test (Intentionally Fails) diff --git a/pkg/parser/schemas/main_workflow_schema.json b/pkg/parser/schemas/main_workflow_schema.json index 70ffb70ee7b..6c127b48bf6 100644 --- a/pkg/parser/schemas/main_workflow_schema.json +++ b/pkg/parser/schemas/main_workflow_schema.json @@ -2823,9 +2823,17 @@ ] }, "features": { - "description": "Feature flags and configuration options for experimental or optional features in the workflow. Each feature can be a boolean flag or a string value. The 'action-tag' feature (string) specifies the tag or SHA to use when referencing actions/setup in compiled workflows (for testing purposes only).", + "description": "Feature flags and configuration options for experimental or optional features in the workflow. Each feature can be a boolean flag or a string value. The 'action-tag' feature (string) specifies the tag or SHA to use when referencing actions/setup in compiled workflows (for testing purposes only). The 'intentional-failure' feature (boolean) marks the workflow as one that is intentionally expected to fail (e.g., guardrail stress-tests); such workflows are excluded from fleet-health and prod-main success-rate rollups.", "type": "object", "additionalProperties": true, + "properties": { + "intentional-failure": { + "type": "boolean", + "default": false, + "description": "Mark the workflow as one that is intentionally expected to fail (e.g., guardrail stress-tests). Workflows tagged with features.intentional-failure: true are excluded from fleet-health and prod-main success-rate rollups so they do not depress real-regression baselines.", + "examples": [true, false] + } + }, "examples": [ { "action-tag": "v1.0.0" @@ -2833,6 +2841,9 @@ { "action-tag": "abc123def456", "experimental-feature": true + }, + { + "intentional-failure": true } ] }, @@ -11352,12 +11363,6 @@ "description": "Mark the workflow as private, preventing it from being added to other repositories via 'gh aw add'. A workflow with private: true is not meant to be shared outside its repository.", "examples": [true, false] }, - "intentional-failure": { - "type": "boolean", - "default": false, - "description": "Mark the workflow as one that is intentionally expected to fail (e.g., guardrail stress-tests). Workflows tagged with intentional-failure: true are excluded from fleet-health and prod-main success-rate rollups so they do not depress real-regression baselines.", - "examples": [true, false] - }, "check-for-updates": { "type": "boolean", "default": true, diff --git a/pkg/workflow/resolve.go b/pkg/workflow/resolve.go index fe3c886e515..a8adf5e90a9 100644 --- a/pkg/workflow/resolve.go +++ b/pkg/workflow/resolve.go @@ -312,7 +312,15 @@ func IsIntentionalFailure(workflowPath string) bool { return false } - val, ok := result.Frontmatter["intentional-failure"] + featuresRaw, ok := result.Frontmatter["features"] + if !ok { + return false + } + features, ok := featuresRaw.(map[string]any) + if !ok { + return false + } + val, ok := features["intentional-failure"] if !ok { return false } diff --git a/pkg/workflow/resolve_test.go b/pkg/workflow/resolve_test.go index d2faf2e4c0a..830a603674f 100644 --- a/pkg/workflow/resolve_test.go +++ b/pkg/workflow/resolve_test.go @@ -528,7 +528,7 @@ func TestIsIntentionalFailure(t *testing.T) { // Create a workflow marked with intentional-failure: true intentionalMD := filepath.Join(workflowsDir, "daily-credit-limit-test.md") - require.NoError(t, os.WriteFile(intentionalMD, []byte("---\nintentional-failure: true\nprivate: true\n---\n\nSome content"), 0644)) + require.NoError(t, os.WriteFile(intentionalMD, []byte("---\nfeatures:\n intentional-failure: true\nprivate: true\n---\n\nSome content"), 0644)) // Create a normal workflow (no intentional-failure) normalMD := filepath.Join(workflowsDir, "normal-workflow.md") From c9e9350060361a3bd5237ecb54fff1ab74320f97 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Wed, 22 Jul 2026 12:11:13 +0000 Subject: [PATCH 4/4] Address review feedback: remote-repo safety, add buildLogsData test, fix misleading summary text Co-authored-by: gh-aw-bot <259018956+gh-aw-bot@users.noreply.github.com> --- pkg/cli/health_command.go | 17 ++++++++----- pkg/cli/logs_report.go | 14 +++++++++- pkg/cli/logs_report_test.go | 51 +++++++++++++++++++++++++++++++++++++ 3 files changed, 75 insertions(+), 7 deletions(-) diff --git a/pkg/cli/health_command.go b/pkg/cli/health_command.go index 8ea52058cb6..053eac4730b 100644 --- a/pkg/cli/health_command.go +++ b/pkg/cli/health_command.go @@ -236,14 +236,19 @@ func displayHealthSummary(runs []WorkflowRun, config HealthConfig) error { // Calculate health for each workflow, marking intentional-failure workflows so they // are excluded from the fleet-health / prod-main success-rate rollup in CalculateHealthSummary. + // Only classify against the local checkout when no remote repo override is active; when + // --repo targets a different repository we cannot reliably read its frontmatter from the + // local filesystem, so we fail open (IntentionalFailure stays false). workflowHealths := make([]WorkflowHealth, 0, len(groupedRuns)) for workflowName, workflowRuns := range groupedRuns { health := CalculateWorkflowHealth(workflowName, workflowRuns, config.Threshold) - // Derive the workflow path from the first available run and check frontmatter. - for _, r := range workflowRuns { - if r.WorkflowPath != "" { - health.IntentionalFailure = workflow.IsIntentionalFailure(r.WorkflowPath) - break + if config.RepoOverride == "" { + // Derive the workflow path from the first available run and check frontmatter. + for _, r := range workflowRuns { + if r.WorkflowPath != "" { + health.IntentionalFailure = workflow.IsIntentionalFailure(r.WorkflowPath) + break + } } } workflowHealths = append(workflowHealths, health) @@ -339,7 +344,7 @@ func outputHealthTable(summary HealthSummary, threshold float64) error { fmt.Fprintln(os.Stderr, console.FormatWarningMessage(fmt.Sprintf("%d workflow(s) below %.0f%% success threshold", summary.BelowThreshold, threshold))) fmt.Fprintln(os.Stderr, console.FormatInfoMessage(fmt.Sprintf("Run '%s health ' for details", string(constants.CLIExtensionPrefix)))) } else { - fmt.Fprintln(os.Stderr, console.FormatSuccessMessage(fmt.Sprintf("All workflows above %.0f%% success threshold", threshold))) + fmt.Fprintln(os.Stderr, console.FormatSuccessMessage(fmt.Sprintf("All evaluated workflows above %.0f%% success threshold", threshold))) } return nil diff --git a/pkg/cli/logs_report.go b/pkg/cli/logs_report.go index 1ef2e7215df..923b53cd078 100644 --- a/pkg/cli/logs_report.go +++ b/pkg/cli/logs_report.go @@ -210,6 +210,14 @@ func buildLogsData(processedRuns []ProcessedRun, outputDir string, continuation engineCounts := make(map[string]int) var intentionalFailureRuns int + // Get the local repository slug once to guard against cross-repo misclassification. + // IsIntentionalFailure reads from the local filesystem; when a run belongs to a + // different repository the local file may not exist (fail-open) or, in edge cases, + // may match an unrelated local file. We skip detection when the run's repository + // is known and does not match the local checkout. Fails open (empty string) when + // the slug cannot be determined. + localRepo, _ := GetCurrentRepoSlug() + // Build runs data // Initialize as empty slice to ensure JSON marshals to [] instead of null runs := make([]RunData, 0, len(processedRuns)) @@ -374,7 +382,11 @@ func buildLogsData(processedRuns []ProcessedRun, outputDir string, continuation } // Mark runs from workflows tagged intentional-failure: true so that // agents and dashboards can exclude them from fleet-health success-rate rollups. - runData.IntentionalFailure = workflow.IsIntentionalFailure(runData.WorkflowPath) + // Only classify when the run comes from the same repository as the local checkout + // (or when either side is unknown), to avoid cross-repo misclassification. + if localRepo == "" || runData.Repository == "" || strings.EqualFold(localRepo, runData.Repository) { + runData.IntentionalFailure = workflow.IsIntentionalFailure(runData.WorkflowPath) + } if runData.IntentionalFailure { intentionalFailureRuns++ } diff --git a/pkg/cli/logs_report_test.go b/pkg/cli/logs_report_test.go index fa4466483b3..d31b44e2f8e 100644 --- a/pkg/cli/logs_report_test.go +++ b/pkg/cli/logs_report_test.go @@ -1230,3 +1230,54 @@ func TestBuildLogsDataNoFailuresProducesZeroDriverExitCount(t *testing.T) { t.Errorf("Expected TotalAgentLogicFailures = 0, got %d", data.Summary.TotalAgentLogicFailures) } } + +// TestBuildLogsDataIntentionalFailure verifies that buildLogsData correctly marks +// runs[].intentional_failure for workflows tagged with features.intentional-failure: true +// and accumulates the count in summary.intentional_failure_runs. +func TestBuildLogsDataIntentionalFailure(t *testing.T) { + // Set up a temp dir with a real workflow file so IsIntentionalFailure can read it. + tempDir := t.TempDir() + workflowsDir := filepath.Join(tempDir, ".github", "workflows") + if err := os.MkdirAll(workflowsDir, 0755); err != nil { + t.Fatalf("failed to create workflows dir: %v", err) + } + t.Chdir(tempDir) + + // Intentional-failure workflow. + intentionalMD := filepath.Join(workflowsDir, "credit-guardrail.md") + if err := os.WriteFile(intentionalMD, []byte("---\nfeatures:\n intentional-failure: true\n---\n"), 0644); err != nil { + t.Fatalf("failed to write intentional workflow file: %v", err) + } + + processedRuns := []ProcessedRun{ + {Run: WorkflowRun{ + DatabaseID: 1, + WorkflowName: "Credit Guardrail", + WorkflowPath: ".github/workflows/credit-guardrail.lock.yml", + Conclusion: "failure", + }}, + {Run: WorkflowRun{ + DatabaseID: 2, + WorkflowName: "Normal Workflow", + WorkflowPath: ".github/workflows/normal-workflow.lock.yml", + Conclusion: "success", + }}, + } + + data := buildLogsData(processedRuns, "/tmp/logs", nil) + + byID := make(map[int64]RunData) + for _, r := range data.Runs { + byID[r.RunID] = r + } + + if !byID[1].IntentionalFailure { + t.Error("run 1 (credit-guardrail): expected IntentionalFailure=true, got false") + } + if byID[2].IntentionalFailure { + t.Error("run 2 (normal-workflow): expected IntentionalFailure=false, got true") + } + if data.Summary.IntentionalFailureRuns != 1 { + t.Errorf("expected IntentionalFailureRuns=1, got %d", data.Summary.IntentionalFailureRuns) + } +}