From 9b506487cc27a88da687f0867e591c59aec3dee9 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Tue, 18 Aug 2026 17:49:57 +0000 Subject: [PATCH 1/3] Initial plan From 424ebe3685c1be782da7d55962820abb0f363847 Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Tue, 18 Aug 2026 18:00:56 +0000 Subject: [PATCH 2/3] Reduce benchmark noise: time-based benchtime, median baseline, environment-noise guard Co-authored-by: pelikhan <4175913+pelikhan@users.noreply.github.com> --- .../workflows/daily-cli-performance.lock.yml | 2 +- .github/workflows/daily-cli-performance.md | 43 +++++++++++++++---- Makefile | 6 ++- 3 files changed, 40 insertions(+), 11 deletions(-) diff --git a/.github/workflows/daily-cli-performance.lock.yml b/.github/workflows/daily-cli-performance.lock.yml index 1f6745bdd40..cc35791166f 100644 --- a/.github/workflows/daily-cli-performance.lock.yml +++ b/.github/workflows/daily-cli-performance.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c0080016ba139f9c8e441376f19be2aaac3595db831cb13bab06c158bfd6507c","body_hash":"c9d4b015bae721ec85666314efeea7a73bc37a9219904485db91b0b1ec9a8c13","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c0080016ba139f9c8e441376f19be2aaac3595db831cb13bab06c158bfd6507c","body_hash":"ef416be65c3e941ed08ea097f5a9cf810abffa1379211a17ac799a915fdc8e11","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} # gh-aw-manifest: {"version":1,"secrets":["DOCKER_PAT","DOCKER_USERNAME","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1","digest":"sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1@sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1","digest":"sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1@sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.1","digest":"sha256:f931e5e1e13f765605d03ef9511fc755d779a51b76581ea14586e9871506a610","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.1@sha256:f931e5e1e13f765605d03ef9511fc755d779a51b76581ea14586e9871506a610"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1","digest":"sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1@sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/daily-cli-performance.md b/.github/workflows/daily-cli-performance.md index e29394bc4fc..f8707e9b971 100644 --- a/.github/workflows/daily-cli-performance.md +++ b/.github/workflows/daily-cli-performance.md @@ -301,6 +301,7 @@ Analyze benchmark trends and detect performance regressions """ import json import os +import statistics from datetime import datetime, timedelta from pathlib import Path @@ -316,6 +317,11 @@ MAX_HISTORY_ENTRIES = 14 REGRESSION_THRESHOLD = 1.10 # 10% slower is a regression WARNING_THRESHOLD = 1.05 # 5% slower is a warning +# Environment-noise detection: when most unrelated benchmarks regress at once, the +# runner was almost certainly noisy (shared/overloaded CI host), not the code. +NOISE_MIN_REGRESSIONS = 3 # need at least this many regressions to suspect noise +NOISE_REGRESSION_RATIO = 0.5 # ...covering at least this fraction of all benchmarks + def load_history(): """Load historical benchmark data — capped at last MAX_HISTORY_ENTRIES entries to bound context size""" history = [] @@ -353,9 +359,10 @@ def analyze_benchmark(name, current_ns, history_data): 'change_percent': 0 } - # Calculate average of recent history (last 7 data points) + # Use the median of recent history (last 7 data points) as the baseline — + # the median is far less sensitive to one-off noisy runs than the mean recent_history = historical_values[-7:] if len(historical_values) >= 7 else historical_values - avg_historical = sum(recent_history) / len(recent_history) + avg_historical = statistics.median(recent_history) # Calculate change percentage change_percent = ((current_ns - avg_historical) / avg_historical) * 100 @@ -416,12 +423,24 @@ def main(): elif result['status'] == 'stable': analysis['summary']['stable'] += 1 + # Detect likely environment noise: many unrelated benchmarks regressing in the + # same run points at a noisy runner rather than a real code regression + summary_counts = analysis['summary'] + regressions = summary_counts['regressions'] + total = summary_counts['total'] + likely_noise = ( + regressions >= NOISE_MIN_REGRESSIONS + and total > 0 + and regressions / total >= NOISE_REGRESSION_RATIO + ) + analysis['summary']['likely_environment_noise'] = likely_noise + # Save analysis with open(OUTPUT_FILE, 'w') as f: json.dump(analysis, f, indent=2) summary = analysis['summary'] - print(f"Analysis complete! total={summary['total']} regressions={summary['regressions']} warnings={summary['warnings']} improvements={summary['improvements']}") + print(f"Analysis complete! total={summary['total']} regressions={summary['regressions']} warnings={summary['warnings']} improvements={summary['improvements']} likely_environment_noise={summary['likely_environment_noise']}") if __name__ == '__main__': main() @@ -448,10 +467,11 @@ cat /tmp/gh-aw/agent/benchmarks/analysis.json | python3 -m json.tool If regressions are detected, open issues with detailed information. **Rules for opening issues:** -1. Open one issue per regression detected (max 3 as per safe-outputs config) -2. Include benchmark name, current performance, historical average, and change percentage -3. Add "performance" and "automation" labels -4. Use title format: `[performance] Regression in [BenchmarkName]: X% slower` +1. Do **not** open regression issues when the analysis reports `"likely_environment_noise": true` (many unrelated benchmarks regressing at once indicates a noisy runner, not a code regression) — mention it in the report instead and recommend a re-run +2. Otherwise, open one issue per regression detected (max 3 as per safe-outputs config) +3. Include benchmark name, current performance, historical median, and change percentage +4. Add "performance" and "automation" labels +5. Use title format: `[performance] Regression in [BenchmarkName]: X% slower` **Issue template:** @@ -535,6 +555,13 @@ def main(): print("✅ No performance regressions detected!") return + if analysis['summary'].get('likely_environment_noise'): + print(f"⏭️ {len(regressions)} regression(s) detected across unrelated benchmarks — " + "classified as likely environment noise, no regression issues will be opened.") + with open('/tmp/gh-aw/agent/benchmarks/regressions.json', 'w') as f: + json.dump([], f, indent=2) + return + print(f"⚠️ Found {len(regressions)} regression(s):") for reg in regressions: print(f" - {reg['name']}: {reg['change_percent']:+.1f}%") @@ -551,7 +578,7 @@ chmod +x /tmp/gh-aw/agent/benchmarks/create_issues.py python3 /tmp/gh-aw/agent/benchmarks/create_issues.py ``` -Now, for each regression found, use the `create issue` tool to open an issue with the details. +Now, for each regression found in `regressions.json` (empty when the run was classified as likely environment noise), use the `create issue` tool to open an issue with the details. ## Phase 5: Generate Performance Report diff --git a/Makefile b/Makefile index 8030cfca19e..9ed7db34810 100644 --- a/Makefile +++ b/Makefile @@ -170,17 +170,19 @@ bench: go test -bench=. -benchmem -benchtime=3x -run=^$$ ./pkg/... | tee bench_results.txt # Run only critical performance benchmarks for daily monitoring +# Uses time-based -benchtime (not a fixed low iteration count) so that results are +# averaged over many iterations and are far less sensitive to shared CI runner noise. .PHONY: bench-performance bench-performance: @echo "Running critical performance benchmarks..." @echo "This includes: CompileSimpleWorkflow, CompileComplexWorkflow, CompileMCPWorkflow," @echo " CompileMemoryUsage, ParseWorkflow, Validation, YAMLGeneration" @go test -bench='Benchmark(CompileSimpleWorkflow|CompileComplexWorkflow|CompileMCPWorkflow|CompileMemoryUsage|ParseWorkflow|Validation|YAMLGeneration)$$' \ - -benchmem -benchtime=3x -run=^$$ ./pkg/workflow | tee bench_performance.txt + -benchmem -benchtime=2s -run=^$$ ./pkg/workflow | tee bench_performance.txt @echo "" @echo "Also running CLI helper benchmarks..." @go test -bench='Benchmark(ExtractWorkflowNameFromFile|FindIncludesInContent)$$' \ - -benchmem -benchtime=1s -run=^$$ ./pkg/cli >> bench_performance.txt + -benchmem -benchtime=2s -run=^$$ ./pkg/cli >> bench_performance.txt @echo "" @echo "Performance benchmark results saved to bench_performance.txt" From 490b21ef748b1dc16512673d8bb0fe9104e7b1bf Mon Sep 17 00:00:00 2001 From: "copilot-swe-agent[bot]" <198982749+Copilot@users.noreply.github.com> Date: Tue, 18 Aug 2026 18:06:15 +0000 Subject: [PATCH 3/3] Count warnings toward environment-noise ratio Co-authored-by: pelikhan <4175913+pelikhan@users.noreply.github.com> --- .github/workflows/daily-cli-performance.lock.yml | 2 +- .github/workflows/daily-cli-performance.md | 8 +++++--- 2 files changed, 6 insertions(+), 4 deletions(-) diff --git a/.github/workflows/daily-cli-performance.lock.yml b/.github/workflows/daily-cli-performance.lock.yml index cc35791166f..4754e5743b9 100644 --- a/.github/workflows/daily-cli-performance.lock.yml +++ b/.github/workflows/daily-cli-performance.lock.yml @@ -1,4 +1,4 @@ -# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c0080016ba139f9c8e441376f19be2aaac3595db831cb13bab06c158bfd6507c","body_hash":"ef416be65c3e941ed08ea097f5a9cf810abffa1379211a17ac799a915fdc8e11","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} +# gh-aw-metadata: {"schema_version":"v4","frontmatter_hash":"c0080016ba139f9c8e441376f19be2aaac3595db831cb13bab06c158bfd6507c","body_hash":"aaf0bab621217d5a51e99946318338045c88380b8102901bf9844044419ba8ad","strict":true,"agent_id":"copilot","engine_versions":{"copilot":"1.0.80","copilot-sdk":"1.0.11"}} # gh-aw-manifest: {"version":1,"secrets":["DOCKER_PAT","DOCKER_USERNAME","GH_AW_GITHUB_MCP_SERVER_TOKEN","GH_AW_GITHUB_TOKEN","GH_AW_OTEL_GRAFANA_AUTHORIZATION","GH_AW_OTEL_GRAFANA_ENDPOINT","GH_AW_OTEL_SENTRY_AUTHORIZATION","GH_AW_OTEL_SENTRY_ENDPOINT","GITHUB_TOKEN"],"actions":[{"repo":"actions/cache/restore","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/cache/save","sha":"55cc8345863c7cc4c66a329aec7e433d2d1c52a9","version":"v6.1.0"},{"repo":"actions/checkout","sha":"3d3c42e5aac5ba805825da76410c181273ba90b1","version":"v7.0.1"},{"repo":"actions/download-artifact","sha":"3e5f45b2cfb9172054b4087a40e8e0b5a5461e7c","version":"v8.0.1"},{"repo":"actions/github-script","sha":"3a2844b7e9c422d3c10d287c895573f7108da1b3","version":"v9.0.0"},{"repo":"actions/setup-node","sha":"820762786026740c76f36085b0efc47a31fe5020","version":"v7.0.0"},{"repo":"actions/upload-artifact","sha":"043fb46d1a93c77aae656e7c1c64a875d1fc6a0a","version":"v7.0.1"}],"containers":[{"image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1","digest":"sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d","pinned_image":"ghcr.io/github/gh-aw-firewall/agent:0.28.1@sha256:5e3f6ee27eeae07195838b97ac4aa2f8aea42a7c55f1c0d3e17d8e88e294ad0d"},{"image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1","digest":"sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c","pinned_image":"ghcr.io/github/gh-aw-firewall/api-proxy:0.28.1@sha256:288e7d2a12d5b430500d739f9c16e20bb1ed51b91f986f3f3eccde189f489f5c"},{"image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.1","digest":"sha256:f931e5e1e13f765605d03ef9511fc755d779a51b76581ea14586e9871506a610","pinned_image":"ghcr.io/github/gh-aw-firewall/cli-proxy:0.28.1@sha256:f931e5e1e13f765605d03ef9511fc755d779a51b76581ea14586e9871506a610"},{"image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1","digest":"sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f","pinned_image":"ghcr.io/github/gh-aw-firewall/squid:0.28.1@sha256:9d428af47899bf18ef2d5618075777d76ef344c91e76c1f44ec1aaa0ee347e5f"},{"image":"ghcr.io/github/gh-aw-mcpg:v0.4.9","digest":"sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f","pinned_image":"ghcr.io/github/gh-aw-mcpg:v0.4.9@sha256:e5a1569aeaf41820fa7bdee3e94468cae448133cdbf00119ad24f5b74db1ab9f"},{"image":"ghcr.io/github/gh-aw-node","digest":"sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196","pinned_image":"ghcr.io/github/gh-aw-node@sha256:0d9f1fb5fd6610c0ac1f5194a38e45a8a1e81f8a390d5142d8e4e6f26a4b3196"},{"image":"ghcr.io/github/github-mcp-server:v1.9.0","digest":"sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e","pinned_image":"ghcr.io/github/github-mcp-server:v1.9.0@sha256:881b53d6f75f69bdbc1b5b10fc2f1361717c19054143b3a8529fb5c32061a50e"}]} # This file was automatically generated by gh-aw. DO NOT EDIT. To debug this workflow, load the skill at https://github.com/github/gh-aw/blob/main/debug.md # diff --git a/.github/workflows/daily-cli-performance.md b/.github/workflows/daily-cli-performance.md index f8707e9b971..aed8d696b15 100644 --- a/.github/workflows/daily-cli-performance.md +++ b/.github/workflows/daily-cli-performance.md @@ -423,15 +423,17 @@ def main(): elif result['status'] == 'stable': analysis['summary']['stable'] += 1 - # Detect likely environment noise: many unrelated benchmarks regressing in the - # same run points at a noisy runner rather than a real code regression + # Detect likely environment noise: many unrelated benchmarks slowing down in the + # same run points at a noisy runner rather than a real code regression. + # Warnings count as degraded too, since noise spreads unevenly across benchmarks. summary_counts = analysis['summary'] regressions = summary_counts['regressions'] + degraded = regressions + summary_counts['warnings'] total = summary_counts['total'] likely_noise = ( regressions >= NOISE_MIN_REGRESSIONS and total > 0 - and regressions / total >= NOISE_REGRESSION_RATIO + and degraded / total >= NOISE_REGRESSION_RATIO ) analysis['summary']['likely_environment_noise'] = likely_noise