diff --git a/scripts/fleet_monitor.sh b/scripts/fleet_monitor.sh index e74da6f9f..01b2c4aa2 100755 --- a/scripts/fleet_monitor.sh +++ b/scripts/fleet_monitor.sh @@ -257,6 +257,34 @@ if ! gh api "search/issues?q=org:${ORG}+label:fleet-tracker+is:open+is:issue&per : > "$issues_lookup_file" # ensure file exists but is empty on failure fi +# --------------------------------------------------------------------------- +# 2c. Dev-Lead timeout observability (#1019, story C of #901) +# Count dev-lead status markers by failure reason over the window, per repo, so a +# rising timeout ("wall") rate is visible distinctly from generic engine-error. +# Markers are HTML comments (``, +# Story B / #1018) that GitHub search cannot index, so — like auto_rebase_health.sh +# — we pull each repo's issue comments server-side filtered by `since=`. Only +# repos with >=1 marker in the window get a row; the section is omitted entirely +# when the fleet had no dev-lead activity. Best-effort: a comment-read failure on +# one repo is a per-repo warning, not a fatal error for the monitor. +# --------------------------------------------------------------------------- +dev_lead_reason_file=$(mktemp) +for repo in "${repos[@]}"; do + if ! comments_raw=$(gh api \ + "repos/${repo}/issues/comments?since=${CUTOFF}&per_page=100" --paginate \ + --jq '.[] | {body}' 2>/dev/null); then + echo "::warning::Cannot read issue comments for ${repo} — dev-lead timeout counts may undercount" + continue + fi + comments_json=$(printf '%s' "$comments_raw" | jq -s '.' 2>/dev/null || echo '[]') + IFS=$'\t' read -r dl_timeout dl_engine_error dl_total \ + < <(summarize_dev_lead_timeouts "$comments_json") + if [ "${dl_total:-0}" -gt 0 ]; then + printf '%s\t%s\t%s\t%s\n' "$repo" "$dl_timeout" "$dl_engine_error" "$dl_total" \ + >> "$dev_lead_reason_file" + fi +done + # --------------------------------------------------------------------------- # 3. Generate reports # --------------------------------------------------------------------------- @@ -278,16 +306,24 @@ stub_drift_section() { done } +# dev_lead_timeout_section — appends the Dev-Lead timeout ("wall") observability +# block (#1019) when any repo had dev-lead status markers in the window. +dev_lead_timeout_section() { + [ -s "$dev_lead_reason_file" ] || return 0 + printf '\n' + generate_dev_lead_timeout_report "$dev_lead_reason_file" +} + # Step Summary — Tier 1 visualizations only (Mermaid not rendered there) # GitHub Step Summary has a 1 MB hard limit per job. At ~200 bytes per row # this supports ~5 000 workflows before truncation. if [ -n "${GITHUB_STEP_SUMMARY:-}" ]; then - { report_header; generate_report "$metrics_file" "$failed_file" "false" "$issues_lookup_file"; stub_drift_section; } \ + { report_header; generate_report "$metrics_file" "$failed_file" "false" "$issues_lookup_file"; dev_lead_timeout_section; stub_drift_section; } \ >> "$GITHUB_STEP_SUMMARY" fi # Report file — full report with Mermaid charts (used as Issue body) -{ report_header; generate_report "$metrics_file" "$failed_file" "true" "$issues_lookup_file"; stub_drift_section; } \ +{ report_header; generate_report "$metrics_file" "$failed_file" "true" "$issues_lookup_file"; dev_lead_timeout_section; stub_drift_section; } \ > "$REPORT_FILE" # --------------------------------------------------------------------------- @@ -336,7 +372,7 @@ if [ -n "${GITHUB_ENV:-}" ]; then [ "$stub_drift_count" -gt 0 ] && echo "HAS_STUB_DRIFT=true" >> "$GITHUB_ENV" fi -rm -f "$metrics_file" "$failed_file" "$issues_lookup_file" +rm -f "$metrics_file" "$failed_file" "$issues_lookup_file" "$dev_lead_reason_file" [ ${#stub_drift_files[@]} -gt 0 ] && rm -f "${stub_drift_files[@]}" # --------------------------------------------------------------------------- diff --git a/scripts/fleet_report.sh b/scripts/fleet_report.sh index 5888de54b..13ec7068e 100755 --- a/scripts/fleet_report.sh +++ b/scripts/fleet_report.sh @@ -15,6 +15,16 @@ # gate list for testing; add permanent gates to the default below. FLEET_GATE_WORKFLOWS="${FLEET_GATE_WORKFLOWS:-test-deletion-guard.yml holdout-guard.yml}" +# Dev-Lead status-marker prefix (#1019, story C of #901). Mirrors +# ISSUE_MARKER_PREFIX in dev-lead-fix-issue.sh / dev-lead-retry.sh: every +# dev-lead failure/retry/escalation posts an HTML-comment marker of the form +# +# GitHub search cannot index HTML comments, so callers pass the repo's windowed +# issue comments (issues/comments?since=) and summarize_dev_lead_timeouts tallies +# the markers by their reason= field. Kept in one place so a rename in the +# dev-lead scripts is a one-line fix here. +DEV_LEAD_ISSUE_MARKER="${DEV_LEAD_ISSUE_MARKER:-\nx"}, + {"body":"\ny"}, + {"body":""} + ]' + run summarize_dev_lead_timeouts "$json" + [ "$status" -eq 0 ] + [ "$output" = $'2\t1\t3' ] +} + +@test "dev-lead timeouts: ignores comments that merely mention reason= but lack the marker" { + local json='[ + {"body":""}, + {"body":"a normal comment mentioning reason=timeout but not a marker"} + ]' + run summarize_dev_lead_timeouts "$json" + [ "$output" = $'1\t0\t1' ] +} + +@test "dev-lead timeouts: empty array yields zero counts" { + run summarize_dev_lead_timeouts '[]' + [ "$output" = $'0\t0\t0' ] +} + +@test "dev-lead timeouts: absent argument yields zero counts" { + run summarize_dev_lead_timeouts + [ "$output" = $'0\t0\t0' ] +} + +@test "dev-lead timeouts: rate-limited counts toward total but not timeout/engine-error" { + local json='[ + {"body":""}, + {"body":""} + ]' + run summarize_dev_lead_timeouts "$json" + [ "$output" = $'1\t0\t2' ] +} + +# --------------------------------------------------------------------------- +# generate_dev_lead_timeout_report — renders the per-repo section (#1019) +# Input: TSV file rows `repo timeout engine_error total`. +# --------------------------------------------------------------------------- + +@test "dev-lead timeout report: renders per-repo rows, fleet total, and timeout rate" { + local f + f="$(mktemp "$BATS_TEST_TMPDIR/fr.XXXXXX")" + printf '%s\n' \ + $'petry-projects/a\t2\t1\t4' \ + $'petry-projects/b\t1\t1\t3' \ + > "$f" + run generate_dev_lead_timeout_report "$f" + [ "$status" -eq 0 ] + [[ "$output" =~ "petry-projects/a" ]] + [[ "$output" =~ "petry-projects/b" ]] + # Fleet totals: timeout 3, engine-error 2, total 7 + [[ "$output" =~ "Fleet total" ]] + [[ "$output" =~ "reason=timeout" ]] + # Rate = 3 / 7 = 42.9% + [[ "$output" =~ "42.9%" ]] + rm -f "$f" +} + +@test "dev-lead timeout report: whole-number rate renders without a decimal" { + local f + f="$(mktemp "$BATS_TEST_TMPDIR/fr.XXXXXX")" + printf '%s\n' $'petry-projects/a\t1\t1\t2' > "$f" + run generate_dev_lead_timeout_report "$f" + [[ "$output" =~ "50%" ]] + [[ ! "$output" =~ "50.0%" ]] + rm -f "$f" +} + +@test "dev-lead timeout report: empty file produces no output" { + local f + f="$(mktemp "$BATS_TEST_TMPDIR/fr.XXXXXX")" + : > "$f" + run generate_dev_lead_timeout_report "$f" + [ "$status" -eq 0 ] + [ -z "$output" ] + rm -f "$f" +}