diff --git a/scripts/canary-rollout.sh b/scripts/canary-rollout.sh index c81f6aba..b118d5e3 100755 --- a/scripts/canary-rollout.sh +++ b/scripts/canary-rollout.sh @@ -775,31 +775,108 @@ _failure_suspect() { return 1 } -# _cumulative_health — failures + -# startup_failures across EVERY given tier repo since the candidate cut. Failures +# Memoization for _run_reusable_sha. Callers run it in command substitutions (subshells), so the +# in-memory map alone is lost between the health pass and the blocker-evidence pass; when the +# sweep has a _RUNS_CACHE_DIR the result is also persisted there so each run's log is fetched once. +declare -A _RUN_SHA_CACHE=() + +# _run_reusable_sha — every distinct commit SHA the run's calls to THIS +# agent's reusable resolved to (one per line), read from the "Uses: /@ ()" +# lines GitHub prints in the run log (#1176). Only lines naming the registry host + full reusable +# path count, so a different workflow with a similar filename is never attributed. Empty (exit 1) +# when the log is unreadable or has no such line; an unknown SHA is never attributed to an older +# release (fail closed — see _run_is_stale). Lookup failures are cached as empty too. +_run_reusable_sha() { + local agent="$1" repo="$2" id="$3" key="$1:$2:$3" reusable host cachef="" keyhash log shas="" line + { [ -z "$repo" ] || [ "$repo" = '*' ] || [ -z "$id" ]; } && { echo ""; return 1; } + if [[ -v _RUN_SHA_CACHE["$key"] ]]; then + shas="${_RUN_SHA_CACHE[$key]}"; [ -n "$shas" ] && printf '%s\n' "$shas"; [ -n "$shas" ]; return + fi + if [ -n "${_RUNS_CACHE_DIR:-}" ] && [ -d "$_RUNS_CACHE_DIR" ]; then + # Hash the key into the filename (as _repo_wf_runs_cached does): char-substitution would map + # distinct (agent, repo, run) keys to one file and could hand one run another run's SHA. + keyhash="$(printf '%s' "$key" | { sha256sum 2>/dev/null || shasum -a 256 2>/dev/null; } | cut -d' ' -f1)" + [ -n "$keyhash" ] || keyhash="${key//[^A-Za-z0-9._-]/_}" + cachef="$_RUNS_CACHE_DIR/sha_${keyhash}" + if [ -f "$cachef" ]; then + shas="$(<"$cachef")"; _RUN_SHA_CACHE["$key"]="$shas" + [ -n "$shas" ] && printf '%s\n' "$shas"; [ -n "$shas" ]; return + fi + fi + reusable="$(_agent_field "$agent" reusable)" + host="$(_agent_field "$agent" host)"; host="${host:-$THIS_REPO}" + log="$(gh run view "$id" --repo "$repo" --log 2>/dev/null)" || log="" + if [ -n "$reusable" ]; then + # `gh run view --log` lines are "\t\t ". GitHub prints the genuine + # "Uses:" line while setting a job up, before any of that job's own output, so only the FIRST + # "Uses:" line per job is trusted: a later line a job merely echoes (a forged old SHA) is ignored. + # The step column is NOT used — real logs often show it as "UNKNOWN STEP" (verified). + local -A seen_job=() + local jobcol text + while IFS=$'\t' read -r jobcol _ text; do + [[ "$text" =~ ^[0-9]{4}-[0-9]{2}-[0-9]{2}T[0-9:.]+Z\ (Uses:\ .*)$ ]] || continue + line="${BASH_REMATCH[1]}" + [[ -v seen_job["$jobcol"] ]] && continue + seen_job["$jobcol"]=1 + [[ "$line" == "Uses: $host/$reusable@"* ]] || continue + [[ "$line" =~ \(([0-9a-f]{7,40})\)$ ]] && shas+="${BASH_REMATCH[1]}"$'\n' + done <<< "$log" + shas="$(printf '%s' "$shas" | sort -u)" + fi + _RUN_SHA_CACHE["$key"]="$shas" + [ -n "$cachef" ] && printf '%s' "$shas" > "$cachef" 2>/dev/null || true + [ -n "$shas" ] && printf '%s\n' "$shas"; [ -n "$shas" ] +} + +# _run_is_stale — 0 only when the run PROVABLY executed a release +# other than the candidate (#1176): a ring's members run the previous release until promoted, so +# their failures say nothing about the candidate. Anything undeterminable (no cand, no Uses: line) +# or ambiguous (a run that called the reusable at several SHAs, any of them the candidate) is NOT +# stale, so it still counts and blocks — the gate never gets more permissive on a guess. +_run_is_stale() { + local agent="$1" cand="$2" repo="$3" id="$4" sha shas + { [ -z "$cand" ] || [ "$cand" = "-" ]; } && return 1 + shas="$(_run_reusable_sha "$agent" "$repo" "$id")" || return 1 + while IFS= read -r sha; do + [ -z "$sha" ] && continue + case "$cand" in "$sha"*) return 1 ;; esac + case "$sha" in "$cand"*) return 1 ;; esac + done <<< "$shas" + return 0 +} + +# _cumulative_health — failures + +# startup_failures across EVERY given tier repo since the candidate cut. Only runs that executed +# the candidate count (#1176): a failure from a run provably on an OLDER release (see +# _run_is_stale) is tallied separately as target-ring health and never blocks. cand "-" disables +# that attribution (every failure counts, the pre-#1176 behaviour). Failures # matching the per-reusable known-benign allowlist (#1025 P2) are counted separately and # excluded from the blocking total; which allowlist entries apply depends on whether the # candidate changed the reusable (differs — see _benign_patterns, #668). A counted (non- # benign) failure that matches a `suspect_failure_classes` entry sets the suspect flag # (#668 increment 2) — it still counts toward the blocking total (SUSPECT blocks like # REGRESSION), but downstream triage renders SUSPECT + guidance instead of a bare -# REGRESSION. Prints " ". +# REGRESSION. Prints " ". _cumulative_health() { - local agent="$1" since="$2" differs="$3"; shift 3 - local wf repo json fail=0 startup=0 benign=0 suspect=0 patterns="" suspect_patterns="" rid rwf + local agent="$1" since="$2" differs="$3" cand="${4:--}"; shift 4 + local wf repo json fail=0 startup=0 benign=0 suspect=0 stale=0 patterns="" suspect_patterns="" rid rwf wf="$(_agent_field "$agent" run_workflow)" patterns="$(_benign_patterns "$agent" "$differs")" suspect_patterns="$(_suspect_patterns "$agent")" for repo in "$@"; do json="$(_run_json "$repo" "$wf" "$since")" + # startup_failure runs never executed a job, so no release can be attributed to them (no + # "Uses:" log line): they always count, fail closed (#1176). startup=$(( startup + $(jq '[.[]?|select(.conclusion=="startup_failure")]|length' 2>/dev/null <<< "${json:-[]}" || echo 0) )) - if [ -z "$patterns" ] && [ -z "$suspect_patterns" ]; then - # No benign or suspect patterns to match — count all failures with one jq pass, - # avoiding a gh run view call per failure. + if [ -z "$patterns" ] && [ -z "$suspect_patterns" ] && [ "$cand" = "-" ]; then + # No benign/suspect patterns and no candidate attribution — count all failures with one + # jq pass, avoiding a gh run view call per failure. fail=$(( fail + $(jq '[.[]?|select(.conclusion=="failure")]|length' 2>/dev/null <<< "${json:-[]}" || echo 0) )) else while IFS=$'\t' read -r rid rwf; do - if [ -n "$patterns" ] && _failure_benign "$repo" "$rid" "$rwf" "$patterns"; then + if _run_is_stale "$agent" "$cand" "$repo" "$rid"; then + stale=$(( stale + 1 )) + elif [ -n "$patterns" ] && _failure_benign "$repo" "$rid" "$rwf" "$patterns"; then benign=$(( benign + 1 )) else fail=$(( fail + 1 )) @@ -810,7 +887,7 @@ _cumulative_health() { done < <(jq -r '.[]?|select(.conclusion=="failure")|[(.databaseId // "" | tostring),(.workflowName // "")]|@tsv' 2>/dev/null <<< "$json") fi done - echo "$fail $startup $benign $suspect" + echo "$fail $startup $benign $suspect $stale" } # _baseline_daily — per-day EXECUTED counts on the @@ -953,19 +1030,25 @@ _correctness_verdict() { # _correctness_verdict (the per-run classes are memoized in _RUN_DECISION_CACHE, so the second # pass is cache-warm). Empty when the agent has no gate.correctness or the frontier is resolved. _decision_mix_table() { - local agent="$1" cand="$2" + local agent="$1" cand="$2" pair_transition="${3:-}" local knobs; knobs="$(_jq -c --arg a "$agent" '.agents[$a].gate.correctness // ""')" [ -z "$knobs" ] || [ "$knobs" = '""' ] && return 0 local chans frontier="" ch c chans="$(ordered_channels "$agent")" - local chan_array=(); IFS=, read -r -a chan_array <<< "$chans" - for ch in "${chan_array[@]}"; do - c="$(channel_commit "$agent" "$ch")" - if [ "$ch" = "next" ] || [ "$c" = "$cand" ]; then :; else frontier="$ch"; break; fi - done - [ -z "$frontier" ] && return 0 local transition source cut_z prefix - transition="$(transition_key "$frontier" "$chans")" + if [ -n "$pair_transition" ]; then + # The blocked pair is known (#1118): use ITS source tier, not one re-derived on the assumption + # that the candidate always sits on `next`. + transition="$pair_transition" + else + local chan_array=(); IFS=, read -r -a chan_array <<< "$chans" + for ch in "${chan_array[@]}"; do + c="$(channel_commit "$agent" "$ch")" + if [ "$ch" = "next" ] || [ "$c" = "$cand" ]; then :; else frontier="$ch"; break; fi + done + [ -z "$frontier" ] && return 0 + transition="$(transition_key "$frontier" "$chans")" + fi source="${transition%%->*}" cut_z="$(candidate_cut_date "$agent" "$cand")" [ -z "$cut_z" ] && return 0 @@ -997,38 +1080,21 @@ _decision_mix_table() { ' 2>/dev/null || return 0 } -# _frontier_state — compute the rollout frontier and graduated gate, echoing: -# " " -# frontier = first ring (after next) not yet on the candidate commit; triage is "-" -# unless state is BLOCKED (then REGRESSION | PRE_EXISTING | SUSPECT). mix_shift is "SHIFT" -# when a gate.correctness decision-mix shift is holding the promotion (#668 L2), else "-". -# downgrade is "DOWNGRADE" when a SUSPECT was auto-downgraded to PRE_EXISTING (#668 inc6) — -# then triage already reads PRE_EXISTING and dg_* carry the candidate-vs-baseline permille -# rates + sample sizes that drove it; "-"/0 otherwise. -_frontier_state() { - local agent="$1" - local cand chans frontier="" - cand="$(channel_commit "$agent" next)" - chans="$(ordered_channels "$agent")" - - local chan_array=() - IFS=, read -r -a chan_array <<< "$chans" - local ch - for ch in "${chan_array[@]}"; do - local c; c="$(channel_commit "$agent" "$ch")" - if [ "$ch" = "next" ] || [ "$c" = "$cand" ]; then :; else frontier="$ch"; break; fi - done - if [ -z "$frontier" ]; then - echo "$cand - - COMPLETE 0 0 0 0 0 0 0 - -"; return 0 - fi - - local transition source cut_z now_epoch - transition="$(transition_key "$frontier" "$chans")" - source="${transition%%->*}" +# _pair_state — evaluate the graduated gate for +# ONE ring transition (#1118). The candidate is the commit CURRENTLY ON (not always +# `next`); the gate measures dwell from THAT candidate's own cut, samples on the tier, +# and scopes cumulative health to every tier EXCEPT a ring strictly BELOW that runs a +# DIFFERENT (newer) candidate — so a newer candidate churning on a lower tier can never block an +# older, independently-clean candidate on a higher pair. Echoes the same 18-field line documented +# on _frontier_state. triage/mix_shift/downgrade semantics are unchanged from the single-frontier +# implementation this was extracted from. +_pair_state() { + local agent="$1" source="$2" frontier="$3" cand="$4" transition="$5" chans="$6" commits_csv="$7" + local cut_z now_epoch cut_z="$(candidate_cut_date "$agent" "$cand")" if [ -z "$cut_z" ]; then # Cannot determine the per-candidate window start — fail closed to prevent unbounded history queries. - echo "$cand $frontier $transition BLOCKED 0 0 0 0 0 0 0 - -"; return 0 + echo "${cand:--} $frontier $transition BLOCKED 0 0 0 0 0 0 0 - -"; return 0 fi now_epoch="$(date -u +%s)" @@ -1053,18 +1119,62 @@ _frontier_state() { # (#668) — inherently context-caused failures that can never be candidate-introduced — # and every other class is disabled, so the allowlist can never mask a candidate-introduced # regression (#1025 P2). - local prior differs - prior="$(channel_commit "$agent" "$frontier")" + # chans / commits_csv are the ring order and each ring's commit ("-" = unresolvable), resolved ONCE + # by _frontier_state: channel_commit runs in a subshell, so its own cache is lost between calls and + # every lookup would be another API round-trip. + local chan_array=() commit_array=() idx src_idx=-1 dst_idx=-1 + IFS=, read -r -a chan_array <<< "$chans" + IFS=, read -r -a commit_array <<< "$commits_csv" + for idx in "${!chan_array[@]}"; do + [ "${chan_array[$idx]}" = "$source" ] && src_idx="$idx" + [ "${chan_array[$idx]}" = "$frontier" ] && dst_idx="$idx" + done + local prior="${commit_array[$dst_idx]:--}" differs + [ "$prior" = "-" ] && prior="" differs="$(_reusable_differs "$agent" "$cand" "$prior")" - # Cumulative health across EVERY concrete tier repo since the candidate's own cut. - local all_repos=() ch3 - for ch3 in "${chan_array[@]}"; do - while IFS= read -r r; do [ -n "$r" ] && [ "$r" != '*' ] && all_repos+=("$r"); done \ - < <(resolve_members "$agent" "$ch3") + # Cumulative health since this candidate's own cut, across every concrete tier EXCEPT a ring + # strictly BELOW the source that runs a DIFFERENT (newer) candidate (#1118 AC5): that ring's + # failures belong to the newer candidate's own gate, so counting them here would let a churning + # lower candidate block an older, independently-clean higher pair. In a single-candidate rollout + # no ring is below the source with a different commit, so this is byte-identical to the prior + # all-tiers scope; it diverges only when a newer candidate is soaking further down the pipeline. + # Two scopes (#1176 + CodeRabbit on #1220): a tier that does NOT yet run the candidate (the + # destination and beyond, tag provably on a different commit) has its failures attributed to the + # release that actually ran, so old-release failures there never block. Every other tier — the + # source tier and below, plus any tier whose commit is unknown — counts RAW, as before #1176: + # the source tier's sample still includes runs from before the candidate reached it, so dropping + # only their failures would let a candidate with no executions there reach PROMOTE. + local raw_repos=() attr_repos=() ch3 + for idx in "${!chan_array[@]}"; do + ch3="${chan_array[$idx]}" + # Exclude a lower ring only when it PROVABLY runs a different commit. An unresolvable (empty) + # tag is unknown, not "different": keep its failures in scope (fail closed). + local lc="${commit_array[$idx]:--}"; [ "$lc" = "-" ] && lc="" + if [ "$idx" -lt "$src_idx" ] && [ -n "$lc" ] && [ "$lc" != "$cand" ]; then + continue + fi + while IFS= read -r r; do + { [ -z "$r" ] || [ "$r" = '*' ]; } && continue + if [ -n "$lc" ] && [ "$lc" != "$cand" ]; then attr_repos+=("$r"); else raw_repos+=("$r"); fi + done < <(resolve_members "$agent" "$ch3") done - local cum_fail cum_startup cum_benign cum_suspect - read -r cum_fail cum_startup cum_benign cum_suspect < <(_cumulative_health "$agent" "$cut_z" "$differs" "${all_repos[@]}") + # A repo that also belongs to a raw tier (e.g. the host, in both next and ring0) stays raw. + local -A raw_set=(); local attr_only=() + for r in ${raw_repos[@]+"${raw_repos[@]}"}; do raw_set["$r"]=1; done + for r in ${attr_repos[@]+"${attr_repos[@]}"}; do [ -z "${raw_set["$r"]:-}" ] && attr_only+=("$r"); done + local cum_fail cum_startup cum_benign cum_suspect cum_stale + local a_fail a_startup a_benign a_suspect a_stale + read -r cum_fail cum_startup cum_benign cum_suspect cum_stale < <(_cumulative_health "$agent" "$cut_z" "$differs" "-" ${raw_repos[@]+"${raw_repos[@]}"}) + read -r a_fail a_startup a_benign a_suspect a_stale < <(_cumulative_health "$agent" "$cut_z" "$differs" "$cand" ${attr_only[@]+"${attr_only[@]}"}) + cum_fail=$(( cum_fail + a_fail )); cum_startup=$(( cum_startup + a_startup )) + cum_benign=$(( cum_benign + a_benign )); cum_suspect=$(( cum_suspect + a_suspect )) + cum_stale=$(( cum_stale + a_stale )) + # Target-ring health (#1176): failures from runs still on the previous release are reported, + # never counted — the candidate cannot have caused them and may be the fix. + if [ "${cum_stale:-0}" -gt 0 ]; then + echo "::notice::target-ring health ($agent $transition): ${cum_stale} failure(s) in runs still on the previous release — informational, not counted against candidate ${cand:0:12}." >&2 + fi # Per-transition knobs (registry-configurable; #548 defaults live in the ring SoT). local dwell_floor waived="false" target=0 @@ -1167,7 +1277,58 @@ _frontier_state() { fi fi - echo "$cand $frontier $transition $state $dwell_h $dwell_floor $sample $target $cum_fail $cum_startup $cum_benign $triage $mix_shift $downgrade $dg_cand_rate $dg_cand_sample $dg_base_rate $dg_base_sample" + echo "${cand:--} $frontier $transition $state $dwell_h $dwell_floor $sample $target $cum_fail $cum_startup $cum_benign $triage $mix_shift $downgrade $dg_cand_rate $dg_cand_sample $dg_base_rate $dg_base_sample" +} + +# _frontier_state — evaluate EVERY ring transition INDEPENDENTLY and echo ONE line per +# PENDING pair (#1118), each with the fields: +# " " +# is "-" when the source ring's commit is unresolvable (a blank leading field would be collapsed +# by `read` and shift every field); consumers must never treat "-" as a SHA. +# For each adjacent src->dst, the candidate is channel_commit(src); the pair is PENDING (and a +# line emitted, computed by _pair_state) iff dst is not already on that candidate. Several pairs +# may be in flight at once — e.g. a newer candidate soaking at next->ring0 while an older one +# holds at ring1->stable, the exact case that used to be dropped when only next's candidate was +# evaluated. A ring only ever advances to the commit on the ring directly below it (channel_commit +# of the source), so no tier can be skipped. When no pair is pending the agent is fully rolled out +# and a single COMPLETE line is emitted (frontier="-"), preserving the shape every consumer keys +# off. triage is "-" unless a pair's state is BLOCKED (then REGRESSION | PRE_EXISTING | SUSPECT); +# mix_shift is "SHIFT" for a gate.correctness decision-mix hold (#668 L2); downgrade is +# "DOWNGRADE" when a SUSPECT auto-downgraded to PRE_EXISTING (#668 inc6). +_frontier_state() { + local agent="$1" chans + chans="$(ordered_channels "$agent")" + local chan_array=() + IFS=, read -r -a chan_array <<< "$chans" + # Resolve every ring's commit ONCE (channel_commit is uncached across calls; see _pair_state). + local -a commits=() + local ch c commits_csv="" + for ch in "${chan_array[@]}"; do + c="$(channel_commit "$agent" "$ch")" + commits+=("$c") + commits_csv+="${c:--}," + done + commits_csv="${commits_csv%,}" + local i prev="" cand dstc transition emitted=0 + for i in "${!chan_array[@]}"; do + ch="${chan_array[$i]}" + if [ -n "$prev" ]; then + # Pending iff dst is not on src's commit. An UNRESOLVABLE src commit (empty) with a populated + # dst is still pending and _pair_state holds it BLOCKED (fail closed, as the single-frontier + # code did) — never silently skipped into a false COMPLETE. + cand="${commits[$((i-1))]}" + dstc="${commits[$i]}" + if [ "$dstc" != "$cand" ]; then + transition="${prev}->${ch}" + _pair_state "$agent" "$prev" "$ch" "$cand" "$transition" "$chans" "$commits_csv" + emitted=1 + fi + fi + prev="$ch" + done + if [ "$emitted" -eq 0 ]; then + echo "${commits[0]:--} - - COMPLETE 0 0 0 0 0 0 0 - -" + fi } cmd_evaluate() { @@ -1185,11 +1346,18 @@ cmd_evaluate() { local mark=" "; [ -n "$cand" ] && [ "$c" = "$cand" ] && mark="* " printf ' %s%-7s -> %s\n' "$mark" "${ch_tag#"$agent"/}" "${c:0:12}" done - read -r _cand frontier transition state dwell floor sample target cum_fail cum_startup cum_benign triage mix_shift downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample < <(_frontier_state "$agent") echo "----" - if [ "$frontier" = "-" ]; then - echo "frontier: none — fully rolled out (all rings on candidate)." - else + # Report EVERY pending pair independently (#1118): several transitions can be in flight at once. + local frontier transition state dwell floor sample target cum_fail cum_startup cum_benign + local triage mix_shift downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample _pcand + local any=0 + while read -r _pcand frontier transition state dwell floor sample target cum_fail cum_startup cum_benign triage mix_shift downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample; do + [ -z "$frontier" ] && continue + if [ "$frontier" = "-" ]; then + echo "frontier: none — fully rolled out (all rings on candidate)." + any=1; continue + fi + any=1 gate_summary_line "$transition" "$state" "$dwell" "$floor" "$sample" "$target" "$cum_fail" "$cum_startup" "$cum_benign" echo "decision for next ring '$frontier' [$transition]: $state" if [ "$state" = "BLOCKED" ]; then @@ -1209,6 +1377,9 @@ cmd_evaluate() { elif [ "$state" = "AWAITING_CONFIRMATION" ]; then echo "::notice::state=AWAITING_CONFIRMATION — reliability PASSED; holding for an opt-in human go/no-go at $transition (#668 Layer 3). Review the canary-confirm issue, then dispatch: promote $agent --confirm (not --override)." fi + done < <(_frontier_state "$agent") + if [ "$any" -eq 0 ]; then + echo "frontier: none — fully rolled out (all rings on candidate)." fi } @@ -1243,88 +1414,104 @@ cmd_promote() { *) echo "::error::unknown promote flag: $1" >&2; return 2 ;; esac; shift done - read -r cand frontier transition state _dwell _floor _sample _target cum_fail _cum_startup _cum_benign triage _mix_shift _dg _dgcr _dgcs _dgbr _dgbs < <(_frontier_state "$agent") - if [ "$frontier" = "-" ]; then - echo "nothing to promote — $agent is fully rolled out."; return 0 - fi - # allow_pre: advance a BLOCKED frontier ONLY when triage=PRE_EXISTING (never REGRESSION). - # Sourced from the per-reusable control block or the --allow-pre-existing flag (#1025 P2). + # Snapshot EVERY pending pair up front (#1118): each dst advances to the commit that was on its + # src BEFORE any move this run, so a ring can never skip a tier even when several pairs advance + # in one sweep, and pairs are decided INDEPENDENTLY — a BLOCKED lower pair never blocks a clean + # higher pair, and `--confirm` clears only an AWAITING_CONFIRMATION pair. + local -a _pairs=() + mapfile -t _pairs < <(_frontier_state "$agent") + # allow_pre: advance a BLOCKED pair ONLY when triage=PRE_EXISTING (never REGRESSION). Sourced + # from the per-reusable control block or the --allow-pre-existing flag (#1025 P2). Computed once. local allow_pre allow_pre="$(_jq -r --arg a "$agent" '.agents[$a].gate?.control?.allow_pre_existing // false')" [ "$allow_pre_flag" = true ] && allow_pre=true - # REGRESSION and SUSPECT both HALT + need a human: neither advances without --override, - # and --allow-pre-existing (which only unblocks PRE_EXISTING) never advances them. For a - # SUSPECT the human answers the class's discriminating question first — `--override` when - # the timeout is unrelated to the diff, or roll back when the candidate got materially - # slower (#668 increment 2). - if [ "$state" = "BLOCKED" ] && { [ "$triage" = "REGRESSION" ] || [ "$triage" = "SUSPECT" ]; } && [ "$override" != true ]; then - echo "::error::gate=BLOCKED (triage=$triage) for '$frontier' [$transition] — candidate regression suspected; not promoting. Investigate + rollback, do not --override blindly." - return 0 - fi - local advance=false - [ "$state" = "PROMOTE" ] && advance=true - [ "$override" = true ] && advance=true - [ "$state" = "BLOCKED" ] && [ "$triage" = "PRE_EXISTING" ] && [ "$allow_pre" = true ] && advance=true - # Layer 3 (#668 increment 3): a human --confirm advances an AWAITING_CONFIRMATION frontier - # (reliability is already PROMOTE — the state is only ever set from an otherwise-PROMOTE - # verdict). --confirm is NOT --override: it clears ONLY this state and can never advance a - # BLOCKED gate, so it cannot bypass reliability. - [ "$state" = "AWAITING_CONFIRMATION" ] && [ "$confirm" = true ] && advance=true - if [ "$advance" != true ]; then - if [ "$state" = "AWAITING_CONFIRMATION" ]; then - echo "gate=AWAITING_CONFIRMATION for ring '$frontier' [$transition] — reliability PASSED; holding for an opt-in human go/no-go. Review the canary-confirm issue, then dispatch: promote $agent --confirm (--confirm advances ONLY this reliability-clean state; it is NOT --override)." - else - echo "gate=$state for ring '$frontier' [$transition] (cum_fail=$cum_fail, triage=$triage) — not promoting. (use --override, or --allow-pre-existing for a PRE_EXISTING triage, after investigating)" - fi - return 0 - fi - if [ "$state" = "AWAITING_CONFIRMATION" ] && [ "$confirm" = true ]; then - echo "::notice::human confirmation received (--confirm) — advancing $agent/$frontier [$transition] past the confirmation go/no-go (reliability was already PROMOTE)." - elif [ "$state" != "PROMOTE" ]; then - echo "::warning::advancing $agent/$frontier despite gate state '$state' (triage=$triage)" - fi - # Consistent move (#1076): EVERY agent moves its channel tag via `gh api` on its HOST - # repo — never a local `git push`. A local force-push is NOT granted the release-manager - # App's ruleset bypass for a tag UPDATE, so it 013s on a protected channel tag such as - # dev-lead/next; the API path (same App token) IS honored as a bypass actor. host - # defaults to THIS_REPO for an agent whose registry entry omits it. + # Consistent move (#1076): EVERY agent moves its channel tag via `gh api` on its HOST repo — + # never a local `git push`. host defaults to THIS_REPO when the registry entry omits it. local host host="$(_jq -r --arg a "$agent" '.agents[$a].host // "" | tostring')" host="${host:-$THIS_REPO}" - # Move the RESOLVED frontier tag (major-scoped-channels epic #657, F4): advance the - # v-scoped `/v-` within its major line when it exists, else the legacy - # bare `/`. A v2 promotion never touches a v1 tag. The logical tier - # ($frontier) is unchanged — it still drives the gate + is reported as promoted_ring. - local frontier_tag; frontier_tag="$(_resolved_channel_tag "$agent" "$frontier")" - echo "advancing $frontier_tag -> ${cand:0:12} on $host" - if [ "$dry" = true ]; then - echo "[DRY-RUN] would: gh api PATCH repos/$host/git/refs/tags/$frontier_tag sha=$cand (force)" - return 0 - fi - _gh_move_tag "$host" "$frontier_tag" "$cand" \ - || { echo "::error::failed to move $frontier_tag -> ${cand:0:12} on $host" >&2 - # Persist this FAILED tag write (#1023 defect 2). It is UNEXPECTED — a permission/API - # rejection on the WRITE — distinct from an expected gate-block, which returns above - # before ever reaching the move. sync-promotion-failures turns a repeatedly-failing - # write into a durable, escalating blocker issue instead of a scrolling log line. - _log_promotion_failure "$agent" "$frontier" "$cand" "$host" "tag write rejected ($frontier_tag on $host)" - return 1; } - echo "promoted $frontier_tag -> ${cand:0:12}" - # Expose the move for the workflow's GitHub Deployment (traceability, #502). The - # deployment must be created on the repo that OWNS the moved commit: a cross-repo agent's - # candidate SHA lives on its host, NOT on THIS_REPO — creating the deployment against - # GITHUB_REPOSITORY 422s with "No ref found" (#1059). So emit the owning repo too. - local deploy_repo="$host" # the repo that OWNS the moved commit (#1059); host==THIS_REPO for this-repo agents - # GITHUB_OUTPUT is single-valued (last write wins), fine for a single `promote`. For - # `promote-all` (many promotions per run) the workflow reads CANARY_PROMOTIONS_LOG — one - # TSV line per promotion — so it can record a deployment for EVERY move, not just the last. - if [ -n "${GITHUB_OUTPUT:-}" ]; then - { echo "promoted_agent=$agent"; echo "promoted_ring=$frontier" - echo "promoted_sha=$cand"; echo "promoted_host=$deploy_repo"; } >> "$GITHUB_OUTPUT" - fi - if [ -n "${CANARY_PROMOTIONS_LOG:-}" ]; then - printf '%s\t%s\t%s\t%s\n' "$agent" "$frontier" "$cand" "$deploy_repo" >> "$CANARY_PROMOTIONS_LOG" - fi + local rc=0 pending=0 line first_pair=1 pair_override + local cand frontier transition state _dwell _floor _sample _target cum_fail _cum_startup _cum_benign triage _mix_shift _dg _dgcr _dgcs _dgbr _dgbs + for line in "${_pairs[@]}"; do + read -r cand frontier transition state _dwell _floor _sample _target cum_fail _cum_startup _cum_benign triage _mix_shift _dg _dgcr _dgcs _dgbr _dgbs <<< "$line" + { [ -z "$frontier" ] || [ "$frontier" = "-" ]; } && continue + pending=1 + # --override is an operator decision about ONE promotion — the lowest pending pair, i.e. the + # newest candidate, exactly what the single-frontier code overrode. It must NOT sweep every + # in-flight pair: that could push an AWAITING_CONFIRMATION pair for a different candidate past + # its human go/no-go (#1118 keeps that gate). Higher pairs are still decided by their own gates. + pair_override=false + if [ "$first_pair" -eq 1 ] && [ "$override" = true ]; then pair_override=true; fi + first_pair=0 + # An unresolvable candidate ("-") cannot be promoted, even with --override: there is no commit to + # move a tag to. It stays held BLOCKED (fail closed) until the source ring's tag resolves. + if [ "$cand" = "-" ]; then + echo "::error::gate=$state for '$frontier' [$transition] — the source ring's commit is unresolvable; not promoting (even with --override). Clears once the tag resolves." >&2 + continue + fi + # REGRESSION and SUSPECT both HALT + need a human: neither advances without --override, and + # --allow-pre-existing (which only unblocks PRE_EXISTING) never advances them. For a SUSPECT + # the human answers the class's discriminating question first (#668 increment 2). + if [ "$state" = "BLOCKED" ] && { [ "$triage" = "REGRESSION" ] || [ "$triage" = "SUSPECT" ]; } && [ "$pair_override" != true ]; then + echo "::error::gate=BLOCKED (triage=$triage) for '$frontier' [$transition] — candidate regression suspected; not promoting. Investigate + rollback, do not --override blindly." + continue + fi + local advance=false + [ "$state" = "PROMOTE" ] && advance=true + [ "$pair_override" = true ] && advance=true + [ "$state" = "BLOCKED" ] && [ "$triage" = "PRE_EXISTING" ] && [ "$allow_pre" = true ] && advance=true + # Layer 3 (#668 increment 3): a human --confirm advances an AWAITING_CONFIRMATION pair + # (reliability is already PROMOTE — the state is only ever set from an otherwise-PROMOTE + # verdict). --confirm is NOT --override: it clears ONLY this state and can never advance a + # BLOCKED gate, so it cannot bypass reliability. + [ "$state" = "AWAITING_CONFIRMATION" ] && [ "$confirm" = true ] && advance=true + if [ "$advance" != true ]; then + if [ "$state" = "AWAITING_CONFIRMATION" ]; then + echo "gate=AWAITING_CONFIRMATION for ring '$frontier' [$transition] — reliability PASSED; holding for an opt-in human go/no-go. Review the canary-confirm issue, then dispatch: promote $agent --confirm (--confirm advances ONLY this reliability-clean state; it is NOT --override)." + else + echo "gate=$state for ring '$frontier' [$transition] (cum_fail=$cum_fail, triage=$triage) — not promoting. (use --override, or --allow-pre-existing for a PRE_EXISTING triage, after investigating)" + fi + continue + fi + if [ "$state" = "AWAITING_CONFIRMATION" ] && [ "$confirm" = true ]; then + echo "::notice::human confirmation received (--confirm) — advancing $agent/$frontier [$transition] past the confirmation go/no-go (reliability was already PROMOTE)." + elif [ "$state" != "PROMOTE" ]; then + echo "::warning::advancing $agent/$frontier despite gate state '$state' (triage=$triage)" + fi + # Move the RESOLVED frontier tag (major-scoped-channels epic #657, F4): advance the v-scoped + # `/v-` within its major line when it exists, else the legacy bare + # `/`. The logical tier ($frontier) is unchanged — it still drives the gate + is + # reported as promoted_ring. + local frontier_tag; frontier_tag="$(_resolved_channel_tag "$agent" "$frontier")" + echo "advancing $frontier_tag -> ${cand:0:12} on $host" + if [ "$dry" = true ]; then + echo "[DRY-RUN] would: gh api PATCH repos/$host/git/refs/tags/$frontier_tag sha=$cand (force)" + continue + fi + if ! _gh_move_tag "$host" "$frontier_tag" "$cand"; then + echo "::error::failed to move $frontier_tag -> ${cand:0:12} on $host" >&2 + # Persist this FAILED tag write (#1023 defect 2). It is UNEXPECTED — a permission/API + # rejection on the WRITE — distinct from an expected gate-block, which never reaches the move. + _log_promotion_failure "$agent" "$frontier" "$cand" "$host" "tag write rejected ($frontier_tag on $host)" + rc=1 + continue + fi + echo "promoted $frontier_tag -> ${cand:0:12}" + # Expose the move for the workflow's GitHub Deployment (traceability, #502). The deployment must + # be created on the repo that OWNS the moved commit (#1059). GITHUB_OUTPUT is single-valued + # (last write wins); the workflow reads CANARY_PROMOTIONS_LOG — one TSV line per move — to + # record a deployment for EVERY promotion (a run may now advance several rings, #1118). + local deploy_repo="$host" + if [ -n "${GITHUB_OUTPUT:-}" ]; then + { echo "promoted_agent=$agent"; echo "promoted_ring=$frontier" + echo "promoted_sha=$cand"; echo "promoted_host=$deploy_repo"; } >> "$GITHUB_OUTPUT" + fi + if [ -n "${CANARY_PROMOTIONS_LOG:-}" ]; then + printf '%s\t%s\t%s\t%s\n' "$agent" "$frontier" "$cand" "$deploy_repo" >> "$CANARY_PROMOTIONS_LOG" + fi + done + [ "$pending" -eq 0 ] && echo "nothing to promote — $agent is fully rolled out." + return "$rc" } # _log_promotion_failure — append a FAILED tag-write to the @@ -1455,16 +1642,30 @@ _blocker_evidence() { while IFS= read -r r; do [ -n "$r" ] && [ "$r" != '*' ] && all+=("$r"); done < <(resolve_members "$agent" "$ch") done for r in "${all[@]}"; do case "$seen" in *" $r "*) ;; *) dedup+=("$r"); seen+="$r ";; esac; done + # Same attribution scope as _pair_state: only a repo that belongs solely to tiers provably NOT on + # the candidate has its old-release failures skipped; any repo in a tier on (or of unknown) commit + # lists every failure, so the evidence matches cum_fail. + local -A raw_evidence=(); local ev_c + for ch in "${chan_array[@]}"; do + ev_c="$(channel_commit "$agent" "$ch" || true)" + if [ -z "$ev_c" ] || [ "$ev_c" = "$cand" ]; then + while IFS= read -r r; do [ -n "$r" ] && raw_evidence["$r"]=1; done < <(resolve_members "$agent" "$ch") + fi + done for repo in "${dedup[@]}"; do json="$(_run_json "$repo" "$wf" "$cut_z")" local rid sig - while IFS= read -r rid; do + local concl + while IFS=$'\t' read -r rid concl; do [ -z "$rid" ] && continue + # Old release: not counted (#1176). Only `failure` runs are attributable — a startup_failure + # never ran a job, so it has no "Uses:" line and always counts (cum_startup), like the gate. + if [ "$concl" = "failure" ] && [ -z "${raw_evidence["$repo"]:-}" ] && _run_is_stale "$agent" "$cand" "$repo" "$rid"; then continue; fi if [ "$n" -ge 8 ]; then out+="- _(…more failing runs; truncated at 8)_"$'\n'; printf '%s' "$out"; return 0; fi sig="$(_run_signature "$repo" "$rid" | tr '\n' ';' | sed 's/;$//')" out+="- \`$repo\` — run [$rid](https://github.com/$repo/actions/runs/$rid); failed steps: ${sig:-unknown}"$'\n' n=$((n+1)) - done < <(jq -r '.[]?|select(.conclusion=="failure" or .conclusion=="startup_failure")|(.databaseId|tostring)' 2>/dev/null <<< "$json") + done < <(jq -r '.[]?|select(.conclusion=="failure" or .conclusion=="startup_failure")|[(.databaseId|tostring),.conclusion]|@tsv' 2>/dev/null <<< "$json") done [ -z "$out" ] && out="_(no failing runs in the per-candidate window — cum_fail may be startup_failures or a transient count)_"$'\n' printf '%s' "$out" @@ -1572,8 +1773,11 @@ EOF # _confirm_body — the body of the # evidence-carrying human-confirmation issue for an AWAITING_CONFIRMATION frontier (#668 Layer 3). # Reliability has PASSED; a human confirms the candidate is behaving CORRECTLY (not merely exiting -# green) before it reaches the stable tier (all consumers). Marker-keyed (canary-confirm:) -# so sync-issues upserts it idempotently and auto-closes it on promote/candidate change. +# green) before it reaches the stable tier (all consumers). Marker-keyed on the AGENT AND the +# ring1 CANDIDATE (canary-confirm::) so sync-issues upserts it idempotently, keeps it +# open across newer cuts landing on next/ring0 (the candidate is unchanged), and closes it only +# when stable advances or the ring1 candidate itself changes — a recut ring1 candidate never +# silently inherits a human's pending go/no-go for a different commit (#1118 AC3). _confirm_body() { local agent="$1" transition="$2" cand="$3" prior="$4" host="$5" sample="$6" target="$7" local suspect; suspect="$(_suspect_guidance "$agent")" @@ -1591,7 +1795,7 @@ $suspect" local display_prior="${prior:0:12}" display_prior="${display_prior:-none}" cat < + **Canary rollout — human confirmation requested (\`$transition\`).** The release gate holds \`$agent\` in **AWAITING_CONFIRMATION**: reliability has PASSED (dwell + sample + cumulative health all clean), but this transition is flagged \`require_confirmation\` (#668 Layer 3), so a human confirms the candidate is behaving *correctly* — not merely exiting green — before it reaches the stable tier (all consumers). Filed + maintained by the Canary Rollout workflow; **regenerated each run and auto-closes** when the promotion is confirmed or the candidate changes — do not edit by hand. @@ -1648,7 +1852,9 @@ EOF # FAIL CLOSED to BLOCKED so the tracked issue is upserted (never a green no-op). Emits the 18 # _frontier_state fields PLUS a trailing DATA-GAP flag (0=normal state resolved, 1=partial # run-history). Returns NON-ZERO only when even the candidate/frontier cannot be resolved — a -# TOTAL inability stays a hard error, surfaced by the caller. +# TOTAL inability stays a hard error, surfaced by the caller. Since #1118 _frontier_state emits +# one line per PENDING pair, so the wrapper appends the datagap flag to EACH line and, on a data +# gap, reconstructs one BLOCKED line per pending pair. _frontier_state_resilient() { local agent="$1" out="" rc=0 fetch_failed=0 # Arm a file flag that _run_json appends to on a sustained fetch failure. A file (not a shell @@ -1668,9 +1874,13 @@ _frontier_state_resilient() { [ -s "$flag" ] && fetch_failed=1 rm -f "$flag" fi - # Normal path (no fetch failure, a state line was produced): transparent pass-through, datagap=0. + # Normal path (no fetch failure, state lines produced): transparent pass-through, datagap=0 + # appended to EVERY pending-pair line (#1118). if [ "$fetch_failed" -eq 0 ] && [ -n "$out" ] && [ "$rc" -eq 0 ]; then - printf '%s 0\n' "$out" + local line + while IFS= read -r line; do + [ -n "$line" ] && printf '%s 0\n' "$line" + done <<< "$out" return 0 fi # Non-data-gap _frontier_state failure: the fetch succeeded but _frontier_state still failed — @@ -1679,28 +1889,53 @@ _frontier_state_resilient() { [ "$rc" -ne 0 ] && return "$rc" return 1 fi - # Data gap: the run-history fetch failed — reconstruct the tag-only facts (none read run history). - local cand chans frontier="" ch c transition prior differs triage - cand="$(channel_commit "$agent" next || true)" - chans="$(ordered_channels "$agent" || true)" + # Data gap: the run-history fetch failed — reconstruct the tag-only facts for EVERY pending pair + # (none read run history). Each is failed CLOSED to BLOCKED so its tracked issue is upserted. + local chans; chans="$(ordered_channels "$agent" || true)" local chan_array=() IFS=, read -r -a chan_array <<< "$chans" + local prev="" prev_commit="" ch ch_commit cand dstc transition prior differs triage emitted=0 for ch in "${chan_array[@]}"; do - c="$(channel_commit "$agent" "$ch" || true)" - if [ "$ch" = "next" ] || [ "$c" = "$cand" ]; then :; else frontier="$ch"; break; fi + ch_commit="$(channel_commit "$agent" "$ch" || true)" # once per ring (uncached across calls) + if [ -n "$prev" ]; then + cand="$prev_commit" + dstc="$ch_commit" + # Same pending rule as _frontier_state: dst not on src's commit, including an unresolvable + # (empty) src commit, which stays tracked as BLOCKED rather than vanishing (fail closed). + if [ "$dstc" != "$cand" ]; then + transition="${prev}->${ch}" + prior="$dstc" + differs="$(_reusable_differs "$agent" "$cand" "$prior")" + # classify_failure with an unknown category + no suspect signal: differs=1 → REGRESSION + # (fail closed — a changed reusable with UNREADABLE health is a suspected regression that + # needs a human), differs=0 → PRE_EXISTING (a byte-identical reusable cannot be a + # candidate regression). Either verdict still tracks the pair as BLOCKED. + triage="$(classify_failure "$differs" unknown 0)" + echo "${cand:--} $ch $transition BLOCKED 0 0 0 0 0 0 0 $triage - - 0 0 0 0 1" + emitted=1 + fi + fi + prev="$ch"; prev_commit="$ch_commit" done - # Total inability: cannot resolve even the candidate or a pending frontier → hard error. - # Fail closed (never a silent green); the caller surfaces ::error:: and ends non-zero. - { [ -z "$cand" ] || [ -z "$frontier" ]; } && return 1 - transition="$(transition_key "$frontier" "$chans")" - prior="$(channel_commit "$agent" "$frontier" || true)" - differs="$(_reusable_differs "$agent" "$cand" "$prior")" - # classify_failure with an unknown category + no suspect signal: differs=1 → REGRESSION - # (fail closed — a changed reusable with UNREADABLE health is treated as a suspected - # regression that needs a human), differs=0 → PRE_EXISTING (a byte-identical reusable cannot - # be a candidate regression). Either verdict still tracks the agent as BLOCKED. - triage="$(classify_failure "$differs" unknown 0)" - echo "$cand $frontier $transition BLOCKED 0 0 0 0 0 0 0 $triage - - 0 0 0 0 1" + # Total inability: cannot resolve even one pending pair → hard error. Fail closed (never a silent + # green); the caller surfaces ::error:: and ends non-zero. + [ "$emitted" -eq 0 ] && return 1 + return 0 +} + +# _confirm_issues_for_agent — "\t" per canary-confirm issue whose marker +# belongs to , matching BOTH the legacy agent-only marker (`canary-confirm:`) and +# the #1118 agent:candidate marker (`canary-confirm::`). Used to close any confirm +# issue that is no longer the current ring1 candidate's — so a recut ring1 candidate (or a +# no-longer-awaiting agent) never leaves a stale go/no-go issue open. Empty if none. +_confirm_issues_for_agent() { + local agent="$1" + gh issue list --repo "$ISSUE_REPO" --label canary-confirm --state all -L 100 \ + --json number,state,body 2>/dev/null \ + | jq -r --arg a "$agent" \ + '.[] | select((.body // "") | test("")) + | [(.number|tostring), (.state|ascii_upcase)] | @tsv' 2>/dev/null \ + || echo "" } # cmd_sync_issues [--dry-run] — upsert one blocker issue per BLOCKED agent, and render the @@ -1731,121 +1966,162 @@ cmd_sync_issues() { local rows="" agent hard_fail=0 while IFS= read -r agent; do [ -z "$agent" ] && continue - local cand frontier transition state _d _f _s _t cum_fail cum_startup _cb triage mix_shift host - local downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample datagap - # Resilient read (#820): the wrapper fails closed to a BLOCKED+datagap line on a run-history - # fetch outage, and returns non-zero (no output) only on a TOTAL inability to determine - # state. A failed `read` there means we could not even resolve the candidate/frontier — - # fail closed rather than report a false all-clear (a green no-op could mask a regression). - if ! read -r cand frontier transition state _d _f _s _t cum_fail cum_startup _cb triage mix_shift downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample datagap < <(_frontier_state_resilient "$agent"); then + # Resilient read (#820), now per PENDING pair (#1118): the wrapper fails closed to a + # BLOCKED+datagap line per pending pair on a run-history fetch outage, and returns non-zero + # (no output) only on a TOTAL inability to determine state. A non-zero here means we could not + # resolve even one pair — fail closed rather than report a false all-clear (a green no-op could + # mask a regression). + local pairs_out pairs_rc=0 + pairs_out="$(_frontier_state_resilient "$agent")" || pairs_rc=$? + if [ "$pairs_rc" -ne 0 ]; then echo "::error::sync-issues: cannot determine canary state for '$agent' (run-history AND tag resolution unavailable) — failing closed rather than reporting a false all-clear." >&2 hard_fail=1 rows+="| \`$agent\` | UNKNOWN | \`-\` | ? | ? | ⚠️ data unavailable (fail-closed) |"$'\n' continue fi - host="$(_agent_field "$agent" host)" - local blk="—" num_state num istate - # Best-effort (#1081): these substitutions call gh/jq. Under `set -euo pipefail` - # a bare assignment propagates a non-zero exit and would abort the whole step — - # before the fleet dashboard renders and before the intended fallback warnings - # below. `|| true` keeps sync-issues degrading gracefully (empty → handled). + local host; host="$(_agent_field "$agent" host)" + local cand frontier transition state _d _f _s _t cum_fail cum_startup _cb triage mix_shift + local downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample datagap + # Pass 1: select the blocker pair (first BLOCKED) and the confirm pair (first AWAITING). The two + # concerns are independent (#1118 AC5), so a BLOCKED lower pair and an AWAITING higher pair are + # tracked by SEPARATE issues in the same tick. + local have_blocked=0 bl_cand="" bl_transition="" bl_triage="-" bl_cum_fail=0 bl_cum_startup=0 + local bl_mix_shift="-" bl_downgrade="-" bl_dgcr=0 bl_dgcs=0 bl_dgbr=0 bl_dgbs=0 bl_datagap=0 + local have_awaiting=0 cf_cand="" cf_transition="" cf_sample=0 cf_target=0 + while read -r cand frontier transition state _d _f _s _t cum_fail cum_startup _cb triage mix_shift downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample datagap; do + { [ -z "$frontier" ] || [ "$frontier" = "-" ]; } && continue + if [ "$state" = "BLOCKED" ] && [ "$have_blocked" -eq 0 ]; then + have_blocked=1 + bl_cand="$cand"; bl_transition="$transition"; bl_triage="$triage" + bl_cum_fail="$cum_fail"; bl_cum_startup="$cum_startup"; bl_mix_shift="$mix_shift" + bl_downgrade="$downgrade"; bl_dgcr="$dg_cand_rate"; bl_dgcs="$dg_cand_sample" + bl_dgbr="$dg_base_rate"; bl_dgbs="$dg_base_sample"; bl_datagap="${datagap:-0}" + fi + if [ "$state" = "AWAITING_CONFIRMATION" ] && [ "$have_awaiting" -eq 0 ]; then + have_awaiting=1 + cf_cand="$cand"; cf_transition="$transition"; cf_sample="$_s"; cf_target="$_t" + fi + done <<< "$pairs_out" + + # Blocker issue (per agent, keyed canary-blocker:). Driven by the selected BLOCKED pair. + local bl_link="—" num_state num istate num_state="$(_issue_find canary-blocker "" || true)" num="${num_state%%$'\t'*}"; istate="${num_state##*$'\t'}" - if [ "$state" = "BLOCKED" ]; then + if [ "$have_blocked" -eq 1 ]; then local evidence body title mix_table="" - if [ "${datagap:-0}" = "1" ]; then - # Run history was unreadable this tick — there are no listable failing runs to cite. + if [ "$bl_datagap" = "1" ]; then evidence="_(⚠️ run-history fetch failed this tick — the failing runs could not be listed. The gate FAILS CLOSED: the promotion is held and this issue stays open until run history is readable again and the gate can re-evaluate.)_" else - evidence="$(_blocker_evidence "$agent" "$cand" || true)" + if [ "$bl_cand" = "-" ]; then + evidence="_(the source ring's commit is unresolvable this tick, so there is no candidate whose failing runs can be listed. The gate FAILS CLOSED and holds this pair until the tag resolves.)_" + else + evidence="$(_blocker_evidence "$agent" "$bl_cand" || true)" + fi fi - [ "$mix_shift" = "SHIFT" ] && mix_table="$(_decision_mix_table "$agent" "$cand" || true)" - body="$(_blocker_body "$agent" "$transition" "$cand" "$cum_fail" "$cum_startup" "$triage" "$host" "$evidence" "$mix_shift" "$mix_table" "$downgrade" "$dg_cand_rate" "$dg_cand_sample" "$dg_base_rate" "$dg_base_sample" "${datagap:-0}")" - if [ "$mix_shift" = "SHIFT" ]; then - title="Canary blocker: $agent $transition (decision-mix shift, SUSPECT)" - elif [ "${datagap:-0}" = "1" ]; then - title="Canary blocker: $agent $transition ($triage, partial run-history — fail-closed)" + [ "$bl_mix_shift" = "SHIFT" ] && mix_table="$(_decision_mix_table "$agent" "$bl_cand" "$bl_transition" || true)" + body="$(_blocker_body "$agent" "$bl_transition" "$bl_cand" "$bl_cum_fail" "$bl_cum_startup" "$bl_triage" "$host" "$evidence" "$bl_mix_shift" "$mix_table" "$bl_downgrade" "$bl_dgcr" "$bl_dgcs" "$bl_dgbr" "$bl_dgbs" "$bl_datagap")" + if [ "$bl_mix_shift" = "SHIFT" ]; then + title="Canary blocker: $agent $bl_transition (decision-mix shift, SUSPECT)" + elif [ "$bl_datagap" = "1" ]; then + title="Canary blocker: $agent $bl_transition ($bl_triage, partial run-history — fail-closed)" else - title="Canary blocker: $agent $transition (cum_fail=$cum_fail, $triage)" + title="Canary blocker: $agent $bl_transition (cum_fail=$bl_cum_fail, $bl_triage)" fi if [ -z "$num" ]; then - if [ "$dry" = true ]; then echo " [DRY] would OPEN blocker issue for $agent ($triage)"; blk="(new)"; else + if [ "$dry" = true ]; then echo " [DRY] would OPEN blocker issue for $agent ($bl_triage)"; bl_link="(new)"; else num="$(_gh_issue_create "$title" "$body" "canary-blocker" || true)" if [ -n "$num" ]; then gh issue edit "$num" --repo "$ISSUE_REPO" --add-label dev-lead >/dev/null 2>&1 || true - { [ "$triage" = "REGRESSION" ] || [ "$triage" = "SUSPECT" ]; } && gh issue edit "$num" --repo "$ISSUE_REPO" --add-label needs-human >/dev/null 2>&1 || true - echo " opened blocker issue #$num for $agent"; blk="#$num" + { [ "$bl_triage" = "REGRESSION" ] || [ "$bl_triage" = "SUSPECT" ]; } && gh issue edit "$num" --repo "$ISSUE_REPO" --add-label needs-human >/dev/null 2>&1 || true + echo " opened blocker issue #$num for $agent"; bl_link="#$num" else echo "::warning::could not open blocker issue for $agent (Issues:write on the App?)"; fi fi else - if [ "$dry" = true ]; then echo " [DRY] would UPDATE blocker issue #$num for $agent"; blk="#$num"; else + if [ "$dry" = true ]; then echo " [DRY] would UPDATE blocker issue #$num for $agent"; bl_link="#$num"; else [ "$istate" = "OPEN" ] || gh issue reopen "$num" --repo "$ISSUE_REPO" >/dev/null 2>&1 || true gh issue edit "$num" --repo "$ISSUE_REPO" --title "$title" --body "$body" >/dev/null 2>&1 \ || echo "::warning::could not update blocker issue #$num for $agent" gh issue edit "$num" --repo "$ISSUE_REPO" --add-label dev-lead >/dev/null 2>&1 || true - if [ "$triage" = "REGRESSION" ] || [ "$triage" = "SUSPECT" ]; then + if [ "$bl_triage" = "REGRESSION" ] || [ "$bl_triage" = "SUSPECT" ]; then gh issue edit "$num" --repo "$ISSUE_REPO" --add-label needs-human >/dev/null 2>&1 || true else gh issue edit "$num" --repo "$ISSUE_REPO" --remove-label needs-human >/dev/null 2>&1 || true fi - echo " updated blocker issue #$num for $agent"; blk="#$num" + echo " updated blocker issue #$num for $agent"; bl_link="#$num" fi fi else - # Not blocked — close a stale open blocker issue (the gate cleared). + # No BLOCKED pair — close a stale open blocker issue (the gate cleared). if [ -n "$num" ] && [ "$istate" = "OPEN" ]; then - if [ "$dry" = true ]; then echo " [DRY] would CLOSE cleared blocker issue #$num for $agent ($state)"; else + if [ "$dry" = true ]; then echo " [DRY] would CLOSE cleared blocker issue #$num for $agent"; else gh issue close "$num" --repo "$ISSUE_REPO" \ - --comment "✅ Gate cleared — \`$agent\` is now \`$state\`. Closed automatically by canary-rollout." >/dev/null 2>&1 || true + --comment "✅ Gate cleared — \`$agent\` is no longer BLOCKED. Closed automatically by canary-rollout." >/dev/null 2>&1 || true echo " closed cleared blocker issue #$num for $agent" fi - blk="#$num (closed)" + bl_link="#$num (closed)" fi fi - # Layer 3 (#668 increment 3): the human go/no-go issue for an AWAITING_CONFIRMATION frontier. - # A SEPARATE marker/label from the blocker issue (the two concerns are independent), upserted - # idempotently and auto-closed once the agent is no longer awaiting confirmation (promoted, - # rolled back, or a new candidate cut). Labelled needs-human — a human confirms, dev-lead does - # not action it. Best-effort under set -e, like the blocker path (`|| true`). - local cnum_state cnum cistate - cnum_state="$(_issue_find canary-confirm "" || true)" - cnum="${cnum_state%%$'\t'*}"; cistate="${cnum_state##*$'\t'}" - if [ "$state" = "AWAITING_CONFIRMATION" ]; then - local prior cbody ctitle - prior="$(channel_commit "$agent" "$frontier" || true)" - cbody="$(_confirm_body "$agent" "$transition" "$cand" "$prior" "$host" "$_s" "$_t" || true)" - ctitle="Canary confirm: $agent $transition — human go/no-go before stable" + # Layer 3 (#668 increment 3, #1118 AC3): the human go/no-go issue for an AWAITING_CONFIRMATION + # pair, keyed on the AGENT and the ring1 CANDIDATE (canary-confirm::). Keyed on the + # candidate so it PERSISTS across newer cuts landing on next/ring0 (the candidate is unchanged) + # and closes only when stable advances or the ring1 candidate itself changes. Labelled + # needs-human — a human confirms; dev-lead does not action it. + local cf_link="—" cnum_state cnum cistate keep_num="" + if [ "$have_awaiting" -eq 1 ]; then + local cf_dst prior cbody ctitle + cf_dst="${cf_transition##*->}" + prior="$(channel_commit "$agent" "$cf_dst" || true)" + cbody="$(_confirm_body "$agent" "$cf_transition" "$cf_cand" "$prior" "$host" "$cf_sample" "$cf_target" || true)" + ctitle="Canary confirm: $agent $cf_transition — human go/no-go before stable" + cnum_state="$(_issue_find canary-confirm "" || true)" + cnum="${cnum_state%%$'\t'*}"; cistate="${cnum_state##*$'\t'}" if [ -z "$cnum" ]; then - if [ "$dry" = true ]; then echo " [DRY] would OPEN confirm issue for $agent"; blk="(new confirm)"; else + if [ "$dry" = true ]; then echo " [DRY] would OPEN confirm issue for $agent"; cf_link="(new confirm)"; else cnum="$(_gh_issue_create "$ctitle" "$cbody" "canary-confirm" || true)" if [ -n "$cnum" ]; then gh issue edit "$cnum" --repo "$ISSUE_REPO" --add-label needs-human >/dev/null 2>&1 || true - echo " opened confirm issue #$cnum for $agent"; blk="#$cnum (confirm)" + echo " opened confirm issue #$cnum for $agent"; cf_link="#$cnum (confirm)" else echo "::warning::could not open confirm issue for $agent (Issues:write on the App?)"; fi fi else - if [ "$dry" = true ]; then echo " [DRY] would UPDATE confirm issue #$cnum for $agent"; blk="#$cnum (confirm)"; else + if [ "$dry" = true ]; then echo " [DRY] would UPDATE confirm issue #$cnum for $agent"; cf_link="#$cnum (confirm)"; else [ "$cistate" = "OPEN" ] || gh issue reopen "$cnum" --repo "$ISSUE_REPO" >/dev/null 2>&1 || true gh issue edit "$cnum" --repo "$ISSUE_REPO" --title "$ctitle" --body "$cbody" >/dev/null 2>&1 \ || echo "::warning::could not update confirm issue #$cnum for $agent" gh issue edit "$cnum" --repo "$ISSUE_REPO" --add-label needs-human >/dev/null 2>&1 || true - echo " updated confirm issue #$cnum for $agent"; blk="#$cnum (confirm)" + echo " updated confirm issue #$cnum for $agent"; cf_link="#$cnum (confirm)" fi fi - else - # No longer awaiting — close a stale open confirm issue (confirmed, rolled back, or recut). - if [ -n "$cnum" ] && [ "$cistate" = "OPEN" ]; then - if [ "$dry" = true ]; then echo " [DRY] would CLOSE cleared confirm issue #$cnum for $agent ($state)"; else - gh issue close "$cnum" --repo "$ISSUE_REPO" \ - --comment "✅ No longer awaiting confirmation — \`$agent\` is now \`$state\`. Closed automatically by canary-rollout." >/dev/null 2>&1 || true - echo " closed cleared confirm issue #$cnum for $agent" - fi - blk="#$cnum (confirm closed)" - fi + keep_num="$cnum" fi - - rows+="| \`$agent\` | $state | \`$transition\` | $cum_fail | $triage | $blk |"$'\n' + # Close any confirm issue for this agent that is NOT the current ring1 candidate's (a recut + # ring1 candidate, or an agent no longer awaiting at all) — a stale go/no-go must never linger. + local ci_num ci_state + while IFS=$'\t' read -r ci_num ci_state; do + [ -z "$ci_num" ] && continue + [ -n "$keep_num" ] && [ "$ci_num" = "$keep_num" ] && continue + [ "$ci_state" = "OPEN" ] || continue + if [ "$dry" = true ]; then echo " [DRY] would CLOSE cleared confirm issue #$ci_num for $agent"; else + gh issue close "$ci_num" --repo "$ISSUE_REPO" \ + --comment "✅ No longer awaiting confirmation for this candidate — \`$agent\` go/no-go is stale. Closed automatically by canary-rollout." >/dev/null 2>&1 || true + echo " closed cleared confirm issue #$ci_num for $agent" + fi + [ "$cf_link" = "—" ] && cf_link="#$ci_num (confirm closed)" + done < <(_confirm_issues_for_agent "$agent") + + # Pass 2: one dashboard row per pending pair (a COMPLETE agent gets a single COMPLETE row). + local rendered=0 disp_tr link + while read -r cand frontier transition state _d _f _s _t cum_fail cum_startup _cb triage mix_shift downgrade dg_cand_rate dg_cand_sample dg_base_rate dg_base_sample datagap; do + [ -z "$state" ] && continue + disp_tr="$transition"; [ "$frontier" = "-" ] && disp_tr="-" + link="—" + [ "$state" = "BLOCKED" ] && [ "$transition" = "$bl_transition" ] && link="$bl_link" + [ "$state" = "AWAITING_CONFIRMATION" ] && [ "$transition" = "$cf_transition" ] && link="$cf_link" + rows+="| \`$agent\` | $state | \`$disp_tr\` | ${cum_fail:-0} | ${triage:--} | $link |"$'\n' + rendered=1 + done <<< "$pairs_out" + [ "$rendered" -eq 0 ] && rows+="| \`$agent\` | COMPLETE | \`-\` | 0 | - | — |"$'\n' done <<< "$agents" # Render the fleet-status table into the run's job summary (a read-only snapshot, not a @@ -1949,16 +2225,16 @@ cmd_sync_promotion_failures() { echo "::error::could not list canary-promotion-failure issues on $ISSUE_REPO — aborting to avoid creating duplicate tracking issues." >&2 return 1 fi - local -A failed_map=() ok_map=() - local fa - while IFS= read -r fa; do - [ -n "$fa" ] && failed_map["$fa"]=1 - done < <(printf '%s\n' "$failed_agents") - local oa - while IFS= read -r oa; do - [ -n "$oa" ] && ok_map["$oa"]=1 - done < <(printf '%s\n' "$ok_agents") - local -A promo_failures=() + # A successful write cancels a failure only for the SAME ring: one run can now advance several + # rings (#1118), so a ring0 success must not hide a ring1 failure for the same agent. + local -A ok_ring=() + if [ -n "${CANARY_PROMOTIONS_LOG:-}" ] && [ -s "$CANARY_PROMOTIONS_LOG" ]; then + local ok_a ok_r + while IFS=$'\t' read -r ok_a ok_r _; do + [ -n "$ok_a" ] && ok_ring["$ok_a|$ok_r"]=1 + done < "$CANARY_PROMOTIONS_LOG" + fi + local -A promo_failures=() unresolved=() local _pf_agent _pf_num _pf_state _pf_count while IFS=$'\t' read -r _pf_agent _pf_num _pf_state _pf_count; do [ -n "$_pf_agent" ] && promo_failures["$_pf_agent"]="${_pf_num}"$'\t'"${_pf_state}"$'\t'"${_pf_count}" @@ -1969,6 +2245,9 @@ cmd_sync_promotion_failures() { local lf_agent lf_ring lf_cand lf_host lf_reason while IFS=$'\t' read -r lf_agent lf_ring lf_cand lf_host lf_reason; do if [ -n "$lf_agent" ]; then + # Superseded by a successful write of the same ring this run (a retry) → not a failure. + [ -n "${ok_ring["$lf_agent|$lf_ring"]:-}" ] && continue + unresolved["$lf_agent"]=1 fail_ring["$lf_agent"]="$lf_ring" fail_cand["$lf_agent"]="$lf_cand" fail_host["$lf_agent"]="$lf_host" @@ -1980,7 +2259,7 @@ cmd_sync_promotion_failures() { while IFS= read -r agent; do [ -z "$agent" ] && continue local outcome="ok" - [ -n "${failed_map["$agent"]:-}" ] && [ -z "${ok_map["$agent"]:-}" ] && outcome="failed" + [ -n "${unresolved["$agent"]:-}" ] && outcome="failed" local ns num istate prior ns="${promo_failures["$agent"]:-}" num="$(printf '%s' "$ns" | cut -f1)"; istate="$(printf '%s' "$ns" | cut -f2)"; prior="$(printf '%s' "$ns" | cut -f3)" diff --git a/tests/canary_rollout.bats b/tests/canary_rollout.bats index 2ba7ad3c..851c96c1 100644 --- a/tests/canary_rollout.bats +++ b/tests/canary_rollout.bats @@ -785,7 +785,7 @@ GHEOF printf '#!/usr/bin/env bash\necho "gh: HTTP 503: server error" >&2\nexit 1\n' > "$STUB_BIN/gh" chmod +x "$STUB_BIN/gh" run env CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 \ - bash -c "source '$ORCH' && _cumulative_health dev-lead '' 0 some/repo" + bash -c "source '$ORCH' && _cumulative_health dev-lead '' 0 - some/repo" # Fail-closed path returns non-zero, but NEVER with an arithmetic syntax error. [[ "$output" != *"syntax error"* ]] [[ "$output" != *"operand expected"* ]] @@ -1193,6 +1193,146 @@ GITEOF [[ "$output" == *"PRE_EXISTING"* ]] } +# ── orchestrator: failures from runs still on the OLD release don't block (#1176) ── +# next = candidate (cccc), every ring = prior (bbbb); all tier repos return 3 failed runs. +# `gh run view --log` prints the resolved "Uses: …@refs/tags/ ()" line, with +# = $1 (the release the run actually executed); "none" prints a log with no Uses: line. +_executed_sha_stub() { + local executed="$1" conclusion="${2:-failure}" fail_sub="${3:-repo petry-projects/TalkTerm --workflow}" cut_iso run_iso + STUB_BIN="$(mktemp -d "$BATS_TEST_TMPDIR/stub.XXXXXX")"; export PATH="$STUB_BIN:$PATH" + cut_iso="$(date -u -d "-3 days" +%Y-%m-%dT%H:%M:%SZ 2>/dev/null || date -u -v"-3d" +%Y-%m-%dT%H:%M:%SZ 2>/dev/null)" + run_iso="$(date -u -d "-2 days" +%Y-%m-%dT%H:%M:%SZ 2>/dev/null || date -u -v"-2d" +%Y-%m-%dT%H:%M:%SZ 2>/dev/null)" + cat > "$STUB_BIN/gh" < (default: TalkTerm, a ring1 member = the DESTINATION side of next->ring0) + # fails; every other repo is green — mirrors #1176's real case. + __c=success; case "\$*" in *"$fail_sub"*) __c="$conclusion" ;; esac + jq -nc --arg d "$run_iso" --arg c "\$__c" '[range(3)|{conclusion:\$c,createdAt:\$d,databaseId:(1000+.),workflowName:"Dev-Lead Agent"}]' ;; + *"run view"*"--log"*) + # Mirrors real \`gh run view --log\` output: " ". + ts="2026-09-25T00:00:00.1234567Z" + U="Uses: petry-projects/.github-private/.github/workflows/dev-lead-reusable.yml@refs/tags/dev-lead/v2-ring1" + if [ "$executed" = logfail ]; then exit 1 + elif [ "$executed" = none ]; then printf 'build\tUNKNOWN STEP\t%s nothing relevant\n' "\$ts" + elif [ "$executed" = collision ]; then printf 'build\tUNKNOWN STEP\t%s Uses: someone/else/.github/workflows/dev-lead-reusable.yml@refs/tags/x (bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb)\n' "\$ts" + elif [ "$executed" = forgedcand ]; then + # genuine OLD line first, then the same job echoes a forged CANDIDATE-sha line + printf 'build\tUNKNOWN STEP\t%s %s (bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb)\n' "\$ts" "\$U" + printf 'build\tUNKNOWN STEP\t%s %s (cccccccccccccccccccccccccccccccccccccccc)\n' "\$ts" "\$U" + elif [ "$executed" = forged ]; then + # genuine candidate line first, then the same job echoes a forged OLD-sha line + printf 'build\tUNKNOWN STEP\t%s %s (cccccccccccccccccccccccccccccccccccccccc)\n' "\$ts" "\$U" + printf 'build\tUNKNOWN STEP\t%s %s (bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb)\n' "\$ts" "\$U" + else + n=0; for sha in $executed; do n=\$((n+1)); printf 'job%s\tUNKNOWN STEP\t%s %s (%s)\n' "\$n" "\$ts" "\$U" "\$sha"; done + fi ;; + *"run view"*) echo '{"jobs":[{"steps":[{"name":"Compile TypeScript","conclusion":"failure"}]}]}' ;; + *) echo "{}" ;; +esac +GHEOF + chmod +x "$STUB_BIN/gh" + cat > "$STUB_BIN/git" <<'GITEOF' +#!/usr/bin/env bash +: +GITEOF + chmod +x "$STUB_BIN/git" +} + +@test "orchestrator: failures from runs on the OLD release do not block the candidate (#1176)" { + _executed_sha_stub bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" != *"BLOCKED"* ]] + [[ "$output" == *"target-ring health"* ]] + [[ "$output" == *"informational"* ]] +} + +@test "orchestrator: failures from runs that executed the candidate SHA still BLOCK (#1176)" { + _executed_sha_stub cccccccccccccccccccccccccccccccccccccccc + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] + [[ "$output" != *"target-ring health"* ]] +} + +@test "orchestrator: a run that called the reusable at the old AND candidate SHA still BLOCKS (#1176)" { + _executed_sha_stub "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb cccccccccccccccccccccccccccccccccccccccc" + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] +} + +@test "orchestrator: a Uses: line for a different host's same-named workflow is not attributed — BLOCKS (#1176)" { + _executed_sha_stub collision + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] +} + +@test "orchestrator: a forged old-SHA Uses: line echoed after the genuine one is ignored — still BLOCKS (#1176)" { + _executed_sha_stub forged + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] +} + +@test "orchestrator: only the FIRST Uses: line per job is trusted — a later forged candidate-SHA line cannot re-block an old-release run (#1176)" { + _executed_sha_stub forgedcand + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" != *"BLOCKED"* ]] + [[ "$output" == *"target-ring health"* ]] +} + +@test "orchestrator: old-release failures on the SOURCE tier still BLOCK — attribution only applies to tiers not yet on the candidate (#1176)" { + # The source tier's sample still counts runs from before the candidate reached it, so dropping only + # their failures would let a candidate with no executions there reach PROMOTE. The host (.github-private) + # is the source tier of next->ring0 and its failing run's log names the OLD sha: it must still count. + _executed_sha_stub bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb failure "repo petry-projects/.github-private --workflow" + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] + [[ "$output" != *"target-ring health"* ]] +} + +@test "orchestrator: a short (7-char) candidate SHA in the Uses: line is matched by prefix and BLOCKS (#1176)" { + _executed_sha_stub ccccccc + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] +} + +@test "orchestrator: a failing 'gh run view --log' lookup fails closed and BLOCKS (#1176)" { + _executed_sha_stub logfail + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] +} + +@test "orchestrator: startup_failure runs are never attributed to an old release — they still BLOCK (#1176)" { + _executed_sha_stub bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb startup_failure + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] +} + +@test "orchestrator: a failure whose executed release cannot be determined fails closed and BLOCKS (#1176)" { + _executed_sha_stub none + run env CANARY_RINGS="$RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"BLOCKED"* ]] +} + # ── orchestrator: promote --allow-pre-existing (control override, #1025 P2) ───── @test "orchestrator: promote --allow-pre-existing advances a BLOCKED+PRE_EXISTING frontier (dry-run)" { _graduated_stub 3 2 failure 0 @@ -1830,6 +1970,18 @@ GHEOF grep -q -- "issue edit 901 .*--add-label dev-lead --add-label needs-human" "$ISSUE_LOG" } +@test "orchestrator: sync-promotion-failures — a successful ring move does not mask a failed write for the same agent (#1023)" { + local existing='[{"number":903,"state":"OPEN","body":"\n"}]' + _promo_fail_sync_stub "$existing" + local flog="$BATS_TEST_TMPDIR/pf.tsv" slog="$BATS_TEST_TMPDIR/ok.tsv" + printf 'dev-lead\tstable\tccccccccccccccccc\tpetry-projects/.github-private\ttag write rejected\n' > "$flog" + printf 'dev-lead\tring0\tccccccccccccccccc\tpetry-projects/.github-private\n' > "$slog" + run env ISSUE_REPO="petry-projects/.github" CANARY_PROMOTION_FAILURE_ESCALATE_AFTER=2 CANARY_PROMOTIONS_FAILED_LOG="$flog" CANARY_PROMOTIONS_LOG="$slog" bash "$ORCH" sync-promotion-failures + [ "$status" -eq 0 ] + [[ "$output" == *"updated promotion-failure issue #903 for dev-lead (count=2)"* ]] + ! grep -q "^CLOSE|" "$ISSUE_LOG" +} + @test "orchestrator: sync-promotion-failures auto-closes the tracking issue when the write recovers (#1023)" { # dev-lead succeeded this run (in the SUCCESS log) but an OPEN failure issue exists → close it. local existing='[{"number":902,"state":"OPEN","body":"\n"}]' @@ -1854,6 +2006,21 @@ GHEOF grep -q "CLOSE|.*903" "$ISSUE_LOG" } +@test "orchestrator: sync-promotion-failures — a ring0 success must NOT hide a ring1 failure from the same run (#1118)" { + # One run can now advance several rings. ring0 moved, ring1's tag write was rejected: the agent is + # FAILED (its tracking issue stays/gets opened), not "recovered". + local existing='[{"number":905,"state":"OPEN","body":"\n"}]' + _promo_fail_sync_stub "$existing" + local slog="$BATS_TEST_TMPDIR/ok.tsv"; printf 'dev-lead\tring0\tdddddddddddddddd\tpetry-projects/.github-private\n' > "$slog" + local flog="$BATS_TEST_TMPDIR/pf.tsv"; printf 'dev-lead\tring1\tccccccccccccccccc\tpetry-projects/.github-private\ttag write rejected\n' > "$flog" + run env ISSUE_REPO="petry-projects/.github" CANARY_PROMOTIONS_LOG="$slog" CANARY_PROMOTIONS_FAILED_LOG="$flog" bash "$ORCH" sync-promotion-failures + [ "$status" -eq 0 ] + [[ "$output" == *"updated promotion-failure issue #905 for dev-lead (count=2)"* ]] + [[ "$output" != *"closed recovered"* ]] + run grep -q "CLOSE|.*905" "$ISSUE_LOG" + [ "$status" -eq 1 ] +} + @test "orchestrator: sync-promotion-failures does NOT seed a new streak from a CLOSED issue (#1023)" { # A CLOSED tracking issue with a high prior count must not seed the next streak: # the first new failure after recovery should start at count=1 (not count=6). @@ -3302,6 +3469,210 @@ GHEOF grep -q "CLOSE|.*502" "$ISSUE_LOG" } +# ── #1118: independent per-pair evaluation — a ring1->stable hold survives newer cuts ── +# Before #1118 `_frontier_state` evaluated ONLY next's candidate at a SINGLE frontier: once +# autocut moved `next`, the frontier fell back to ring0 and a pending ring1->stable +# AWAITING_CONFIRMATION hold vanished (its canary-confirm issue auto-closed, unconfirmed). Now +# EVERY adjacent src->dst pair is evaluated independently on the commit currently sitting on +# `src`, so several transitions can be in flight at once and an older ring1 candidate keeps its +# human go/no-go regardless of newer cuts landing on next/ring0. +# +# _multicand_stub [fail_sub] [differ] [issue_list] +# t_* ∈ {C1,C2,C3,PRIOR} — the commit each tier carries (C1=cccc…, C2=dddd…, C3=eeee…, +# PRIOR=bbbb…). off_cN — cut age (a `date -d` offset string, e.g. "1 hours"/"3 days") of the +# release tag for C1/C2/C3. fail_sub — a repo substring whose `run list` returns FAILURES +# (default: none → all clean). differ="C1" makes C1's reusable blob differ from the prior +# (default: all identical → differs=0). issue_list — JSON returned by `gh issue list`. +# dev-lead is cross-repo (GITHUB_REPOSITORY=.github forces THIS_REPO=.github): all tag/blob/ +# release resolution goes via gh api. Sets MC_RINGS (dev-lead-only registry); logs issue ops +# to ISSUE_LOG. +_multicand_stub() { + local t_next="$1" t_ring0="$2" t_ring1="$3" t_stable="$4" + local off1="$5" off2="$6" off3="$7" fail_sub="${8:-__NEVER_FAIL__}" differ="${9:-}" issue_list="${10:-[]}" + local C1="cccccccccccccccccccccccccccccccccccccccc" + local C2="dddddddddddddddddddddddddddddddddddddddd" + local C3="eeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeeee" + local PRIOR="bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb" + local -A M=( [C1]="$C1" [C2]="$C2" [C3]="$C3" [PRIOR]="$PRIOR" [NONE]="" ) + local n="${M[$t_next]}" r0="${M[$t_ring0]}" r1="${M[$t_ring1]}" st="${M[$t_stable]}" + local blob_c1="reuseSAME"; [ "$differ" = "C1" ] && blob_c1="reuseCAND" + STUB_BIN="$(mktemp -d "$BATS_TEST_TMPDIR/stub.XXXXXX")"; export PATH="$STUB_BIN:$PATH" + export ISSUE_LOG="$STUB_BIN/issue.log"; : > "$ISSUE_LOG" + local c1_iso c2_iso c3_iso run_iso + c1_iso="$(date -u -d "-$off1" +%Y-%m-%dT%H:%M:%SZ)" + c2_iso="$(date -u -d "-$off2" +%Y-%m-%dT%H:%M:%SZ)" + c3_iso="$(date -u -d "-$off3" +%Y-%m-%dT%H:%M:%SZ)" + run_iso="$(date -u -d '-12 hours' +%Y-%m-%dT%H:%M:%SZ)" + cat > "$STUB_BIN/git" <<'GITEOF' +#!/usr/bin/env bash +: # dev-lead is cross-repo; all tag/blob/release resolution goes via gh api +GITEOF + chmod +x "$STUB_BIN/git" + cat > "$STUB_BIN/gh" <> "$ISSUE_LOG"; echo "https://github.com/petry-projects/.github-private/issues/909" ;; + "issue edit"*) echo "EDIT|\$*" >> "$ISSUE_LOG" ;; + "issue close"*) echo "CLOSE|\$*" >> "$ISSUE_LOG" ;; + "issue reopen"*) echo "REOPEN|\$*" >> "$ISSUE_LOG" ;; + "label create"*) : ;; + *) echo "{}" ;; +esac +GHEOF + chmod +x "$STUB_BIN/gh" + MC_RINGS="$BATS_TEST_TMPDIR/mc-rings.json" + jq '{org_infra_repos, agents: {"dev-lead": .agents["dev-lead"]}}' "$RINGS" > "$MC_RINGS" +} + +@test "#1118: a newer next cut does NOT drop a pending ring1->stable AWAITING_CONFIRMATION hold" { + # next=C1 cut 1h ago (next->ring0 SOAKING), ring0=ring1=C3 cut 30h ago, stable=prior. + # ring1->stable must STILL be evaluated as AWAITING_CONFIRMATION even though `next` moved. + _multicand_stub C1 C3 C3 PRIOR "1 hours" "3 days" "30 hours" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"ring1->stable"* ]] + [[ "$output" == *"AWAITING_CONFIRMATION"* ]] + # the newer next candidate is independently in flight, still soaking (it does NOT cancel the hold) + [[ "$output" == *"next->ring0"* ]] + [[ "$output" == *"SOAKING"* ]] +} + +@test "#1118: sync-issues KEEPS (updates, not closes) the ring1-candidate confirm issue across a newer next cut" { + # An OPEN canary-confirm issue already exists keyed on ring1's candidate C3. A newer next cut + # (C1) must NOT close it — the hold is keyed on the ring1 candidate, which is unchanged (AC3/AC7a). + _multicand_stub C1 C3 C3 PRIOR "1 hours" "3 days" "30 hours" "" "" \ + '[{"number":901,"state":"OPEN","body":""}]' + local summ="$BATS_TEST_TMPDIR/mc-a.md"; : > "$summ" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" ISSUE_REPO="petry-projects/.github-private" GITHUB_STEP_SUMMARY="$summ" bash "$ORCH" sync-issues + [ "$status" -eq 0 ] + grep -q "EDIT|.*901" "$ISSUE_LOG" # the C3 confirm issue is refreshed, not closed + run grep -q "CLOSE|.*901" "$ISSUE_LOG" # the hold survived the newer next cut + [ "$status" -eq 1 ] # (exactly 1: a missing log, status 2, must not pass) + grep -q "AWAITING_CONFIRMATION" "$summ" +} + +@test "#1118: promote --confirm advances stable to ring1's candidate while next->ring0 keeps soaking" { + # next=C1 (soaking), ring0=ring1=C3 (AWAITING at ring1->stable). --confirm clears ONLY the + # ring1->stable hold; the newer, still-soaking next candidate is untouched (AC4/AC7b). + _multicand_stub C1 C3 C3 PRIOR "1 hours" "3 days" "30 hours" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" bash "$ORCH" promote dev-lead --confirm --dry-run + [ "$status" -eq 0 ] + # stable advances to ring1's candidate (C3 = eeee…), confirmed + [[ "$output" == *"tags/dev-lead/stable sha=eeeeeeeeeeee"* ]] + # next->ring0 (the newer candidate) is still soaking — never advanced by --confirm + [[ "$output" == *"SOAKING"* ]] + [[ "$output" != *"tags/dev-lead/ring0 sha="* ]] +} + +@test "#1118: no tier skip — ring1 only ever advances to the commit currently on ring0" { + # next=C1, ring0=C2, ring1=C3, stable=prior; every pair clean and old enough to PROMOTE. + _multicand_stub C1 C2 C3 PRIOR "3 days" "40 hours" "30 hours" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" bash "$ORCH" promote dev-lead --dry-run + [ "$status" -eq 0 ] + # ring0 advances to next's commit (C1 = cccc…) + [[ "$output" == *"tags/dev-lead/ring0 sha=cccccccccccc"* ]] + # ring1 advances to ring0's OLD commit (C2 = dddd…) — NEVER skips to next's (C1) + [[ "$output" == *"tags/dev-lead/ring1 sha=dddddddddddd"* ]] + [[ "$output" != *"tags/dev-lead/ring1 sha=cccc"* ]] + # ring1->stable holds for human confirmation (require_confirmation; no --confirm passed) + [[ "$output" == *"AWAITING_CONFIRMATION"* ]] +} + +@test "#1118: a BLOCKED lower pair does not block an independently clean higher pair" { + # next=C1 with a REGRESSION on the next tier (reusable differs + a failure there); + # ring0=ring1=C3 independently clean → ring1->stable is AWAITING, unblocked by the lower pair. + _multicand_stub C1 C3 C3 PRIOR "3 days" "3 days" "30 hours" "petry-projects/.github-private" C1 + local summ="$BATS_TEST_TMPDIR/mc-d.md"; : > "$summ" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" ISSUE_REPO="petry-projects/.github-private" GITHUB_STEP_SUMMARY="$summ" bash "$ORCH" sync-issues + [ "$status" -eq 0 ] + # lower pair blocked → its canary-blocker issue opens (REGRESSION at next->ring0) + [[ "$output" == *"opened blocker issue"* ]] + grep -q "REGRESSION" "$summ" + # higher pair independently clean → its canary-confirm issue opens (AWAITING at ring1->stable) + [[ "$output" == *"confirm issue"* ]] + grep -q "AWAITING_CONFIRMATION" "$summ" +} + +@test "#1118: --override advances ONLY the lowest pending pair — never an AWAITING_CONFIRMATION pair for another candidate" { + # next=C1 BLOCKED (REGRESSION), ring0=ring1=C3 AWAITING at ring1->stable. An operator --override of + # the blocked newest candidate must not also push stable past the human go/no-go (#668 L3 stays). + _multicand_stub C1 C3 C3 PRIOR "3 days" "3 days" "30 hours" "petry-projects/.github-private" C1 + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" bash "$ORCH" promote dev-lead --override --dry-run + [ "$status" -eq 0 ] + [[ "$output" == *"tags/dev-lead/ring0 sha=cccccccccccc"* ]] # the overridden lowest pair advances + [[ "$output" != *"tags/dev-lead/stable sha="* ]] # the higher pair keeps its own gate + [[ "$output" == *"AWAITING_CONFIRMATION"* ]] +} + +@test "#1118: an UNRESOLVABLE source commit with a populated destination fails closed (BLOCKED), never a false COMPLETE" { + # next has no resolvable tag but ring0 does: next->ring0 is pending and must hold BLOCKED. + _multicand_stub NONE C3 C3 PRIOR "1 hours" "3 days" "30 hours" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"next->ring0"* ]] + [[ "$output" == *"BLOCKED"* ]] + [[ "$output" != *"fully rolled out"* ]] +} + +@test "#1118: an unresolvable source commit is a held pair that sync-issues SEES (sentinel, not a shifted blank field)" { + # next has no resolvable tag: the pair record must still parse (a leading blank field would shift + # every field, hiding BLOCKED), so sync-issues opens its blocker issue. + _multicand_stub NONE C3 C3 PRIOR "1 hours" "3 days" "30 hours" + local summ="$BATS_TEST_TMPDIR/mc-n.md"; : > "$summ" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" ISSUE_REPO="petry-projects/.github-private" GITHUB_STEP_SUMMARY="$summ" bash "$ORCH" sync-issues + [ "$status" -eq 0 ] + [[ "$output" == *"opened blocker issue"* ]] + grep -q "BLOCKED" "$summ" +} + +@test "#1118: promote --override never acts on an unresolvable candidate (no tag is moved to a sentinel)" { + _multicand_stub NONE C3 C3 PRIOR "1 hours" "3 days" "30 hours" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" bash "$ORCH" promote dev-lead --override --dry-run + [ "$status" -eq 0 ] + [[ "$output" == *"unresolvable"* ]] + [[ "$output" != *"tags/dev-lead/ring0 sha=-"* ]] + [[ "$output" != *"advancing dev-lead/ring0 -> -"* ]] +} + +@test "#1118: a lower ring whose tag is UNRESOLVABLE stays in scope — its failures still block a higher pair (fail closed)" { + # ring0 has no resolvable tag and its member (.github) is failing; ring1->stable must not shed + # those failures just because ring0's commit is unknown rather than provably different. + _multicand_stub C1 NONE C3 PRIOR "3 days" "3 days" "30 hours" "repo petry-projects/.github --workflow" + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$MC_RINGS" bash "$ORCH" evaluate dev-lead + [ "$status" -eq 0 ] + [[ "$output" == *"[ring1->stable]: BLOCKED"* ]] +} + +# The confirm issue's idempotency marker is keyed on the ring1 CANDIDATE (#1118 AC3), so a +# recut ring1 candidate does not silently transfer a human's pending go/no-go to a new commit. +@test "#1118: _confirm_body keys the canary-confirm marker on agent AND candidate" { + run env GITHUB_REPOSITORY="petry-projects/.github" CANARY_RINGS="$RINGS" bash -c \ + "source '$ORCH' && _confirm_body dev-lead 'ring1->stable' cafe1234cafe bbbbbbbbbbbb petry-projects/.github-private 5 1" + [ "$status" -eq 0 ] + [[ "$output" == *""* ]] +} + # ── #668 increment 4 (Layer 2): decision telemetry — pure core + engine overlay ── # decision_class(): the taken `decision: ` no-op step (skipped branches ignored, prefix # stripped) off a `gh run view --json jobs` payload. decide_decision_shift(): the pure gate over