diff --git a/scripts/canary-rollout.sh b/scripts/canary-rollout.sh index 6803f7e6..c9c1da6b 100755 --- a/scripts/canary-rollout.sh +++ b/scripts/canary-rollout.sh @@ -595,8 +595,13 @@ _gh_retry_after() { # populated in one call would not survive to the next. When the dir is unset (a unit test # calling _run_json directly), every call fetches. Tunable via CANARY_GH_RETRIES / # CANARY_GH_RETRY_SLEEP (tests set 0). CANARY_GH_RETRY_AFTER_CAP caps server hints (default 900s). +# +# Optional 3rd arg `strict` (#1224): when "1", a not-found workflow returns [] with exit 3 +# instead of 0, so the caller can tell "this repo has no such workflow" from "it has the +# workflow but no runs" — the signal that a repo may be ADR-0007-collapsed. The not-found +# verdict is cached alongside the [] (a `.nf` marker) so a strict cache hit still reports it. _repo_wf_runs_cached() { - local repo="$1" wf="$2" out err summary ra delay span expo errfile cachef key keyhash + local repo="$1" wf="$2" strict="${3:-0}" limit="${4:-1000}" out err summary ra delay span expo errfile cachef key keyhash local attempts="${CANARY_GH_RETRIES:-6}" base="${CANARY_GH_RETRY_SLEEP:-2}" attempt=1 local ra_cap="${CANARY_GH_RETRY_AFTER_CAP:-900}" case "$ra_cap" in ''|*[!0-9]*) ra_cap=900 ;; esac @@ -612,13 +617,19 @@ _repo_wf_runs_cached() { # so this is a real collision surface. sha256 (not sha1/md5 — those trip weak-hash linters # and are collision-broken) keeps the mapping injective; fall back to substitution only if # no hasher exists. This is a filename derivation, not a security context. - key="${repo}//${wf}" + # The run limit is part of the key: a shorter list cached for the default limit must never serve + # the ingress read that asks for more runs (it would look like a quieter repo). + key="${repo}//${wf}//${limit}" # sha256sum on Linux runners, shasum -a 256 on macOS; substitution only if neither # exists (and then, at worst, the pre-existing collision surface — never a crash). keyhash="$(printf '%s' "$key" | { sha256sum 2>/dev/null || shasum -a 256 2>/dev/null; } | cut -d' ' -f1)" [ -n "$keyhash" ] || keyhash="${key//[^A-Za-z0-9._-]/_}" cachef="$_RUNS_CACHE_DIR/${keyhash}.json" - [ -s "$cachef" ] && { cat "$cachef"; return 0; } + if [ -s "$cachef" ]; then + cat "$cachef" + [ "$strict" = "1" ] && [ -e "${cachef}.nf" ] && return 3 + return 0 + fi fi errfile="$(mktemp)" declare -p tmpfiles &>/dev/null || declare -g -a tmpfiles=() @@ -630,7 +641,7 @@ _repo_wf_runs_cached() { trap 'rm -f "${tmpfiles[@]+"${tmpfiles[@]}"}"' EXIT while :; do if out="$(gh run list --repo "$repo" --workflow "$wf" \ - -L 1000 --json conclusion,createdAt,databaseId,workflowName 2>"$errfile")"; then + -L "$limit" --json conclusion,createdAt,databaseId,workflowName 2>"$errfile")"; then rm -f "$errfile"; out="${out:-[]}" [ -n "${cachef:-}" ] && [ -d "$_RUNS_CACHE_DIR" ] && printf '%s' "$out" > "$cachef" 2>/dev/null || true printf '%s\n' "$out"; return 0 @@ -644,8 +655,12 @@ _repo_wf_runs_cached() { # counted as an outage; retrying it was the #810→#803 cancellation storm. if [[ "${err,,}" == *"could not find any workflow"* ]] || [[ "${err,,}" == *"no workflows"* ]]; then rm -f "$errfile" - [ -n "${cachef:-}" ] && [ -d "$_RUNS_CACHE_DIR" ] && printf '%s' '[]' > "$cachef" 2>/dev/null || true - echo '[]'; return 0 + # Marker FIRST, then the cached []: a concurrent strict reader that sees the cache file is + # guaranteed to see the .nf marker too (it never reads "workflow exists, no runs"). + [ -n "${cachef:-}" ] && [ -d "$_RUNS_CACHE_DIR" ] && : > "${cachef}.nf" 2>/dev/null && printf '%s' '[]' > "$cachef" 2>/dev/null || true + echo '[]' + [ "$strict" = "1" ] && return 3 + return 0 fi summary="$(_gh_err_summary "$err")" if [ "$attempt" -ge "$attempts" ]; then @@ -692,14 +707,242 @@ _run_json() { <<< "${raw:-[]}" 2>/dev/null || echo '[]' } +# ── ADR-0007 collapsed repos: attribute runs by ingress JOB (#1224) ────────────────── +# A collapsed repo folds its per-role Class-1 caller stubs into ONE `Agent Ingress` workflow +# with one job per role, so `gh run list --workflow ""` finds nothing there. The +# agent's runs are then the ingress runs in which its role job (registry `ingress_job`) ran. +# +# DELIBERATE DUPLICATION: the per-run role reduction below mirrors .github-private's +# scripts/lib/run-attribution.sh `normalize_ingress_runs` (#1727) — role = job-name segment +# before " / ", worst-outcome precedence across a role's jobs, an all-skipped role did not run, +# a run with no jobs is never silently dropped. It is NOT importable: this engine runs from a +# .github checkout with no .github-private tree (the same constraint as the inline autocut, +# #613/#1069). Keep the two in step if the attribution rules change. + +# _ingress_workflow — the collapsed ingress workflow's display name (registry `.ingress.workflow`). +_ingress_workflow() { + local w; w="$(_jq -r '.ingress?.workflow? // empty' 2>/dev/null || true)" + printf '%s' "${w:-Agent Ingress}" +} + +# _record_unresolved — note a ring member whose runs could not be +# attributed. Appends "\t" to $_CANARY_UNRESOLVED_FLAG (armed by _frontier_state) +# and warns once per distinct entry. A file, not a var: callers run in subshells. +_record_unresolved() { + local agent="$1" repo="$2" reason="$3" line + line="$(printf '%s\t%s' "$repo" "$reason")" + if [ -n "${_CANARY_UNRESOLVED_FLAG:-}" ]; then + grep -qxF -- "$line" "$_CANARY_UNRESOLVED_FLAG" 2>/dev/null && return 0 + # A failed append (temp fs full) must not leave a clean-looking empty flag: remove it, so the + # gate sees the flag GONE and fails closed (see the "evidence write failed" check in _frontier_state). + # If the cleanup fails too (read-only fs) the empty flag survives; _pair_state re-probes its + # writability and fails closed in that case. + printf '%s\n' "$line" >> "$_CANARY_UNRESOLVED_FLAG" 2>/dev/null || rm -f "$_CANARY_UNRESOLVED_FLAG" 2>/dev/null \ + || echo "::error::canary: cannot record or remove the unresolved-member flag '$_CANARY_UNRESOLVED_FLAG'; the gate will fail closed" >&2 + fi + echo "::warning::canary: UNRESOLVED member $repo for '$agent' — $reason" >&2 +} + +# _unresolved_flag_path — per-process UNRESOLVED flag path, so concurrent +# invocations for the same agent/candidate never truncate each other's evidence. +_unresolved_flag_path() { + printf '%s/.canary-unresolved-%s-%s-%s.txt' "${TMPDIR:-/tmp}" "$$" "$1" "$2" +} + +# _run_jobs_json — `gh run view --json jobs` for one run, file-memoized under +# $_RUNS_CACHE_DIR (an ingress run is shared by every collapsed role, so each run's jobs are read +# once per sweep). Bounded retry (CANARY_GH_RETRIES, default 6; an optional 3rd arg caps the attempts — +# the legacy single-call readers pass 1, preserving their pre-#1224 "one call, fail fast" behavior). +# Returns 0 on success; 2 when the run is permanently gone/unreadable +# (deleted or expired — HTTP 404); 1 when the jobs endpoint stayed unreadable after the full backoff +# (a transient/systemic failure — callers use that to stop reading, see _ingress_agent_runs). +_run_jobs_json() { + local repo="$1" id="$2" cachef="" keyhash out attempt=1 + local attempts="${3:-${CANARY_GH_RETRIES:-6}}" base="${CANARY_GH_RETRY_SLEEP:-2}" delay span expo + case "$attempts" in ''|*[!0-9]*) attempts=6 ;; esac + case "$base" in ''|*[!0-9]*) base=2 ;; esac + [ "$attempts" -lt 1 ] && attempts=1 + if [ -n "${_RUNS_CACHE_DIR:-}" ]; then + keyhash="$(printf 'jobs//%s//%s' "$repo" "$id" | { sha256sum 2>/dev/null || shasum -a 256 2>/dev/null; } | cut -d' ' -f1)" + [ -n "$keyhash" ] || keyhash="jobs_${repo//[^A-Za-z0-9._-]/_}_${id}" + cachef="$_RUNS_CACHE_DIR/${keyhash}.jobs.json" + [ -s "$cachef" ] && { cat "$cachef"; return 0; } + fi + local errf; errf="$(mktemp 2>/dev/null || echo /dev/null)" + while :; do + if out="$(gh run view "$id" --repo "$repo" --json jobs 2>"$errf")" && jq -e '.jobs|type=="array"' >/dev/null 2>&1 <<< "$out"; then + [ "$errf" != /dev/null ] && rm -f "$errf" + [ -n "$cachef" ] && [ -d "$_RUNS_CACHE_DIR" ] && printf '%s' "$out" > "$cachef" 2>/dev/null || true + printf '%s\n' "$out"; return 0 + fi + # A run deleted/expired since `gh run list` is permanent — fail fast, don't burn the backoff. + if grep -qiE 'could not find (any )?(workflow )?run|HTTP 404|run [0-9]+ not found' "$errf" 2>/dev/null; then + [ "$errf" != /dev/null ] && rm -f "$errf"; return 2 + fi + if [ "$attempt" -ge "$attempts" ]; then [ "$errf" != /dev/null ] && rm -f "$errf"; return 1; fi + # Same policy as the run-list path: exponential backoff with full jitter in [base, 30s]. + expo=$((attempt - 1)); [ "$expo" -gt 20 ] && expo=20 + delay=$(( base << expo )); span=$(( delay - base )); [ "$span" -lt 0 ] && span=0 + delay=$(( base + RANDOM % (span + 1) )); [ "$delay" -gt 30 ] && delay=30 + sleep "$delay"; attempt=$((attempt + 1)) + done +} + +# _ingress_agent_runs (ingress runs JSON on stdin) +# — the agent's runs on a collapsed repo, one record per completed ingress run in which the +# job ran: {conclusion, createdAt, databaseId, workflowName: , role}. +# workflowName is the agent's registered run_workflow so its benign/suspect classes keep +# matching across the collapse (parity with the legacy per-role bucket). Only runs at/after +# have their jobs read. A run with no jobs (unless it is a startup_failure, which hit +# every role) or unreadable jobs makes the member UNRESOLVED. +_ingress_agent_runs() { + local agent="$1" repo="$2" role="$3" since="$4" wf iraw id created rconc jobs rec recs="" + wf="$(_agent_field "$agent" run_workflow)" + iraw="$(cat)" + # Bound the per-run `gh run view` fan-out (#810/#819): read at most CANARY_INGRESS_JOBS_MAX + # (default 300) of the newest in-window runs. Beyond that the member is UNRESOLVED (fail closed) — + # EXCEPT for the trailing BASELINE read (_baseline_daily arms _CANARY_INGRESS_TRUNC_FLAG): the + # baseline only sizes the sample target, so a truncated read is a valid sample of the NEWEST days. + # The first unread run's day is recorded and _baseline_daily drops that (incomplete) day and every + # older one, rather than blocking a busy collapsed repo forever. Gating windows stay strict. + local jobs_max="${CANARY_INGRESS_JOBS_MAX:-300}" nread=0 jrc role_oldest="" skipkind norole_created="" ncr + case "$jobs_max" in ''|*[!0-9]*) jobs_max=300 ;; esac + [ "$jobs_max" -lt 1 ] && jobs_max=300 + while IFS=$'\t' read -r id created rconc; do + [ -z "$id" ] && continue + [ "$created" = "-" ] && created="" # "-" kept the empty field from collapsing in the tab-split read + if [ "$nread" -ge "$jobs_max" ]; then + if [ -n "${_CANARY_INGRESS_TRUNC_FLAG:-}" ] && printf '%s\n' "${created:0:10}" >> "$_CANARY_INGRESS_TRUNC_FLAG" 2>/dev/null; then + echo "::warning::$agent $repo: more than $jobs_max ingress runs in the baseline window; using the newest days only (CANARY_INGRESS_JOBS_MAX)." >&2 + else + _record_unresolved "$agent" "$repo" "more than $jobs_max ingress runs in the window; job reads capped (CANARY_INGRESS_JOBS_MAX)" + fi + break + fi + nread=$((nread + 1)) + jrc=0; jobs="$(_run_jobs_json "$repo" "$id")" || jrc=$? + if [ "$jrc" -eq 1 ]; then + # Circuit breaker: the jobs endpoint stayed unreadable after the full backoff, so every later + # run would burn the same ~30s+ of retries (300 runs can outlive the job timeout before + # sync-issues records the fail-closed state). Stop reading; the member is UNRESOLVED. + _record_unresolved "$agent" "$repo" "jobs endpoint unreadable (run $id) — stopped reading jobs for this member"; break + elif [ "$jrc" -ne 0 ]; then + _record_unresolved "$agent" "$repo" "jobs of ingress run $id are unreadable"; continue + fi + rec="$(jq -c --arg role "$role" --arg wf "$wf" --arg id "$id" --arg created "$created" --arg rconc "$rconc" ' + def rank(c): + # failure/timed_out rank STRICTLY above action_required/startup_failure: max_by keeps the last of + # tied elements, so a failed job followed by an action_required one must not reduce the role to + # action_required (downstream counters select conclusion=="failure" and would drop the failure). + if c == "failure" or c == "timed_out" then 7 + elif c == "action_required" or c == "startup_failure" then 6 + elif c == "cancelled" then 5 # an interrupted job must not be masked by a green sibling + elif c == "success" then 4 + elif c == null or c == "skipped" then 0 + else 2 end; + (.jobs // []) as $jobs + | if ($jobs | length) == 0 then + # A run cancelled before any job started (e.g. superseded by workflow concurrency) never + # executed this role: it is not a success or a failure, so it is simply not a record. + (if $rconc == "startup_failure" then {conclusion: "startup_failure", nojobs: true} + elif $rconc == "cancelled" then {skipped: "cancelled"} + else "UNATTRIBUTED" end) + else + ([ $jobs[] | select(((.name // "") | split(" / ")[0]) == $role) | {c: .conclusion, r: rank(.conclusion)} ]) as $rj + # No job carries the role: either the role was added to the ingress AFTER this run + # (pre-adoption — benign) or ingress_job is renamed/misspelled (blind). The caller decides, + # once it knows whether ANY run in the window carries the role. + | if ($rj | length) == 0 then "NOROLE" + else ($rj | max_by(.r)) as $w + | if $w.r == 0 then {skipped: "role"} + # A job-level timeout surfaces as a failed run on a legacy per-role workflow. + else {conclusion: (if $w.c == "timed_out" then "failure" else $w.c end)} end + end + end + | if . == "UNATTRIBUTED" or . == "NOROLE" or (type == "object" and has("skipped")) then . + else . + {createdAt: $created, databaseId: ($id | tonumber), workflowName: $wf, role: $role} end + ' <<< "$jobs" 2>/dev/null)" || rec='"UNATTRIBUTED"' + if [ "$rec" = '"UNATTRIBUTED"' ]; then + _record_unresolved "$agent" "$repo" "ingress run $id has no attributable jobs"; continue + fi + if [ "$rec" = '"NOROLE"' ]; then + # Defer: pre-adoption runs are benign, a renamed ingress_job is not (see after the loop). + # Keep the run's own createdAt (no payload re-parse later); "-" stands in for a missing one so + # word-splitting cannot drop it — a run with an unknown date must still count as blind. + norole_created+="${created:--} "; continue + fi + skipkind="$(jq -r 'if type == "object" then (.skipped // empty) else empty end' <<< "$rec" 2>/dev/null || true)" + if [ -n "$skipkind" ]; then + [ "$skipkind" = "role" ] && role_oldest="$created" # the role job ran (skipped): it exists here + continue + fi + # newest→oldest: the last assignment is the OLDEST run carrying the role. A job-less startup_failure + # (nojobs) carries NO role job, so it must not count as the role's appearance. + [ "$(jq -r '.nojobs // empty' <<< "$rec" 2>/dev/null || true)" = "true" ] || role_oldest="$created" + [ -n "$rec" ] && recs+="$(jq -c 'del(.nojobs)' <<< "$rec")"$'\n' + done < <(jq -r --arg since "$since" \ + '[ .[]? | select(.conclusion != null and .conclusion != "") + | select($since == "" or (.createdAt // "") >= $since) ] + | sort_by(.createdAt // "") | reverse | .[] + | [(.databaseId|tostring), (if (.createdAt // "") == "" then "-" else .createdAt end), .conclusion] | @tsv' 2>/dev/null <<< "${iraw:-[]}") + # Runs where no job carried the role. If the role appears in an OLDER run, a no-role run that is + # older still predates the role's adoption by the ingress and is benign; one NEWER than the role's + # oldest appearance, or any when no run in the window carries the role at all (renamed/misspelled + # ingress_job), cannot be attributed → UNRESOLVED. The member stays blocked only for that. + if [ -n "$norole_created" ]; then + local blind=0 + for ncr in $norole_created; do + if [ -z "$role_oldest" ] || [ "$ncr" = "-" ] || ! [[ "$ncr" < "$role_oldest" ]]; then blind=$((blind + 1)); fi + done + [ "$blind" -gt 0 ] && _record_unresolved "$agent" "$repo" "$blind ingress run(s) carry no '$role' job and are not older than the role's first appearance (ingress_job renamed or misspelled?)" + fi + printf '%s' "$recs" | jq -cs '.' +} + +# _agent_run_json — the agent's runs on one ring member since the given +# Zulu timestamp (same record shape as _run_json, plus `role` for ingress-attributed runs). +# Resolution order (#1224): +# 1. the per-role workflow (`run_workflow`) exists → its runs (legacy, unchanged); +# 2. it is absent but the ingress exists and the agent registers `ingress_job` → the ingress +# runs in which that role job ran; +# 3. neither exists → [] (the repo does not consume this agent, #747); +# 4. the ingress exists but the role cannot be attributed (no `ingress_job`, a job-less run, +# unreadable jobs) → the member is UNRESOLVED (see _record_unresolved), never silent. +# A genuine fetch failure fails CLOSED (non-zero), exactly like _run_json. +_agent_run_json() { + local agent="$1" repo="$2" since="$3" wf iwf raw iraw role rc=0 + if [ -z "$repo" ] || [ "$repo" = '*' ]; then echo '[]'; return 0; fi + wf="$(_agent_field "$agent" run_workflow)" + raw="$(_repo_wf_runs_cached "$repo" "$wf" 1)" || rc=$? + if [ "$rc" -eq 3 ]; then + rc=0; iwf="$(_ingress_workflow)" + iraw="$(_repo_wf_runs_cached "$repo" "$iwf" 1 "${CANARY_INGRESS_RUN_LIMIT:-5000}")" || rc=$? + case "$rc" in + 0) + role="$(_jq -r --arg a "$agent" '(.agents[$a].ingress_job)? // empty' 2>/dev/null || true)" + if [ -z "$role" ]; then + _record_unresolved "$agent" "$repo" "no '$wf' workflow but '$iwf' is present and the agent registers no ingress_job" + echo '[]'; return 0 + fi + _ingress_agent_runs "$agent" "$repo" "$role" "$since" <<< "$iraw"; return 0 ;; + 3) echo '[]'; return 0 ;; + *) return 1 ;; + esac + elif [ "$rc" -ne 0 ]; then + return 1 + fi + jq -c --arg since "$since" \ + '[ .[]? | select($since == "" or (.createdAt // "") >= $since) ]' \ + <<< "${raw:-[]}" 2>/dev/null || echo '[]' +} + # _tier_sample — EXECUTED runs (success+failure) on the # source tier since the candidate cut. Prints " ". _tier_sample() { local agent="$1" since="$2"; shift 2 - local wf repo json executed=0 earliest="" e - wf="$(_agent_field "$agent" run_workflow)" + local repo json executed=0 earliest="" e for repo in "$@"; do - json="$(_run_json "$repo" "$wf" "$since")" + json="$(_agent_run_json "$agent" "$repo" "$since")" executed=$(( executed + $(jq '[.[]?|select(.conclusion=="success" or .conclusion=="failure")]|length' 2>/dev/null <<< "${json:-[]}" || echo 0) )) e="$(jq -r '[.[]?|select(.conclusion=="success" or .conclusion=="failure")|.createdAt?]|min // empty' 2>/dev/null <<< "$json" || echo "")" if [ -n "$e" ] && { [ -z "$earliest" ] || [[ "$e" < "$earliest" ]]; }; then earliest="$e"; fi @@ -761,8 +1004,7 @@ _suspect_downgrade_patterns() { # candidate window). _suspect_class_counts() { local agent="$1" since="$2" before="$3" wf_re="$4" step_re="$5"; shift 5 - local wf repo json rid rwf sig sig_rc matched=0 executed=0 incomplete=0 count - wf="$(_agent_field "$agent" run_workflow)" + local repo json rid rwf rrole sig sig_rc matched=0 executed=0 incomplete=0 count local exec_filter='.[]?|select(.conclusion=="success" or .conclusion=="failure")' local fail_filter='.[]?|select(.conclusion=="failure")' if [ "$before" != "-" ]; then @@ -771,15 +1013,16 @@ _suspect_class_counts() { fi for repo in "$@"; do { [ -z "$repo" ] || [ "$repo" = '*' ]; } && continue - json="$(_run_json "$repo" "$wf" "$since")" + json="$(_agent_run_json "$agent" "$repo" "$since")" count="$(jq "[${exec_filter}]|length" 2>/dev/null <<< "$json" || echo 0)" executed=$(( executed + ${count:-0} )) - while IFS=$'\t' read -r rid rwf || [ -n "$rid" ]; do + while IFS=$'\t' read -r rid rwf rrole || [ -n "$rid" ]; do rid="${rid%$'\r'}" rwf="${rwf%$'\r'}" + rrole="${rrole%$'\r'}" [ -z "$rid" ] && continue sig_rc=0 - sig="$(_run_signature "$repo" "$rid")" || sig_rc=$? # || prevents set -e on lookup failure + sig="$(_run_signature "$repo" "$rid" "$rrole")" || sig_rc=$? # || prevents set -e on lookup failure if [ -z "$sig" ]; then [ "$sig_rc" -ne 0 ] && incomplete=$(( incomplete + 1 )) # lookup failed; genuine empty sig is fine continue @@ -787,26 +1030,28 @@ _suspect_class_counts() { if [ "$(benign_match "$rwf" "$sig" "$wf_re" "$step_re")" = "yes" ]; then matched=$(( matched + 1 )) fi - done < <(jq -r "[${fail_filter}]|.[]|[(.databaseId // \"\"|tostring),(.workflowName // \"\")]|@tsv" 2>/dev/null <<< "$json") + done < <(jq -r "[${fail_filter}]|.[]|[(.databaseId // \"\"|tostring),(.workflowName // \"\"),(.role // \"\")]|@tsv" 2>/dev/null <<< "$json") done echo "$matched $executed $incomplete" } -# Memoization cache for _run_signature: keyed by "repo:run_id". +# Memoization cache for _run_signature: keyed by "repo:run_id:role". # Avoids duplicate gh run view calls for the same (repo, run_id) across agents # in evaluate-all (where multiple agents can share repos). declare -A _RUN_SIG_CACHE=() -# _run_signature — the failed step names of a run, joined by newlines -# (the "step/error signature" the allowlist matches against). Empty repo/wildcard/id or +# _run_signature [] — the failed step names of a run, joined by newlines +# (the "step/error signature" the allowlist matches against). A (an ADR-0007 ingress- +# attributed run, #1224) scopes it to that role's own jobs, so another role's failure in the +# same ingress run can never make this agent's failure look benign (or suspect). Empty repo/wildcard/id or # any gh error → "" with exit 1 (fail-closed: an unknown signature is never treated as # benign, and callers that need to distinguish a lookup failure from a genuine empty # signature check the exit code). A successful lookup with no failed steps → "" exit 0. # The $'\x01' sentinel in the cache marks a prior lookup failure so it is not retried. _run_signature() { - local repo="$1" id="$2" cache_key sig json + local repo="$1" id="$2" role="${3:-}" cache_key sig json { [ -z "$repo" ] || [ "$repo" = '*' ] || [ -z "$id" ]; } && { echo ""; return 0; } - cache_key="${repo}:${id}" + cache_key="${repo}:${id}:${role}" if [[ -v _RUN_SIG_CACHE["$cache_key"] ]]; then if [ "${_RUN_SIG_CACHE[$cache_key]}" = $'\x01' ]; then echo ""; return 1 # cached lookup failure — signal to caller @@ -817,21 +1062,21 @@ _run_signature() { # Use || { } so set -e does not trigger on a failing gh run view; the block caches the # sentinel and returns 1 to signal the lookup failure to callers that care (e.g. # _suspect_class_counts tracks incomplete evidence and HOLDs the downgrade). - json="$(gh run view "$id" --repo "$repo" --json jobs 2>/dev/null)" || { + json="$(_run_jobs_json "$repo" "$id" 1)" || { _RUN_SIG_CACHE["$cache_key"]=$'\x01' echo ""; return 1 } - sig="$(jq -r '[.jobs[]?|.steps[]?|select(.conclusion=="failure")|.name] | join("\n")' \ - 2>/dev/null <<< "$json" || echo "")" + sig="$(jq -r --arg role "$role" '[.jobs[]?|select($role == "" or ((.name // "")|split(" / ")[0]) == $role) + |.steps[]?|select(.conclusion=="failure")|.name] | join("\n")' 2>/dev/null <<< "$json" || echo "")" _RUN_SIG_CACHE["$cache_key"]="$sig" echo "$sig" } -# _failure_benign — return 0 if this +# _failure_benign [] — return 0 if this # in-window failure matches any allowlist entry, else 1. Fail-closed on an empty signature. _failure_benign() { - local repo="$1" rid="$2" rwf="$3" patterns="$4" sig wf_re step_re - sig="$(_run_signature "$repo" "$rid")" + local repo="$1" rid="$2" rwf="$3" patterns="$4" role="${5:-}" sig wf_re step_re + sig="$(_run_signature "$repo" "$rid" "$role")" [ -z "$sig" ] && return 1 while IFS=$'\t' read -r wf_re step_re; do [ -z "$step_re" ] && continue @@ -840,14 +1085,14 @@ _failure_benign() { return 1 } -# _failure_suspect — return 0 if this +# _failure_suspect [] — return 0 if this # in-window failure matches any SUSPECT allowlist entry (#668 increment 2), else 1. Reuses # the benign_match pure matcher against the run's failed-step signature; fail-closed on an # empty signature (an unknown signature is never treated as suspect). A suspect match does # NOT exclude the failure — it narrows the triage verdict (SUSPECT vs REGRESSION) only. _failure_suspect() { - local repo="$1" rid="$2" rwf="$3" patterns="$4" sig wf_re step_re - sig="$(_run_signature "$repo" "$rid")" + local repo="$1" rid="$2" rwf="$3" patterns="$4" role="${5:-}" sig wf_re step_re + sig="$(_run_signature "$repo" "$rid" "$role")" [ -z "$sig" ] && return 1 while IFS=$'\t' read -r wf_re step_re || [ -n "$wf_re" ]; do wf_re="${wf_re%$'\r'}" @@ -942,12 +1187,11 @@ _run_is_stale() { # REGRESSION. Prints " ". _cumulative_health() { local agent="$1" since="$2" differs="$3" cand="${4:--}"; shift 4 - local wf repo json fail=0 startup=0 benign=0 suspect=0 stale=0 patterns="" suspect_patterns="" rid rwf - wf="$(_agent_field "$agent" run_workflow)" + local repo json fail=0 startup=0 benign=0 suspect=0 stale=0 patterns="" suspect_patterns="" rid rwf rrole patterns="$(_benign_patterns "$agent" "$differs")" suspect_patterns="$(_suspect_patterns "$agent")" for repo in "$@"; do - json="$(_run_json "$repo" "$wf" "$since")" + json="$(_agent_run_json "$agent" "$repo" "$since")" # startup_failure runs never executed a job, so no release can be attributed to them (no # "Uses:" log line): they always count, fail closed (#1176). startup=$(( startup + $(jq '[.[]?|select(.conclusion=="startup_failure")]|length' 2>/dev/null <<< "${json:-[]}" || echo 0) )) @@ -956,40 +1200,75 @@ _cumulative_health() { # jq pass, avoiding a gh run view call per failure. fail=$(( fail + $(jq '[.[]?|select(.conclusion=="failure")]|length' 2>/dev/null <<< "${json:-[]}" || echo 0) )) else - while IFS=$'\t' read -r rid rwf; do + while IFS=$'\t' read -r rid rwf rrole; do if _run_is_stale "$agent" "$cand" "$repo" "$rid"; then stale=$(( stale + 1 )) - elif [ -n "$patterns" ] && _failure_benign "$repo" "$rid" "$rwf" "$patterns"; then + elif [ -n "$patterns" ] && _failure_benign "$repo" "$rid" "$rwf" "$patterns" "$rrole"; then benign=$(( benign + 1 )) else fail=$(( fail + 1 )) - if [ -n "$suspect_patterns" ] && _failure_suspect "$repo" "$rid" "$rwf" "$suspect_patterns"; then + if [ -n "$suspect_patterns" ] && _failure_suspect "$repo" "$rid" "$rwf" "$suspect_patterns" "$rrole"; then suspect=$(( suspect + 1 )) fi fi - done < <(jq -r '.[]?|select(.conclusion=="failure")|[(.databaseId // "" | tostring),(.workflowName // "")]|@tsv' 2>/dev/null <<< "$json") + done < <(jq -r '.[]?|select(.conclusion=="failure")|[(.databaseId // "" | tostring),(.workflowName // ""),(.role // "")]|@tsv' 2>/dev/null <<< "$json") fi done echo "$fail $startup $benign $suspect $stale" } # _baseline_daily — per-day EXECUTED counts on the -# source tier over the trailing window_days (exactly window_days integers, zero-filled), +# source tier over the trailing window_days (window_days integers, zero-filled — fewer only when a +# capped collapsed-repo read left the older days unknown, see _ingress_agent_runs), # feeding the robust spike-capped baseline for the sample target (#548). _baseline_daily() { local agent="$1" window="$2"; shift 2 - local wf since repo json dates="" day i count out="" - wf="$(_agent_field "$agent" run_workflow)" + local since repo json day i count out="" tflag d cutday any_trunc=0 total=0 + local -A dsum=() dknown=() rcount=() + local -a days=() since="$(_iso_now_minus_days "$window")" - for repo in "$@"; do - json="$(_run_json "$repo" "$wf" "$since")" - dates+="$(jq -r '.[]?|select(.conclusion=="success" or .conclusion=="failure")|.createdAt[0:10]?' 2>/dev/null <<< "$json" || true)"$'\n' - done for (( i=0; i/dev/null || date -u -v"-${i}d" +%Y-%m-%d 2>/dev/null || echo "")" - count=$(grep -c "^${day}$" 2>/dev/null <<< "$dates" || true) + days+=("$day") + done + # A capped ingress read (see _ingress_agent_runs) leaves every day OLDER than that repo's first unread + # run's day unknown — NOT zero — and that boundary day itself only partly counted. Truncation is + # tracked PER REPO: a truncated member drops only its own older days, so a fully-read member's history + # still counts. The boundary day is kept only when it has observed runs (a lower bound, never a false + # zero). A day is dropped from the baseline only when NO member knows it. A truncated read must also + # never look like "no caller" (an all-zero baseline lets waive_sample_if_no_caller skip sampling in + # _pair_state), so if nothing countable was observed the baseline is a non-zero floor, which sizes the + # sample target at its clamp minimum. + for repo in "$@"; do + tflag="$(mktemp 2>/dev/null || true)" # unwritable → the ingress cap stays strict (UNRESOLVED) + json="$(_CANARY_INGRESS_TRUNC_FLAG="$tflag" _agent_run_json "$agent" "$repo" "$since")" + cutday="" + if [ -n "$tflag" ] && [ -s "$tflag" ]; then + while IFS= read -r d; do [ -n "$d" ] && { [ -z "$cutday" ] || [[ "$d" > "$cutday" ]]; } && cutday="$d"; done < "$tflag" + fi + [ -n "$tflag" ] && rm -f "$tflag" + [ -n "$cutday" ] && any_trunc=1 + rcount=() + while IFS= read -r d; do [ -n "$d" ] && rcount["$d"]=$(( ${rcount["$d"]:-0} + 1 )); done \ + < <(jq -r '.[]?|select(.conclusion=="success" or .conclusion=="failure")|.createdAt[0:10]?' 2>/dev/null <<< "$json" || true) + for day in "${days[@]}"; do + count="${rcount["$day"]:-0}" + if [ -n "$cutday" ]; then + [[ "$day" < "$cutday" ]] && continue + [ "$day" = "$cutday" ] && [ "$count" -eq 0 ] && continue + fi + dknown["$day"]=1 + dsum["$day"]=$(( ${dsum["$day"]:-0} + count )) + done + done + for day in "${days[@]}"; do + # No truncated member → every day is known (zero-filled); otherwise only days some member knows. + if [ "$any_trunc" -eq 1 ] && [ "$#" -gt 0 ] && [ -z "${dknown["$day"]:-}" ]; then continue; fi + count="${dsum["$day"]:-0}" + total=$(( total + count )) out+="${count} " done + [ "$any_trunc" -eq 1 ] && [ "$total" -eq 0 ] && out="1 " echo "${out% }" } @@ -1032,18 +1311,22 @@ _reusable_differs() { # so the two never fetch the same run twice. declare -A _RUN_DECISION_CACHE=() -# _run_decision_class — the decision class a run took (the taken -# `` no-op step; skipped branches ignored), via `gh run view --json jobs`. +# _run_decision_class [] — the decision class a run took (the taken +# `` no-op step; skipped branches ignored), via `gh run view --json jobs`. A +# (an ADR-0007 ingress-attributed run, #1224) scopes the read to that role's own jobs. # Empty on missing repo/id or any gh error (fail-open: a run with no decision step simply # contributes nothing to the tally, degrading toward INSUFFICIENT, never a false SHIFT). _run_decision_class() { - local repo="$1" id="$2" prefix="$3" cache_key json cls + local repo="$1" id="$2" prefix="$3" role="${4:-}" cache_key json cls { [ -z "$repo" ] || [ "$repo" = '*' ] || [ -z "$id" ]; } && { echo ""; return 0; } - cache_key="${repo}:${id}" + cache_key="${repo}:${id}:${role}" if [[ -v _RUN_DECISION_CACHE["$cache_key"] ]]; then echo "${_RUN_DECISION_CACHE[$cache_key]}"; return 0 fi - json="$(gh run view "$id" --repo "$repo" --json jobs 2>/dev/null || echo '{}')" + json="$(_run_jobs_json "$repo" "$id" 1 2>/dev/null || echo '{}')" + if [ -n "$role" ]; then + json="$(jq -c --arg role "$role" 'if has("jobs") then .jobs |= map(select(((.name // "")|split(" / ")[0]) == $role)) else error("Missing jobs key") end' <<< "$json" 2>/dev/null || echo '{}')" + fi cls="$(decision_class "$prefix" "$json")" _RUN_DECISION_CACHE["$cache_key"]="$cls" echo "$cls" @@ -1057,23 +1340,22 @@ _run_decision_class() { # reads the object; the min-sample knobs gate sufficiency. _sample_decision_counts() { local agent="$1" prefix="$2" max_k="$3" since="$4" before="$5"; shift 5 - local wf repo json rid cls sampled=0 - wf="$(_agent_field "$agent" run_workflow)" + local repo json rid rrole cls sampled=0 declare -A counts=() local filter='.[]?|select(.conclusion=="success" or .conclusion=="failure")' [ "$before" != "-" ] && filter="$filter|select(.createdAt < \"$before\")" for repo in "$@"; do [ "$sampled" -ge "$max_k" ] && break { [ -z "$repo" ] || [ "$repo" = '*' ]; } && continue - json="$(_run_json "$repo" "$wf" "$since")" - while IFS= read -r rid; do + json="$(_agent_run_json "$agent" "$repo" "$since")" + while IFS=$'\t' read -r rid rrole; do [ -z "$rid" ] && continue [ "$sampled" -ge "$max_k" ] && break - cls="$(_run_decision_class "$repo" "$rid" "$prefix")" + cls="$(_run_decision_class "$repo" "$rid" "$prefix" "$rrole")" [ -z "$cls" ] && continue counts["$cls"]=$(( ${counts["$cls"]:-0} + 1 )) sampled=$(( sampled + 1 )) - done < <(jq -r "[${filter}]|sort_by(.createdAt)|reverse|.[]|(.databaseId|tostring)" 2>/dev/null <<< "$json") + done < <(jq -r "[${filter}]|sort_by(.createdAt)|reverse|.[]|[(.databaseId|tostring),(.role // \"\")]|@tsv" 2>/dev/null <<< "$json") done # Build the counts object in a SINGLE jq pass (one process, not one per key) — # jq does the key/value escaping so arbitrary class names stay valid JSON. @@ -1168,7 +1450,8 @@ _decision_mix_table() { # `next`); the gate measures dwell from THAT candidate's own cut, samples on the tier, # and scopes cumulative health to every tier EXCEPT a ring strictly BELOW that runs a # DIFFERENT (newer) candidate — so a newer candidate churning on a lower tier can never block an -# older, independently-clean candidate on a higher pair. Echoes the same 18-field line documented +# older, independently-clean candidate on a higher pair. A ring member whose runs cannot be +# attributed (#1224) holds an otherwise-clean gate BLOCKED with triage UNRESOLVED. Echoes the same 18-field line documented # on _frontier_state. triage/mix_shift/downgrade semantics are unchanged from the single-frontier # implementation this was extracted from. _pair_state() { @@ -1181,6 +1464,24 @@ _pair_state() { fi now_epoch="$(date -u +%s)" + # Arm the UNRESOLVED-member flag (#1224): _agent_run_json records each ring member whose runs + # cannot be attributed (ADR-0007 ingress present, role job unresolvable). A file, because the + # run reads below happen in subshells. A caller may pre-arm it (sync-issues evidence). + local uflag="${_CANARY_UNRESOLVED_FLAG:-}" uflag_owned=0 + if [ -z "$uflag" ]; then + # Per-process path (never shared across concurrent invocations) so _unresolved_evidence in + # this same process can reuse the flag instead of re-walking the runs (#1224). + uflag="$(_unresolved_flag_path "$agent" "$cand")"; uflag_owned=1 + if : >"$uflag" 2>/dev/null; then + export _CANARY_UNRESOLVED_FLAG="$uflag" + else + # Cannot arm the flag (TMPDIR unwritable/full) → unattributable members could not be + # recorded; fail closed rather than let a clean-looking gate PROMOTE. + echo "WARN: cannot create unresolved-member flag '$uflag'; failing closed" >&2 + echo "${cand:--} $frontier $transition BLOCKED 0 0 0 0 0 0 0 FLAG_ERROR - - 0 0 0 0"; return 0 + fi + fi + # Source-tier repos (the tier currently running the candidate). local src_repos=() r while IFS= read -r r; do [ -n "$r" ] && src_repos+=("$r"); done < <(resolve_members "$agent" "$source") @@ -1360,6 +1661,33 @@ _pair_state() { fi fi + # A blind gate fails LOUDLY (#1224): a ring member whose runs could not be attributed is NOT + # passing evidence. Name it in an ::error::, and hold an otherwise-clean gate (PROMOTE / + # SOAKING / AWAITING_CONFIRMATION) as BLOCKED with triage UNRESOLVED so sync-issues tracks it. + # A gate already BLOCKED by real failures keeps that (more specific) triage. + local unresolved="" + if [ -n "$uflag" ] && [ -s "$uflag" ]; then + unresolved="$(cut -f1 "$uflag" | sort -u | paste -sd, - | sed 's/,/, /g')" + elif [ -n "$uflag" ] && [ ! -e "$uflag" ]; then + unresolved="(unresolved-member evidence could not be persisted)" + elif [ -n "$uflag" ] && ! : >>"$uflag" 2>/dev/null; then + # Empty but no longer writable: a failed append whose cleanup also failed (read-only fs) would + # look clean here, so treat an unwritable flag as lost evidence. + unresolved="(unresolved-member evidence could not be persisted)" + fi + if [ "$uflag_owned" -eq 1 ]; then + # Keep a non-empty flag for _unresolved_evidence to reuse (it deletes it); drop an empty one. + [ -n "$unresolved" ] || rm -f "$uflag" + unset _CANARY_UNRESOLVED_FLAG + fi + if [ -n "$unresolved" ]; then + echo "::error::canary gate for '$agent' [$transition] is BLIND on ring member(s) $unresolved — their runs could not be attributed (ADR-0007 ingress present, role job unresolvable). Counted as UNRESOLVED, not passing evidence; register the agent's ingress_job in the ring registry." >&2 + # PRE_EXISTING is advanceable via `promote --allow-pre-existing`, so UNRESOLVED must dominate it. + if [ "$state" != "BLOCKED" ] || [ "$triage" = "PRE_EXISTING" ]; then + state="BLOCKED"; triage="UNRESOLVED"; mix_shift="-"; downgrade="-" + fi + fi + echo "${cand:--} $frontier $transition $state $dwell_h $dwell_floor $sample $target $cum_fail $cum_startup $cum_benign $triage $mix_shift $downgrade $dg_cand_rate $dg_cand_sample $dg_base_rate $dg_base_sample" } @@ -1465,6 +1793,10 @@ cmd_evaluate() { echo "::warning::triage=SUSPECT — failure matches a suspect class (possibly candidate-caused). BLOCKS + needs a human; see the blocker issue's discriminating question, then promote --override if unrelated or roll back if a real regression." elif [ "$downgrade" = "DOWNGRADE" ]; then echo "::notice::triage=PRE_EXISTING (auto-downgraded from SUSPECT, #668 increment 6) — the candidate's suspect-class failure rate (${dg_cand_rate}‰ over ${dg_cand_sample} runs) is no worse than the prior version's (${dg_base_rate}‰ over ${dg_base_sample} runs), so the timeout is environmental, not a candidate regression. Report only; the SUSPECT hold auto-cleared (no human needed). Advances with --allow-pre-existing once dwell/sample pass." + elif [ "$triage" = "UNRESOLVED" ]; then + echo "::error::triage=UNRESOLVED — at least one ring member's runs could not be attributed (ADR-0007 collapsed repo: '$(_ingress_workflow)' present, no per-role workflow, role job unresolvable). The gate is blind there, so it holds rather than promote on incomplete evidence. Register the agent's ingress_job in the ring registry (#1224)." + elif [ "$triage" = "FLAG_ERROR" ]; then + echo "::error::triage=FLAG_ERROR — could not create the unresolved-member flag file (TMPDIR unwritable/full). The gate fails closed and holds; fix the runner's temp dir." elif [ "$triage" = "PRE_EXISTING" ]; then echo "::warning::triage=PRE_EXISTING — failure is pre-existing/environmental. Report only; do NOT rollback. Advances with --allow-pre-existing (or control.allow_pre_existing in the registry) once dwell/sample pass." else @@ -1748,8 +2080,7 @@ _gh_issue_create() { # runs (repo + run link + failed-step signature), capped at 8. Uses the SAME per-candidate # window + tier repos the gate counts, so the evidence matches cum_fail. _blocker_evidence() { - local agent="$1" cand="$2" wf cut_z repo json r n=0 out="" - wf="$(_agent_field "$agent" run_workflow)" + local agent="$1" cand="$2" cut_z repo json r n=0 out="" cut_z="$(candidate_cut_date "$agent" "$cand")" [ -z "$cut_z" ] && { printf '_(no candidate cut date resolved — cannot list failing runs)_\n'; return 0; } local chan_array=() ch all=() seen=" " dedup=() @@ -1769,24 +2100,49 @@ _blocker_evidence() { fi done for repo in "${dedup[@]}"; do - json="$(_run_json "$repo" "$wf" "$cut_z")" - local rid sig + json="$(_agent_run_json "$agent" "$repo" "$cut_z")" + local rid rrole sig local concl - while IFS=$'\t' read -r rid concl; do + while IFS=$'\t' read -r rid concl rrole; do [ -z "$rid" ] && continue # Old release: not counted (#1176). Only `failure` runs are attributable — a startup_failure # never ran a job, so it has no "Uses:" line and always counts (cum_startup), like the gate. if [ "$concl" = "failure" ] && [ -z "${raw_evidence["$repo"]:-}" ] && _run_is_stale "$agent" "$cand" "$repo" "$rid"; then continue; fi if [ "$n" -ge 8 ]; then out+="- _(…more failing runs; truncated at 8)_"$'\n'; printf '%s' "$out"; return 0; fi - sig="$(_run_signature "$repo" "$rid" | tr '\n' ';' | sed 's/;$//')" + sig="$(_run_signature "$repo" "$rid" "$rrole" | tr '\n' ';' | sed 's/;$//')" out+="- \`$repo\` — run [$rid](https://github.com/$repo/actions/runs/$rid); failed steps: ${sig:-unknown}"$'\n' n=$((n+1)) - done < <(jq -r '.[]?|select(.conclusion=="failure" or .conclusion=="startup_failure")|[(.databaseId|tostring),.conclusion]|@tsv' 2>/dev/null <<< "$json") + done < <(jq -r '.[]?|select(.conclusion=="failure" or .conclusion=="startup_failure")|[(.databaseId|tostring),.conclusion,(.role // "")]|@tsv' 2>/dev/null <<< "$json") done [ -z "$out" ] && out="_(no failing runs in the per-candidate window — cum_fail may be startup_failures or a transient count)_"$'\n' printf '%s' "$out" } +# _unresolved_evidence — markdown bullets naming each ring member whose +# runs could not be attributed (#1224), with the reason. Re-walks the same per-candidate window +# and tier repos as _blocker_evidence under a fresh UNRESOLVED flag. +_unresolved_evidence() { + local agent="$1" cand="$2" flag repo reason out="" flag_is_persistent=0 reuse=0 + # Reuse the flag left by _frontier_state (a caller's pre-armed flag first) to avoid re-walking (#1224) + flag="${_CANARY_UNRESOLVED_FLAG:-$(_unresolved_flag_path "$agent" "$cand")}" + if [ -s "$flag" ]; then + flag_is_persistent=1 + # A caller-armed flag belongs to the caller; only delete our own per-process one. + [ -z "${_CANARY_UNRESOLVED_FLAG:-}" ] && reuse=1 + else + # Flag not found or empty, create a new one and re-evaluate + flag="$(mktemp 2>/dev/null || echo "")" + [ -z "$flag" ] && { printf '_(could not list the unresolved members — see the workflow log ::error:: annotations)_\n'; return 0; } + _CANARY_UNRESOLVED_FLAG="$flag" _blocker_evidence "$agent" "$cand" >/dev/null 2>&1 || true + fi + while IFS=$'\t' read -r repo reason; do + [ -n "$repo" ] && out+="- \`$repo\` — $reason"$'\n' + done < <(sort -u "$flag") + if [ "$flag_is_persistent" -eq 0 ] || [ "$reuse" -eq 1 ]; then rm -f "$flag"; fi + [ -z "$out" ] && out="_(no unresolved member found in the blocker scope on re-check — the hold may come from the baseline window or an unpersisted evidence flag; it re-evaluates next tick)_"$'\n' + printf '%s' "$out" +} + # _suspect_guidance — emit the discriminating question(s) for the agent's # suspect_failure_classes (#668 increment 2), one bullet per class, so a SUSPECT blocker # tells the human exactly how to confirm in ~30 seconds. Empty if the agent has none. @@ -1845,6 +2201,10 @@ $(printf '%s\n' "$guidance" | sed 's/^/> /')" > |---|---|---| > | candidate | \`$dg_cand_rate\` | $dg_cand_sample | > | baseline | \`$dg_base_rate\` | $dg_base_sample |" + elif [ "$triage" = "UNRESOLVED" ]; then + note="> ⚠️ **UNRESOLVED (gate blind on a ring member, #1224)** — at least one ring member is an ADR-0007 collapsed repo (its per-role workflow is gone, \`$(_ingress_workflow)\` is present) and this agent's runs there could not be attributed to its role job. That member is **not** counted as passing evidence, so the gate holds instead of promoting on incomplete evidence. This is **not** a detected run failure (cumulative failures: $cum_fail). Fix: register the agent's \`ingress_job\` (its job key in the ingress) in \`standards/canary-rings.json\`. If the evidence below says job reads were capped, the ingress has more runs in the window than \`CANARY_INGRESS_JOBS_MAX\` allows: raise that knob (default 300) instead. This issue auto-closes once every member resolves." + elif [ "$triage" = "FLAG_ERROR" ]; then + note="> ⚠️ **FLAG_ERROR (gate held — runner temp dir unwritable)** — the candidate's cut date resolved fine, but the gate could not create its unresolved-member flag file (\`TMPDIR\` unwritable or full), so it could not record unattributable ring members and fails closed rather than promote on possibly incomplete evidence. This is **not** a detected run failure (cumulative failures: $cum_fail). Fix: free space / make \`TMPDIR\` writable on the runner; this clears on the next tick." elif [ "$triage" = "PRE_EXISTING" ]; then note="> ⚠️ **PRE_EXISTING** — the failure is pre-existing/environmental (reusable byte-identical to the prior channel). Report only; the gate will not roll back or advance. Fix-forward, and the armed timer auto-promotes once clean." else @@ -1858,6 +2218,8 @@ $(printf '%s\n' "$guidance" | sed 's/^/> /')" # are NOT reliable and the triage verdict rests on the reusable diff alone. Prepend a banner # (before the triage note) and blank out the misleading "0" cumulative-failures cell. local cum_row="**$cum_fail** (startup_failures: $cum_startup)" + local evidence_heading="Failing runs in the per-candidate window" + [ "$triage" = "UNRESOLVED" ] && evidence_heading="Unresolved ring members" if [ "$data_gap" = "2" ]; then cum_row="_unknown — channel tags unresolved this tick_" note="> ⚠️ **TAG LOOKUP FAILED — FAILING CLOSED.** A channel-tag lookup for this pair errored this tick (a GitHub API failure — see the workflow log — not an absent tag), so the gate cannot tell where the rings are. Rather than report a false \"fully rolled out\", it holds the promotion and keeps this issue open. This clears automatically once the tags resolve again and the gate re-evaluates (#1225)." @@ -1882,7 +2244,7 @@ $note" $note -### Failing runs in the per-candidate window +### $evidence_heading $evidence --- _Whole-fleet status is in the Canary Rollout workflow run's job summary (Actions → Canary Rollout → latest run → Summary)._ @@ -2169,6 +2531,10 @@ cmd_sync_issues() { evidence="_(⚠️ a channel-tag lookup failed this tick, so this pair's rings could not be resolved and no failing runs can be attributed. The gate FAILS CLOSED: the promotion is held and this issue stays open until the tags resolve again.)_" elif [ "$bl_datagap" = "1" ]; then evidence="_(⚠️ run-history fetch failed this tick — the failing runs could not be listed. The gate FAILS CLOSED: the promotion is held and this issue stays open until run history is readable again and the gate can re-evaluate.)_" + elif [ "$bl_triage" = "UNRESOLVED" ]; then + evidence="$(_unresolved_evidence "$agent" "$bl_cand" || true)" + elif [ "$bl_triage" = "FLAG_ERROR" ]; then + evidence="_(the gate could not create its unresolved-member flag file in \`TMPDIR\`, so no run evidence was gathered this tick. It FAILS CLOSED and holds this pair until the temp dir is writable.)_" else if [ "$bl_cand" = "-" ]; then evidence="_(the source ring's commit is unresolvable this tick, so there is no candidate whose failing runs can be listed. The gate FAILS CLOSED and holds this pair until the tag resolves.)_" diff --git a/standards/canary-rings.json b/standards/canary-rings.json index b86f572f..8af04830 100644 --- a/standards/canary-rings.json +++ b/standards/canary-rings.json @@ -17,11 +17,17 @@ "personas/" ], "_autocut_note": "autocut (#1069): `canary-rollout.sh autocut` (gated by org var CANARY_AUTO_CUT) auto-cuts a new immutable /vX.Y.Z + moves `next` when a reusable's blob on its host main HEAD differs from the current `next` candidate. Since #712 (epic #1083 pillar 2) the bump is DETECTED from the change: a conventional-commit `!`/`BREAKING CHANGE` or a `workflow_call` interface break (removed/renamed input|secret, or a newly-required input) → major (seeding a fresh v-next per #657 F4); a non-breaking `feat` → minor; else patch. The optional `.agents[].autocut.bump` knob (patch|minor|major) is now an OVERRIDE that forces a level (not the sole source); it fails safe to patch on any signal-fetch error.", + "_ingress_note": "ingress / ingress_job (#1224, ADR-0007): a collapsed repo replaces its per-role Class-1 caller stubs with ONE `.github/workflows/agent-ingress.yml` whose display name is `ingress.workflow`, carrying one job per role. There the per-role `run_workflow` no longer exists, so the gate resolves an agent's runs as the ingress runs in which its role job ran: `.agents[].ingress_job` is that job KEY (the token before \" / \" in the jobs API name, e.g. `dev-lead`). Per member: per-role workflow present \u2192 legacy; absent but ingress present \u2192 attribute by `ingress_job` (role-scoped conclusion + failed-step signature, so benign/suspect classes keyed on `run_workflow` still match); neither \u2192 not a consumer ([], #747). An ingress member that cannot be attributed (no `ingress_job`, a job-less ingress run that is neither cancelled nor a startup_failure, unreadable jobs, or an ingress run carrying no job for the role that is NOT older than the role's first appearance in the window) is reported UNRESOLVED and holds the gate BLOCKED (triage UNRESOLVED) \u2014 never silent passing evidence. Not UNRESOLVED: a job-less run that was cancelled before any job started (no record at all); a job-less startup_failure (no job ran, so it hit every role and is recorded as a failure for each, not as blind); and a no-role run older than the role's first appearance (the role was added to the ingress after it). Every collapsed member's ingress must carry a job for EVERY ring-registered role: a member whose ingress lacks the role is UNRESOLVED (the gate cannot tell a non-consumer from a misspelled `ingress_job`) until it carries the job or leaves the ring. The attribution rules mirror .github-private scripts/lib/run-attribution.sh (#1727), duplicated deliberately because that lib is not importable from this checkout.", + "ingress": { + "workflow": "Agent Ingress", + "file": ".github/workflows/agent-ingress.yml" + }, "agents": { "dev-lead": { "host": "petry-projects/.github-private", "reusable": ".github/workflows/dev-lead-reusable.yml", "run_workflow": "Dev-Lead Agent", + "ingress_job": "dev-lead", "rings": [ { "channel": "next", @@ -502,6 +508,7 @@ "host": "petry-projects/.github", "reusable": ".github/workflows/pr-review-mention-reusable.yml", "run_workflow": "PR Review \u2014 Mention Trigger", + "ingress_job": "pr-review-mention", "rings": [ { "channel": "next", @@ -691,6 +698,7 @@ "host": "petry-projects/.github-private", "reusable": ".github/workflows/ci-failure-analyst-reusable.yml", "run_workflow": "CI Failure Analyst", + "ingress_job": "ci-failure-analyst", "rings": [ { "channel": "next", @@ -950,6 +958,7 @@ "host": "petry-projects/.github", "reusable": ".github/workflows/pr-auto-review-reusable.yml", "run_workflow": "PR Auto-Review \u2014 Ready Check", + "ingress_job": "pr-auto-review", "rings": [ { "channel": "next", @@ -1085,6 +1094,7 @@ "reusable": ".github/workflows/pr-review.yml", "caller_stub": ".github/workflows/pr-review-trigger.yml", "run_workflow": "PR Review Agent — Trigger", + "ingress_job": "pr-review", "rings": [ { "channel": "next", diff --git a/tests/canary_rollout.bats b/tests/canary_rollout.bats index f0542bc5..7ad77005 100644 --- a/tests/canary_rollout.bats +++ b/tests/canary_rollout.bats @@ -704,6 +704,27 @@ GHEOF [ "$(wc -l < "$CALLS")" -eq 1 ] } +@test "_repo_wf_runs_cached: the run limit is part of the cache key — a limit-1000 hit never serves a limit-5000 read (#1224)" { + STUB_BIN="$(mktemp -d "$BATS_TEST_TMPDIR/stub.XXXXXX")"; export PATH="$STUB_BIN:$PATH" + export CALLS="$BATS_TEST_TMPDIR/limit-calls"; : > "$CALLS" + cat > "$STUB_BIN/gh" <<'GHEOF' +#!/usr/bin/env bash +echo "$*" >> "$CALLS" +echo '[{"conclusion":"success","createdAt":"2026-01-10T00:00:00Z","databaseId":1,"workflowName":"W"}]' +GHEOF + chmod +x "$STUB_BIN/gh" + run env _RUNS_CACHE_DIR="$BATS_TEST_TMPDIR/rc-limit" bash -c ' + mkdir -p "$_RUNS_CACHE_DIR"; source "'"$ORCH"'" + _repo_wf_runs_cached some/repo W 0 1000 >/dev/null + _repo_wf_runs_cached some/repo W 0 1000 >/dev/null + _repo_wf_runs_cached some/repo W 0 5000 >/dev/null + ' + [ "$status" -eq 0 ] + # Same limit is served from cache (1 fetch); the larger limit is a distinct entry (2nd fetch). + [ "$(wc -l < "$CALLS")" -eq 2 ] + grep -q -- "-L 5000" "$CALLS" +} + @test "_run_json: cache key is collision-free — 'A B' vs 'A/B' workflows don't share a file (#835 CodeRabbit)" { STUB_BIN="$(mktemp -d "$BATS_TEST_TMPDIR/stub.XXXXXX")"; export PATH="$STUB_BIN:$PATH" # gh echoes back the requested workflow name so we can prove which cache entry served the call. @@ -5430,3 +5451,449 @@ GITEOF [ "$output" = "d1d1d1d1d1d1d1d1d1d1d1d1d1d1d1d1d1d1d1d1" ] [ "$output" != "b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0b0" ] } + +# ── ADR-0007 collapsed repos: run attribution by ingress JOB (#1224) ────────── +# A collapsed repo replaces its per-role caller stubs with ONE `Agent Ingress` workflow +# carrying one job per role. `gh run list --workflow "Dev-Lead Agent"` then finds nothing +# there, so the gate must fall back to the ingress and attribute each run to the agent's +# role job (registry `ingress_job`). A member whose runs cannot be attributed is reported +# UNRESOLVED and never counts as passing evidence. +# +# _ingress_stub — gh stub for three repos: +# org/collapsed — no per-role workflows; `Agent Ingress` runs 101,102,104 +# org/legacy — `Dev-Lead Agent` run 201 (no ingress) +# org/none — neither workflow (a non-consumer, #747) +# org/nojobs — `Agent Ingress` run 103 whose jobs list is empty (unattributable) +# Every gh invocation is appended to $GH_LOG. +_ingress_stub() { + # _frontier_state keeps its per-process UNRESOLVED flag under $TMPDIR; scope it to this test. + export TMPDIR="$BATS_TEST_TMPDIR" + STUB_BIN="$(mktemp -d "$BATS_TEST_TMPDIR/stub.XXXXXX")"; export PATH="$STUB_BIN:$PATH" + export GH_LOG="$BATS_TEST_TMPDIR/gh-ingress.log"; : > "$GH_LOG" + cat > "$STUB_BIN/gh" <<'GHEOF' +#!/usr/bin/env bash +echo "$*" >> "$GH_LOG" +repo=""; wf=""; prev="" +for a in "$@"; do + [ "$prev" = "--repo" ] && repo="$a" + [ "$prev" = "--workflow" ] && wf="$a" + prev="$a" +done +nf() { echo "could not find any workflows named $wf" >&2; exit 1; } +case "$1 $2" in + "run list") + case "$repo|$wf" in + "org/collapsed|Agent Ingress") + echo '[{"conclusion":"success","createdAt":"2026-01-02T00:00:00Z","databaseId":101,"workflowName":"Agent Ingress"}, + {"conclusion":"failure","createdAt":"2026-01-02T01:00:00Z","databaseId":102,"workflowName":"Agent Ingress"}, + {"conclusion":"success","createdAt":"2026-01-02T02:00:00Z","databaseId":104,"workflowName":"Agent Ingress"}, + {"conclusion":null,"createdAt":"2026-01-02T03:00:00Z","databaseId":105,"workflowName":"Agent Ingress"}]' ;; + "org/nojobs|Agent Ingress") + echo '[{"conclusion":"failure","createdAt":"2026-01-02T00:00:00Z","databaseId":103,"workflowName":"Agent Ingress"}]' ;; + "org/skipbusy|Agent Ingress") + # Three runs today in which the dev-lead role job was SKIPPED (other roles' events fired the + # shared ingress); the cap leaves the newest-but-one unread, so nothing countable is observed. + t="$(date -u +%Y-%m-%d)" + echo "[{\"conclusion\":\"success\",\"createdAt\":\"${t}T10:00:00Z\",\"databaseId\":321,\"workflowName\":\"Agent Ingress\"}, + {\"conclusion\":\"success\",\"createdAt\":\"${t}T09:00:00Z\",\"databaseId\":320,\"workflowName\":\"Agent Ingress\"}, + {\"conclusion\":\"success\",\"createdAt\":\"${t}T08:00:00Z\",\"databaseId\":319,\"workflowName\":\"Agent Ingress\"}]" ;; + "org/preadopt|Agent Ingress") + echo '[{"conclusion":"success","createdAt":"2026-01-03T00:00:00Z","databaseId":411,"workflowName":"Agent Ingress"}, + {"conclusion":"success","createdAt":"2026-01-02T00:00:00Z","databaseId":410,"workflowName":"Agent Ingress"}]' ;; + "org/rolegap|Agent Ingress") + echo '[{"conclusion":"success","createdAt":"2026-01-02T00:00:00Z","databaseId":421,"workflowName":"Agent Ingress"}, + {"conclusion":"success","createdAt":"2026-01-03T00:00:00Z","databaseId":422,"workflowName":"Agent Ingress"}]' ;; + "org/norole|Agent Ingress") + echo '[{"conclusion":"success","createdAt":"2026-01-02T00:00:00Z","databaseId":431,"workflowName":"Agent Ingress"}]' ;; + "org/cancelled|Agent Ingress") + echo '[{"conclusion":"cancelled","createdAt":"2026-01-02T00:00:00Z","databaseId":441,"workflowName":"Agent Ingress"}]' ;; + "org/tie|Agent Ingress") + echo '[{"conclusion":"failure","createdAt":"2026-01-02T00:00:00Z","databaseId":451,"workflowName":"Agent Ingress"}, + {"conclusion":"failure","createdAt":"2026-01-03T00:00:00Z","databaseId":452,"workflowName":"Agent Ingress"}]' ;; + "org/sfgap|Agent Ingress") + echo '[{"conclusion":"startup_failure","createdAt":"2026-01-03T00:00:00Z","databaseId":461,"workflowName":"Agent Ingress"}, + {"conclusion":"success","createdAt":"2026-01-02T00:00:00Z","databaseId":462,"workflowName":"Agent Ingress"}]' ;; + "org/busy|Agent Ingress") + # Three attributable runs: two today, one yesterday (newest first once sorted by createdAt). + t="$(date -u +%Y-%m-%d)"; y="$(date -u -d yesterday +%Y-%m-%d 2>/dev/null || date -u -v-1d +%Y-%m-%d)" + echo "[{\"conclusion\":\"success\",\"createdAt\":\"${t}T10:00:00Z\",\"databaseId\":303,\"workflowName\":\"Agent Ingress\"}, + {\"conclusion\":\"success\",\"createdAt\":\"${t}T09:00:00Z\",\"databaseId\":302,\"workflowName\":\"Agent Ingress\"}, + {\"conclusion\":\"success\",\"createdAt\":\"${y}T12:00:00Z\",\"databaseId\":301,\"workflowName\":\"Agent Ingress\"}]" ;; + "org/legacy|Dev-Lead Agent") + echo '[{"conclusion":"success","createdAt":"2026-01-02T00:00:00Z","databaseId":201,"workflowName":"Dev-Lead Agent"}]' ;; + "org/legacy|CI Failure Analyst") + echo '[{"conclusion":"success","createdAt":"2026-01-02T00:00:00Z","databaseId":202,"workflowName":"CI Failure Analyst"}]' ;; + *) nf ;; + esac ;; + "run view") + # Transient/systemic jobs-endpoint failure (5xx) — NOT a permanent 404. + [ -n "${STUB_JOBS_FAIL:-}" ] && { echo "HTTP 502: Bad Gateway" >&2; exit 1; } + case "$3" in + 30[123]) echo '{"jobs":[{"name":"dev-lead / run","conclusion":"success","steps":[]}]}' ;; + 411|421) echo '{"jobs":[{"name":"dev-lead / run","conclusion":"success","steps":[]}]}' ;; + 410|422|431) echo '{"jobs":[{"name":"pr-review / review","conclusion":"success","steps":[]}]}' ;; + 441) echo '{"jobs":[]}' ;; + 461) echo '{"jobs":[]}' ;; + 462) echo '{"jobs":[{"name":"pr-review / review","conclusion":"success","steps":[]}]}' ;; + # A failed job and an action_required job of the SAME role, in both orders (max_by tie-break). + 451) echo '{"jobs":[{"name":"dev-lead / build","conclusion":"failure","steps":[]},{"name":"dev-lead / approve","conclusion":"action_required","steps":[]}]}' ;; + 452) echo '{"jobs":[{"name":"dev-lead / approve","conclusion":"action_required","steps":[]},{"name":"dev-lead / build","conclusion":"failure","steps":[]}]}' ;; + 32[01]) echo '{"jobs":[{"name":"dev-lead / run","conclusion":"skipped","steps":[]},{"name":"pr-review / review","conclusion":"success","steps":[]}]}' ;; + 101) echo '{"jobs":[{"name":"dev-lead / run","conclusion":"success","steps":[]}, + {"name":"pr-review / review","conclusion":"failure","steps":[{"name":"Push","conclusion":"failure"}]}, + {"name":"ci-failure-analyst","conclusion":"skipped","steps":[]}]}' ;; + 102) echo '{"jobs":[{"name":"dev-lead / setup","conclusion":"success","steps":[]}, + {"name":"dev-lead / run","conclusion":"failure","steps":[{"name":"Build","conclusion":"failure"}]}, + {"name":"pr-review / review","conclusion":"failure","steps":[{"name":"Push","conclusion":"failure"}]}, + {"name":"ci-failure-analyst","conclusion":"skipped","steps":[]}]}' ;; + 103) echo '{"jobs":[]}' ;; + 104) echo '{"jobs":[{"name":"dev-lead / run","conclusion":"skipped","steps":[]}, + {"name":"ci-failure-analyst / analyse","conclusion":"success","steps":[]}]}' ;; + *) echo "run $3 not found" >&2; exit 1 ;; + esac ;; + *) echo '{}' ;; +esac +GHEOF + chmod +x "$STUB_BIN/gh" + # Test registry: dev-lead (ingress_job registered) + `noingress` (same per-role workflow, NO + # ingress_job) + `cfa` (ingress_job ci-failure-analyst). Each takes members from its rings. + INGRESS_RINGS="$BATS_TEST_TMPDIR/ingress-rings.json" + jq '{org_infra_repos, ingress, + agents: { + "dev-lead": (.agents["dev-lead"] | .ingress_job = "dev-lead"), + "noingress": (.agents["dev-lead"] | del(.ingress_job) | .gate.benign_failure_classes = [] + | .gate.suspect_failure_classes = [] | del(.gate.correctness) + | .rings = [{"channel":"next","order":0,"members":["org/collapsed"]}, + {"channel":"ring0","order":1,"members":["org/legacy"]}] + | .gate.transitions = {"next->ring0":{"dwell_hours":0,"waive_sample":true}}), + "cfa": (.agents["dev-lead"] | .ingress_job = "ci-failure-analyst" | .run_workflow = "CI Failure Analyst" | .gate.benign_failure_classes = [] + | .gate.suspect_failure_classes = [] | del(.gate.correctness) + | .rings = [{"channel":"next","order":0,"members":["org/collapsed"]}, + {"channel":"ring0","order":1,"members":["org/legacy"]}] + | .gate.transitions = {"next->ring0":{"dwell_hours":0,"waive_sample":true}}) + }}' "$RINGS" > "$INGRESS_RINGS" + export INGRESS_RINGS +} + +@test "_agent_run_json: a collapsed repo (ingress role job, no per-role workflow) resolves by JOB (#1224)" { + _ingress_stub + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 \ + bash -c "source '$ORCH' && _agent_run_json dev-lead org/collapsed '' | jq -c 'sort_by(.databaseId)|map([.databaseId,.conclusion,.workflowName,.role])'" + [ "$status" -eq 0 ] + # 101 → dev-lead success; 102 → dev-lead failure (worst-outcome over its two jobs); + # 104 → dev-lead skipped = did not run (no record); 105 → in flight (no record). + [ "$output" = '[[101,"success","Dev-Lead Agent","dev-lead"],[102,"failure","Dev-Lead Agent","dev-lead"]]' ] +} + +@test "_agent_run_json: a non-collapsed repo still resolves by workflow name and never queries the ingress (#1224)" { + _ingress_stub + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 \ + bash -c "source '$ORCH' && _agent_run_json dev-lead org/legacy '' | jq -c 'map(.databaseId)'" + [ "$status" -eq 0 ] + [ "$output" = "[201]" ] + run grep -q "Agent Ingress" "$GH_LOG" + [ "$status" -eq 1 ] +} + +@test "_agent_run_json: a repo with neither workflow nor ingress is a non-consumer [] — not unresolved (#747 preserved, #1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _agent_run_json noingress org/none ''" + [ "$status" -eq 0 ] + [ "$output" = "[]" ] + [ ! -s "$flag" ] +} + +@test "_agent_run_json: ingress present but the agent has no ingress_job → member reported UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _agent_run_json noingress org/collapsed ''" + [ "$status" -eq 0 ] + [[ "$output" == *"UNRESOLVED"* ]] + [[ "$output" == *"org/collapsed"* ]] + grep -q "^org/collapsed" "$flag" +} + +@test "_agent_run_json: CANARY_INGRESS_JOBS_MAX caps newest-first job reads and flags the member UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 CANARY_INGRESS_JOBS_MAX=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "set -o pipefail; source '$ORCH' && _agent_run_json dev-lead org/collapsed '' 2>/dev/null | jq -c 'map(.databaseId)'" + [ "$status" -eq 0 ] + # Newest completed run is 104 (dev-lead skipped → no record); the cap then stops further reads. + [ "$output" = '[]' ] + grep -q "more than 1 ingress runs" "$flag" + grep -q "run view 104 " "$GH_LOG" + ! grep -q "run view 102 " "$GH_LOG" + ! grep -q "run view 101 " "$GH_LOG" +} + +@test "_baseline_daily: a capped ingress read is a valid truncated sample of the NEWEST days, not UNRESOLVED (#1224 liveness)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + # org/busy has 3 attributable runs (2 today, 1 yesterday); the cap of 2 leaves yesterday's run + # unread. The baseline must cover TODAY only (yesterday and older are unknown, not zero) and the + # member must NOT be flagged UNRESOLVED — a busy collapsed repo may not hold the gate forever. + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 CANARY_INGRESS_JOBS_MAX=2 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _baseline_daily dev-lead 3 org/busy 2>/dev/null" + [ "$status" -eq 0 ] + [ "$output" = "2" ] + [ ! -s "$flag" ] +} + +@test "_baseline_daily: more runs on the NEWEST day than the cap keeps that day's partial count, never an empty baseline (#1224 liveness)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + # Cap of 1: only run 303 (today) is read; the first unread run (302) is also today, so today is the + # boundary day. Its partial count (1) is a lower bound and must be kept — dropping it would leave an + # empty baseline that waive_sample_if_no_caller reads as "no caller". + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 CANARY_INGRESS_JOBS_MAX=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _baseline_daily dev-lead 3 org/busy 2>/dev/null" + [ "$status" -eq 0 ] + [ "$output" = "1" ] + [ ! -s "$flag" ] +} + +@test "_baseline_daily: a truncated read that observed NOTHING countable is a non-zero floor, never 'no caller' (#1224 liveness)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 CANARY_INGRESS_JOBS_MAX=2 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _baseline_daily dev-lead 3 org/skipbusy 2>/dev/null" + [ "$status" -eq 0 ] + [ "$output" = "1" ] + [ ! -s "$flag" ] +} + +@test "_baseline_daily: truncation is per repo — a fully-read member keeps its older days known beside a capped one (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + # org/busy is capped to TODAY (older days unknown for it); org/legacy is read completely, so the + # older days are known (zero) for the aggregate and must not be dropped by busy's cutoff. + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 CANARY_INGRESS_JOBS_MAX=2 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _baseline_daily dev-lead 3 org/busy org/legacy 2>/dev/null" + [ "$status" -eq 0 ] + [ "$output" = "2 0 0" ] + [ ! -s "$flag" ] +} + +@test "_baseline_daily: an UNCAPPED baseline still reports every day (zero-filled), nothing dropped (#1224 liveness)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 CANARY_INGRESS_JOBS_MAX=300 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _baseline_daily dev-lead 3 org/busy 2>/dev/null" + [ "$status" -eq 0 ] + [ "$output" = "2 1 0" ] + [ ! -s "$flag" ] +} + +@test "_agent_run_json: the jobs-read circuit breaker stops after the first exhausted 5xx instead of retrying every run (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env STUB_JOBS_FAIL=1 CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _agent_run_json dev-lead org/collapsed '' 2>/dev/null" + [ "$status" -eq 0 ] + # org/collapsed has three completed runs (104, 102, 101); without the breaker all three are read. + [ "$(grep -c '^run view ' "$GH_LOG")" -eq 1 ] + grep -q "jobs endpoint unreadable" "$flag" +} + +@test "_agent_run_json: an ingress run from BEFORE the role was added to the ingress is not-yet-adopted, not UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + # 410 (older) has only pr-review's job; 411 (newer) carries dev-lead's. The role's oldest appearance is + # 411, so 410 predates adoption and must not hold a correctly configured member BLOCKED for ~14 days. + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "set -o pipefail; source '$ORCH' && _agent_run_json dev-lead org/preadopt '' 2>/dev/null | jq -c 'map(.databaseId)'" + [ "$status" -eq 0 ] + [ "$output" = '[411]' ] + [ ! -s "$flag" ] +} + +@test "_agent_run_json: a no-role ingress run NEWER than the role's first appearance is still UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "set -o pipefail; source '$ORCH' && _agent_run_json dev-lead org/rolegap '' 2>/dev/null | jq -c 'map(.databaseId)'" + [ "$status" -eq 0 ] + [ "$output" = '[421]' ] + grep -q "carry no 'dev-lead' job" "$flag" +} + +@test "_agent_run_json: when NO run carries the role (ingress_job renamed/misspelled) the member is UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "set -o pipefail; source '$ORCH' && _agent_run_json dev-lead org/norole '' 2>/dev/null | jq -c 'map(.databaseId)'" + [ "$status" -eq 0 ] + [ "$output" = '[]' ] + grep -q "carry no 'dev-lead' job" "$flag" +} + +@test "_agent_run_json: a failed job plus an action_required job of the same role stays a FAILURE in either order (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + # jq max_by keeps the last of tied elements; with failure and action_required ranked equal, the role's + # conclusion depended on job order and a real failure could be dropped from cum_fail. + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "set -o pipefail; source '$ORCH' && _agent_run_json dev-lead org/tie '' 2>/dev/null | jq -c 'map(.conclusion)'" + [ "$status" -eq 0 ] + [ "$output" = '["failure","failure"]' ] + [ ! -s "$flag" ] +} + +@test "_agent_run_json: a job-less startup_failure is not the role's first appearance — an older no-role run stays UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + # 461 (newer) has no jobs at all (startup_failure, so it counts as a failure); 462 (older) has only another + # role's job. No run carries dev-lead's job, so 462 is blind (misspelled ingress_job?), not pre-adoption. + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "set -o pipefail; source '$ORCH' && _agent_run_json dev-lead org/sfgap '' 2>/dev/null | jq -c 'map([.databaseId,.conclusion])'" + [ "$status" -eq 0 ] + [ "$output" = '[[461,"startup_failure"]]' ] + grep -q "carry no 'dev-lead' job" "$flag" +} + +@test "_run_signature and _run_decision_class make ONE gh call on a failing lookup, not the ingress retry loop (#1224 regression)" { + _ingress_stub + # Before #1224 these legacy single-call readers did one `gh run view` and failed fast. A persistently + # failing lookup (410 expired logs, 403/rate limit) must not cost the 6-attempt backoff per run. + run env STUB_JOBS_FAIL=1 CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=6 \ + bash -c "source '$ORCH'; set +e; _run_signature org/collapsed 101 '' >/dev/null 2>&1; _run_decision_class org/collapsed 102 dev-lead '' >/dev/null 2>&1; true" + [ "$status" -eq 0 ] + [ "$(grep -c '^run view ' "$GH_LOG")" -eq 2 ] +} + +@test "_run_jobs_json: the ingress path keeps the bounded retry (CANARY_GH_RETRIES attempts) on a transient 5xx (#1224)" { + _ingress_stub + run env STUB_JOBS_FAIL=1 CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=3 \ + bash -c "source '$ORCH'; set +e; _run_jobs_json org/collapsed 101 >/dev/null 2>&1; echo rc=\$?" + [[ "$output" == *"rc=1"* ]] + [ "$(grep -c '^run view ' "$GH_LOG")" -eq 3 ] +} + +@test "_agent_run_json: a job-less CANCELLED ingress run never executed the role — no record, not UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "set -o pipefail; source '$ORCH' && _agent_run_json dev-lead org/cancelled '' 2>/dev/null | jq -c 'map(.databaseId)'" + [ "$status" -eq 0 ] + [ "$output" = '[]' ] + [ ! -s "$flag" ] +} + +@test "_agent_run_json: an ingress run with NO jobs cannot be attributed → member reported UNRESOLVED (#1224)" { + _ingress_stub + local flag="$BATS_TEST_TMPDIR/unresolved"; : > "$flag" + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 _CANARY_UNRESOLVED_FLAG="$flag" \ + bash -c "source '$ORCH' && _agent_run_json dev-lead org/nojobs ''" + [ "$status" -eq 0 ] + grep -q "^org/nojobs" "$flag" +} + +@test "_cumulative_health: an ingress role's failure counts, another role's failure in the same run does not leak in (#1224)" { + _ingress_stub + # differs=0 activates dev-lead's [Pp]ush benign class. Run 102's dev-lead job failed at + # 'Build'; only pr-review's job failed at 'Push'. A run-wide signature would wrongly see + # 'Push' and excuse dev-lead's failure as benign — the role-scoped signature must not. + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 \ + bash -c "source '$ORCH' && _cumulative_health dev-lead '' 0 - org/collapsed" + [ "$status" -eq 0 ] + [ "$output" = "1 0 0 0 0" ] +} + +@test "_tier_sample: a MIXED ring (collapsed + legacy member) counts both in one evaluation (#1224)" { + _ingress_stub + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 \ + bash -c "source '$ORCH' && _tier_sample dev-lead '' org/collapsed org/legacy" + [ "$status" -eq 0 ] + # collapsed: 101 + 102 executed; legacy: 201. + [ "$output" = "3 2026-01-02T00:00:00Z" ] +} + +# _frontier_state with tag resolution stubbed out (candidate on next only; cut before the runs). +_ingress_frontier() { + local agent="$1" + env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 bash -c " + source '$ORCH' + channel_commit() { case \"\$2\" in next) echo cand ;; *) echo prior ;; esac; } + candidate_cut_date() { echo 2026-01-01T00:00:00Z; } + _reusable_differs() { echo 0; } + _frontier_state $agent" +} + +@test "_frontier_state: an uncreatable unresolved flag fails closed as FLAG_ERROR with a full 18-field line (#1224)" { + _ingress_stub + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 bash -c " + source '$ORCH' + channel_commit() { case \"\$2\" in next) echo cand ;; *) echo prior ;; esac; } + candidate_cut_date() { echo 2026-01-01T00:00:00Z; } + _reusable_differs() { echo 0; } + _unresolved_flag_path() { echo '$BATS_TEST_TMPDIR/no-such-dir/flag'; } # cannot be created + _frontier_state cfa 2>/dev/null" + [ "$status" -eq 0 ] + local line; line="$(grep ' FLAG_ERROR ' <<< "$output" | head -1)" + [ -n "$line" ] + [[ "$line" == *" BLOCKED "* ]] + # cmd_sync_issues appends the datagap field, so a short line would shift it into `downgrade`. + [ "$(wc -w <<< "$line")" -eq 18 ] +} + +@test "_frontier_state: a MIXED ring that fully resolves gates normally (collapsed + legacy → PROMOTE) (#1224)" { + _ingress_stub + run _ingress_frontier cfa + [ "$status" -eq 0 ] + read -r _c frontier transition state _rest <<< "$(printf '%s\n' "$output" | tail -1)" + [ "$frontier" = "ring0" ]; [ "$state" = "PROMOTE" ] + [[ "$output" != *"UNRESOLVED"* ]] +} + +@test "_frontier_state: an UNRESOLVED member fails the gate loudly — BLOCKED/UNRESOLVED, never PROMOTE (#1224)" { + _ingress_stub + run _ingress_frontier noingress + [ "$status" -eq 0 ] + # The member is named in an ::error:: annotation… + [[ "$output" == *"::error::"*"org/collapsed"* ]] + # …and the state line holds the promotion with triage UNRESOLVED. + local line; line="$(printf '%s\n' "$output" | tail -1)" + read -r _c frontier transition state _d _f _s _t _cf _cs _cb triage _rest <<< "$line" + [ "$state" = "BLOCKED" ] + [ "$triage" = "UNRESOLVED" ] +} + +@test "_blocker_body: UNRESOLVED triage explains the blind member, not a cut-date indeterminate (#1224)" { + run bash -c "source '$ORCH' && _blocker_body dev-lead 'next->ring0' cand 0 0 UNRESOLVED petry-projects/.github-private '- \`org/collapsed\` — ingress present'" + [ "$status" -eq 0 ] + [[ "$output" == *"UNRESOLVED"* ]] + [[ "$output" == *"ingress_job"* ]] + [[ "$output" != *"cut date unresolved"* ]] +} + +@test "registry: every ADR-0007-collapsed role carries ingress_job = its ingress job key, documented (#1224)" { + [ "$(jq -r '.ingress.workflow' "$RINGS")" = "Agent Ingress" ] + [ -n "$(jq -r '._ingress_note // empty' "$RINGS")" ] + local a + for a in dev-lead pr-review-mention pr-auto-review pr-review ci-failure-analyst; do + [ "$(jq -r --arg a "$a" '.agents[$a].ingress_job // empty' "$RINGS")" = "$a" ] + done +} + +@test "_unresolved_evidence: the blocker issue names each blind member and why (#1224)" { + _ingress_stub + run env CANARY_RINGS="$INGRESS_RINGS" CANARY_GH_RETRY_SLEEP=0 CANARY_GH_RETRIES=1 bash -c " + source '$ORCH' + candidate_cut_date() { echo 2026-01-01T00:00:00Z; } + _unresolved_evidence noingress cand" + [ "$status" -eq 0 ] + [[ "$output" == *'`org/collapsed`'*"ingress_job"* ]] + # The resolvable legacy member is not listed. + [[ "$output" != *"org/legacy"* ]] +} + +@test "_record_unresolved: a failed evidence append removes the flag so the gate cannot read it as clean (#1224)" { + # Force the append to fail via a printf shim, not file modes (a root runner bypasses chmod 444). + local flag="$BATS_TEST_TMPDIR/unresolved-ro"; : > "$flag" + run env _CANARY_UNRESOLVED_FLAG="$flag" ORCH="$ORCH" bash -c 'source "$ORCH" && printf() { if [ "$1" = "%s\n" ]; then return 1; fi; builtin printf "$@"; } && _record_unresolved dev-lead org/x blind 2>/dev/null' + [ ! -e "$flag" ] +}