From 22e84f01eae8f3d4e4d517eaee5f22ba8fb0e399 Mon Sep 17 00:00:00 2001 From: "J. Yaunches" Date: Fri, 24 Jul 2026 14:17:51 -0400 Subject: [PATCH 1/4] feat(e2e): require exact pre-tag qualification Signed-off-by: J. Yaunches --- .../SKILL.md | 30 ++- .../skills/nemoclaw-maintainer-e2e/SKILL.md | 215 +++++++++++++++++ .../agents/openai.yaml | 7 + .../scripts/validate-full-e2e-evidence.mts | 218 ++++++++++++++++++ .../nemoclaw-maintainer-evening/SKILL.md | 18 +- .../references/release-train.md | 10 +- .agents/skills/nemoclaw-skills-guide/SKILL.md | 7 +- .github/workflows/e2e.yaml | 57 ++++- test/e2e/support/e2e-workflow.test.ts | 94 ++++++++ test/maintainer-e2e-skill.test.ts | 178 ++++++++++++++ test/maintainer-skills-policy.test.ts | 48 +++- ...upload-e2e-artifacts-workflow-boundary.mts | 1 + tools/e2e/workflow-boundary.mts | 145 +++++++++++- 13 files changed, 1009 insertions(+), 19 deletions(-) create mode 100644 .agents/skills/nemoclaw-maintainer-e2e/SKILL.md create mode 100644 .agents/skills/nemoclaw-maintainer-e2e/agents/openai.yaml create mode 100644 .agents/skills/nemoclaw-maintainer-e2e/scripts/validate-full-e2e-evidence.mts create mode 100644 test/maintainer-e2e-skill.test.ts diff --git a/.agents/skills/nemoclaw-maintainer-cut-release-tag/SKILL.md b/.agents/skills/nemoclaw-maintainer-cut-release-tag/SKILL.md index f7ef87f3a55..802b80e0efc 100644 --- a/.agents/skills/nemoclaw-maintainer-cut-release-tag/SKILL.md +++ b/.agents/skills/nemoclaw-maintainer-cut-release-tag/SKILL.md @@ -40,6 +40,7 @@ The downstream scheduled reconciliation remains available if the event-driven di - Treat the dated MDX entry as the canonical release history. A conventional Release Notes page or post-tag Announcement draft cannot replace it. - If `origin/main` changes after plan generation, regenerate the plan before cutting the tag. - Before asking for release confirmation, satisfy the canonical [pre-tag E2E evidence policy](../nemoclaw-maintainer-policies/references/release-train.md#pre-tag-e2e-evidence) for that commit. +- Load `nemoclaw-maintainer-e2e` and run full mode when the candidate has no applicable exact Brev Launchable evidence. - Ask the maintainer to paste the confirmation phrase from the plan before cutting the tag. - Push only the semver tag (`vX.Y.Z`) from the agent-controlled step. - Never push `latest` or `lkg` from this skill. @@ -115,14 +116,36 @@ When the entry is waived, show the recorded waiver reason in the plan presentati For the plan's full `origin/main` SHA, review `.github/workflows/e2e.yaml` at that commit and build the evidence ledger required by the canonical [pre-tag E2E evidence policy](../nemoclaw-maintainer-policies/references/release-train.md#pre-tag-e2e-evidence). The workflow is the sole source of truth; do not substitute or maintain a separate release-gating test list. +Find an applicable full-mode E2E run for the candidate SHA. +If none exists, load `nemoclaw-maintainer-e2e` and dispatch full mode for that SHA. +Do not substitute an ordinary E2E run or a selective `staging-brev-launchable` run. +Require the full-mode run to include the default-enabled suite and `Exact staging Brev Launchable`. + +Before accepting that run, require: + +- the workflow `head_sha` to equal the plan candidate SHA; +- the trusted dispatch receipt to prove empty selectors and `include_staging_brev_launchable=true`; +- the workflow conclusion to be `success`; +- the `Exact staging Brev Launchable` job conclusion to be `success`; +- the job URL and workflow attempt number; +- qualification identity for the same SHA; and +- cleanup evidence that reports the qualified workspace as `ABSENT`. + +Treat a skipped job as missing evidence even when the workflow concludes `success`. +If the plan candidate SHA changes, discard the run and qualification evidence. +Run full mode again for the new candidate SHA. +No release-note-only delta exception is currently defined. + Before showing the confirmation prompt, present: - the candidate SHA; - the number of tests with green evidence out of the number required by the workflow; - each required test mapped to a successful run or job URL and attempt; and -- an itemized maintainer exception for every test without green evidence, including its current result or failure summary and the rationale for proceeding. +- the full-mode workflow URL, `Exact staging Brev Launchable` job URL, attempt, qualification identity, and cleanup result; and +- a separate itemized maintainer exception for each test without successful evidence, including its test identifier, run links, current result, and rationale; and +- a separate itemized maintainer exception for missing or invalid exact Brev Launchable qualification, including run and job URLs, the current result or missing receipt, and rationale. -Do not ask for the phrase until every test has green evidence or an explicit itemized maintainer exception. If `origin/main` moves or the candidate SHA otherwise changes, regenerate the plan and rebuild the ledger for the new SHA. +Do not ask for the phrase until each test and the exact Brev Launchable qualification has successful evidence or its own itemized maintainer exception. If `origin/main` moves or the candidate SHA otherwise changes, regenerate the plan and rebuild the ledger for the new SHA. Ask the maintainer to paste this phrase: @@ -256,6 +279,9 @@ If the Announcement is valid, return its URL with the release artifacts and mark - Plan generation fails: fix the named precondition, then regenerate the plan. - Planned changelog entry is missing or malformed: stop before plan generation and run the pre-tag `nemoclaw-contributor-update-docs` workflow. Use post-release recovery only when the tag already exists. +- Full-mode E2E readiness is disabled: stop before dispatch. Ask a release administrator to complete the protected environment, secrets, staging Launchable ID, and ownership setup, then enable the persistent readiness variable. +- Full-mode E2E ran for another SHA or skipped `Exact staging Brev Launchable`: reject the run and dispatch full mode for the plan candidate SHA. +- Qualification or cleanup evidence is missing or invalid: reject the run. Do not infer qualification from the workflow conclusion. - `origin/main` moved after plan generation: regenerate the plan and ask for the new confirmation phrase. - Remote semver tag already exists: stop; do not retag unless the maintainer explicitly starts protected-tag remediation. - `latest` workflow fails or times out: report the workflow/status; do not move `latest` manually. diff --git a/.agents/skills/nemoclaw-maintainer-e2e/SKILL.md b/.agents/skills/nemoclaw-maintainer-e2e/SKILL.md new file mode 100644 index 00000000000..ff1d4e39a7c --- /dev/null +++ b/.agents/skills/nemoclaw-maintainer-e2e/SKILL.md @@ -0,0 +1,215 @@ +--- +name: nemoclaw-maintainer-e2e +description: Dispatches and verifies trusted GitHub Actions E2E for NemoClaw maintainers. Use for requests such as run the E2E suite, run the full E2E suite, deploy pre-release full E2E, run pre-tag full E2E, or run release-candidate E2E. +--- + + + + +# Run Maintainer E2E + +Use `.github/workflows/e2e.yaml` from trusted `main`. +Do not substitute local `npm run test:live-e2e` unless the maintainer explicitly requests local execution. + +## Select the Mode + +| Request | Mode | `include_staging_brev_launchable` | +|---|---|---| +| “Run the E2E suite” | Ordinary | `false` | +| “Run the full E2E suite” | Full | `true` | +| “deploy pre-release full E2E” | Full | `true` | +| “run pre-tag full E2E” | Full | `true` | +| “run release-candidate E2E” | Full | `true` | + +A generic E2E request must not authorize the protected Brev path. +Do not infer full mode from words such as “all” or “complete.” +Ask for clarification only when the request contains conflicting mode phrases. + +Ordinary mode runs the default-enabled GitHub Actions suite. +Full mode runs that suite and `Exact staging Brev Launchable` in the same workflow run. + +## Resolve the Candidate + +Run from a trusted NemoClaw checkout: + +```bash +gh auth status +git fetch --prune origin main +CANDIDATE_SHA="$(git rev-parse origin/main)" +``` + +For a pre-tag request, use the full candidate SHA from the generated release plan. +Require that SHA to equal `origin/main` before dispatch. +Stop and regenerate the release plan when they differ. + +Record `CANDIDATE_SHA` for every dispatch. +Do not use a relative revision in the evidence report. + +## Check Full-Mode Readiness + +Skip this step in ordinary mode. + +Read only the persistent readiness variable: + +```bash +gh variable get NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED \ + --repo NVIDIA/NemoClaw --json value --jq .value +``` + +Require the value `true`. +If the variable is absent or disabled, do not dispatch. +Report this prerequisite: + +> A release administrator must set `NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED=true` after the protected environment, secrets, staging Launchable ID, and ownership are configured. + +Do not inspect, print, or handle cloud credentials. +Do not change the readiness variable around a run. + +## Dispatch One Trusted Run + +Generate a unique correlation ID: + +```bash +CORRELATION_ID="$(python3 -c 'import uuid; print(uuid.uuid4())')" +``` + +For ordinary mode: + +```bash +gh workflow run .github/workflows/e2e.yaml \ + --repo NVIDIA/NemoClaw \ + --ref main \ + -f targets= \ + -f jobs= \ + -f inference_mode=mock \ + -f include_staging_brev_launchable=false \ + -f "correlation_id=${CORRELATION_ID}" +``` + +For full mode: + +```bash +gh workflow run .github/workflows/e2e.yaml \ + --repo NVIDIA/NemoClaw \ + --ref main \ + -f targets= \ + -f jobs= \ + -f inference_mode=mock \ + -f include_staging_brev_launchable=true \ + -f "correlation_id=${CORRELATION_ID}" +``` + +Do not set `jobs=staging-brev-launchable` for full mode. +Empty `jobs` and `targets` select the default suite. +The boolean input adds qualification to that same run. + +Find the run by its unique title: + +```bash +RUN_TITLE="E2E main (${CORRELATION_ID})" +for POLL_INDEX in $(seq 1 30); do + RUNS="$(gh run list --repo NVIDIA/NemoClaw --workflow e2e.yaml \ + --event workflow_dispatch --branch main --limit 50 \ + --json databaseId,displayTitle,headSha,status,url)" + MATCHES="$(jq -c --arg title "$RUN_TITLE" \ + '[.[] | select(.displayTitle == $title)]' <<<"$RUNS")" + [ "$(jq 'length' <<<"$MATCHES")" -le 1 ] || { + echo "Correlation matched more than one E2E run" >&2 + exit 1 + } + RUN_ID="$(jq -r '.[0].databaseId // empty' <<<"$MATCHES")" + [ -z "$RUN_ID" ] || break + sleep 10 +done +test -n "${RUN_ID:-}" +RUN_SHA="$(jq -r '.[0].headSha' <<<"$MATCHES")" +test "$RUN_SHA" = "$CANDIDATE_SHA" +``` + +Reject a run for another SHA. +Do not reuse it as evidence. + +Wait for completion: + +```bash +gh run watch "$RUN_ID" --repo NVIDIA/NemoClaw --exit-status +``` + +Full mode can wait for protected-environment approval. +Queued, waiting, or accepted dispatch state is not success. + +## Verify the Result + +Create a private temporary evidence directory: + +```bash +EVIDENCE_DIR="$(mktemp -d)" +chmod 700 "$EVIDENCE_DIR" +trap 'rm -rf "$EVIDENCE_DIR"' EXIT +gh api "repos/NVIDIA/NemoClaw/actions/runs/$RUN_ID" >"$EVIDENCE_DIR/run.json" +gh api "repos/NVIDIA/NemoClaw/actions/runs/$RUN_ID/jobs?filter=latest&per_page=100" \ + >"$EVIDENCE_DIR/jobs.json" +``` + +For ordinary mode, require `run.json` to report: + +- `head_sha` equal to `CANDIDATE_SHA`; +- `status` equal to `completed`; and +- `conclusion` equal to `success`. + +Return the run URL and conclusion. + +For full mode, download the qualification evidence: + +```bash +gh run download "$RUN_ID" --repo NVIDIA/NemoClaw \ + --name "staging-brev-launchable-${CANDIDATE_SHA}-${RUN_ID}" \ + --dir "$EVIDENCE_DIR" +node --experimental-strip-types --no-warnings \ + .agents/skills/nemoclaw-maintainer-e2e/scripts/validate-full-e2e-evidence.mts \ + --candidate-sha "$CANDIDATE_SHA" \ + --run-json "$EVIDENCE_DIR/run.json" \ + --jobs-json "$EVIDENCE_DIR/jobs.json" \ + --dispatch-json "$EVIDENCE_DIR/dispatch.json" \ + --qualification-json "$EVIDENCE_DIR/qualification.json" \ + --cleanup-json "$EVIDENCE_DIR/cleanup.json" +``` + +The validator requires: + +- the workflow run to succeed for the selected SHA; +- `dispatch.json` to bind the run and attempt to empty selectors and `include_staging_brev_launchable=true`; +- `Exact staging Brev Launchable` to conclude `success` in the reported attempt; +- `qualification.json` to identify the selected SHA in the repository and provision records; +- the booted repository to be unmodified; +- the in-guest full E2E to pass; and +- `cleanup.json` to report the same workspace as `ABSENT`. + +A skipped, cancelled, queued, or failed qualification job is not evidence. +A selective `jobs=staging-brev-launchable` run is not full-mode evidence. +A missing, mismatched, or failed cleanup receipt is not evidence. + +## Bind Release Evidence + +If no release plan exists, label a successful full run against `origin/main` as provisional release evidence. +Return: + +- candidate SHA; +- workflow run URL and conclusion; +- `Exact staging Brev Launchable` job URL; +- workflow attempt number; +- qualification identity; and +- cleanup result. + +If the release candidate SHA changes, discard the earlier run and rerun full mode. +No release-note-only delta exception is currently defined. + +When `nemoclaw-maintainer-cut-release-tag` invokes this skill, return the validated fields for its pre-tag E2E evidence ledger. +The trusted `dispatch.json` receipt proves that full mode selected the default suite. +The release evidence ledger proves the result of each default-suite execution. +Do not ask for the release confirmation phrase in this skill. + +## Access Failures + +Follow the shared [Git and GitHub Access Hard Stop](../_shared/git-github-hard-stop.md). +Stop on authentication, authorization, remote-access, or permission failures. diff --git a/.agents/skills/nemoclaw-maintainer-e2e/agents/openai.yaml b/.agents/skills/nemoclaw-maintainer-e2e/agents/openai.yaml new file mode 100644 index 00000000000..53f3b7a9521 --- /dev/null +++ b/.agents/skills/nemoclaw-maintainer-e2e/agents/openai.yaml @@ -0,0 +1,7 @@ +# SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +# SPDX-License-Identifier: Apache-2.0 + +interface: + display_name: "NemoClaw Maintainer E2E" + short_description: "Run trusted release-candidate E2E" + default_prompt: "Use $nemoclaw-maintainer-e2e to run the full E2E suite for the selected release candidate." diff --git a/.agents/skills/nemoclaw-maintainer-e2e/scripts/validate-full-e2e-evidence.mts b/.agents/skills/nemoclaw-maintainer-e2e/scripts/validate-full-e2e-evidence.mts new file mode 100644 index 00000000000..583d2bf8dd8 --- /dev/null +++ b/.agents/skills/nemoclaw-maintainer-e2e/scripts/validate-full-e2e-evidence.mts @@ -0,0 +1,218 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +import fs from "node:fs"; +import { parseArgs } from "node:util"; +import { pathToFileURL } from "node:url"; + +type JsonRecord = Record; + +export interface FullE2eEvidenceInput { + candidateSha: string; + cleanup: unknown; + dispatch: unknown; + jobs: unknown; + qualification: unknown; + run: unknown; +} + +export interface FullE2eEvidenceSummary { + attempt: number; + candidateSha: string; + cleanup: { + status: "ABSENT"; + verifiedAt: string; + workspaceId: string; + workspaceName: string; + }; + dispatch: { + defaultSuiteSelected: true; + includeStagingBrevLaunchable: true; + }; + jobUrl: string; + qualification: { + fullE2e: "passed"; + producerRunId: string; + provisionSha: string; + repoClean: true; + repoSha: string; + }; + runUrl: string; +} + +function record(value: unknown, owner: string): JsonRecord { + if (typeof value !== "object" || value === null || Array.isArray(value)) { + throw new Error(`${owner} must be an object`); + } + return value as JsonRecord; +} + +function stringField(value: JsonRecord, key: string, owner: string): string { + const field = value[key]; + if (typeof field !== "string" || field.length === 0) { + throw new Error(`${owner}.${key} must be a non-empty string`); + } + return field; +} + +function positiveIntegerField(value: JsonRecord, key: string, owner: string): number { + const field = value[key]; + if (!Number.isSafeInteger(field) || Number(field) < 1) { + throw new Error(`${owner}.${key} must be a positive integer`); + } + return Number(field); +} + +function requireEqual(actual: unknown, expected: unknown, owner: string): void { + if (actual !== expected) { + throw new Error(`${owner} must equal ${JSON.stringify(expected)}`); + } +} + +function requireGitHubUrl(value: string, owner: string): void { + if (!value.startsWith("https://github.com/NVIDIA/NemoClaw/actions/")) { + throw new Error(`${owner} must be an NVIDIA/NemoClaw Actions URL`); + } +} + +export function validateFullE2eEvidence(input: FullE2eEvidenceInput): FullE2eEvidenceSummary { + if (!/^[0-9a-f]{40}$/.test(input.candidateSha)) { + throw new Error("candidate SHA must be a lowercase 40-character SHA"); + } + + const run = record(input.run, "run"); + requireEqual(run.head_sha, input.candidateSha, "run.head_sha"); + requireEqual(run.head_branch, "main", "run.head_branch"); + requireEqual(run.event, "workflow_dispatch", "run.event"); + requireEqual(run.path, ".github/workflows/e2e.yaml", "run.path"); + requireEqual(run.status, "completed", "run.status"); + requireEqual(run.conclusion, "success", "run.conclusion"); + const attempt = positiveIntegerField(run, "run_attempt", "run"); + const runUrl = stringField(run, "html_url", "run"); + requireGitHubUrl(runUrl, "run.html_url"); + const runId = positiveIntegerField(run, "id", "run"); + + const dispatch = record(input.dispatch, "dispatch"); + requireEqual(dispatch.kind, "nemoclaw-e2e-dispatch-v1", "dispatch.kind"); + requireEqual(dispatch.candidateSha, input.candidateSha, "dispatch.candidateSha"); + requireEqual(dispatch.eventName, "workflow_dispatch", "dispatch.eventName"); + requireEqual(dispatch.workflowRunId, String(runId), "dispatch.workflowRunId"); + requireEqual(dispatch.workflowRunAttempt, attempt, "dispatch.workflowRunAttempt"); + requireEqual(dispatch.jobs, "", "dispatch.jobs"); + requireEqual(dispatch.targets, "", "dispatch.targets"); + requireEqual( + dispatch.includeStagingBrevLaunchable, + true, + "dispatch.includeStagingBrevLaunchable", + ); + requireEqual(dispatch.defaultSuiteSelected, true, "dispatch.defaultSuiteSelected"); + + const jobsPayload = record(input.jobs, "jobs response"); + if (!Array.isArray(jobsPayload.jobs)) { + throw new Error("jobs response.jobs must be an array"); + } + const matchingJobs = jobsPayload.jobs + .map((job, index) => record(job, `jobs response.jobs[${index}]`)) + .filter((job) => job.name === "Exact staging Brev Launchable"); + if (matchingJobs.length !== 1) { + throw new Error("jobs response must contain exactly one Exact staging Brev Launchable job"); + } + const job = matchingJobs[0]!; + requireEqual(job.status, "completed", "Exact staging Brev Launchable status"); + requireEqual(job.conclusion, "success", "Exact staging Brev Launchable conclusion"); + requireEqual(job.run_attempt, attempt, "Exact staging Brev Launchable run_attempt"); + const jobUrl = stringField(job, "html_url", "Exact staging Brev Launchable"); + requireGitHubUrl(jobUrl, "Exact staging Brev Launchable html_url"); + if (!jobUrl.startsWith(`${runUrl}/job/`)) { + throw new Error("Exact staging Brev Launchable html_url must belong to the workflow run"); + } + + const qualification = record(input.qualification, "qualification"); + requireEqual(qualification.candidateSha, input.candidateSha, "qualification.candidateSha"); + requireEqual(qualification.fullE2e, "passed", "qualification.fullE2e"); + const producer = record(qualification.producer, "qualification.producer"); + requireEqual(producer.status, "success", "qualification.producer.status"); + const producerRunId = stringField(producer, "runId", "qualification.producer"); + const boot = record(qualification.boot, "qualification.boot"); + requireEqual(boot.repoSha, input.candidateSha, "qualification.boot.repoSha"); + requireEqual(boot.provisionSha, input.candidateSha, "qualification.boot.provisionSha"); + requireEqual(boot.repoClean, true, "qualification.boot.repoClean"); + + const workspace = record(qualification.workspace, "qualification.workspace"); + const workspaceName = stringField(workspace, "name", "qualification.workspace"); + const workspaceId = stringField(workspace, "id", "qualification.workspace"); + const cleanup = record(input.cleanup, "cleanup"); + requireEqual(cleanup.workspaceName, workspaceName, "cleanup.workspaceName"); + requireEqual(cleanup.workspaceId, workspaceId, "cleanup.workspaceId"); + requireEqual(cleanup.status, "ABSENT", "cleanup.status"); + const verifiedAt = stringField(cleanup, "verifiedAt", "cleanup"); + if (Number.isNaN(Date.parse(verifiedAt))) { + throw new Error("cleanup.verifiedAt must be an ISO timestamp"); + } + + return { + attempt, + candidateSha: input.candidateSha, + cleanup: { + status: "ABSENT", + verifiedAt, + workspaceId, + workspaceName, + }, + dispatch: { + defaultSuiteSelected: true, + includeStagingBrevLaunchable: true, + }, + jobUrl, + qualification: { + fullE2e: "passed", + producerRunId, + provisionSha: input.candidateSha, + repoClean: true, + repoSha: input.candidateSha, + }, + runUrl, + }; +} + +function readJson(file: string): unknown { + return JSON.parse(fs.readFileSync(file, "utf8")); +} + +function main(): void { + const { values } = parseArgs({ + options: { + "candidate-sha": { type: "string" }, + "cleanup-json": { type: "string" }, + "dispatch-json": { type: "string" }, + "jobs-json": { type: "string" }, + "qualification-json": { type: "string" }, + "run-json": { type: "string" }, + }, + strict: true, + }); + for (const name of [ + "candidate-sha", + "cleanup-json", + "dispatch-json", + "jobs-json", + "qualification-json", + "run-json", + ] as const) { + if (!values[name]) throw new Error(`--${name} is required`); + } + + const summary = validateFullE2eEvidence({ + candidateSha: values["candidate-sha"]!, + cleanup: readJson(values["cleanup-json"]!), + dispatch: readJson(values["dispatch-json"]!), + jobs: readJson(values["jobs-json"]!), + qualification: readJson(values["qualification-json"]!), + run: readJson(values["run-json"]!), + }); + process.stdout.write(`${JSON.stringify(summary, null, 2)}\n`); +} + +if (process.argv[1] && import.meta.url === pathToFileURL(process.argv[1]).href) { + main(); +} diff --git a/.agents/skills/nemoclaw-maintainer-evening/SKILL.md b/.agents/skills/nemoclaw-maintainer-evening/SKILL.md index ad115240370..a12e704bc4e 100644 --- a/.agents/skills/nemoclaw-maintainer-evening/SKILL.md +++ b/.agents/skills/nemoclaw-maintainer-evening/SKILL.md @@ -50,14 +50,28 @@ If a docs PR or any other intended PR merges after `release:plan`, regenerate th ## Step 4: Cut the Tag and Publish Release Notes -Load `cut-release-tag`. The version is already known — default to patch bump, but still show the commit, changelog, post-tag carry-forward and label-retirement plan, and release notes draft for confirmation. After the release plan freezes the candidate SHA, review the pre-tag E2E evidence ledger derived from `.github/workflows/e2e.yaml` at that commit. Do not ask for the release confirmation phrase until every test has green evidence or an explicit itemized maintainer exception. NemoClaw releases are tag-based: tag the confirmed release commit with `vX.Y.Z`, let the workflow move `latest`, automatically carry remaining open issues/PRs to the next patch label, delete the released label, and prepare the release notes announcement for the maintainer to post. +Load `cut-release-tag`. +The version is already known, so use a patch bump unless the maintainer selects another bump. +Show the commit, changelog, carry-forward plan, label-retirement plan, and release notes draft. + +After the release plan freezes the candidate SHA, load `nemoclaw-maintainer-e2e`. +Run full mode when that SHA has no applicable exact Brev Launchable evidence. +Review the pre-tag E2E evidence ledger from `.github/workflows/e2e.yaml` at that commit. +Require a successful `Exact staging Brev Launchable` job, matching qualification identity, and verified workspace absence. +Each missing test result requires its own itemized maintainer exception. +Missing or invalid qualification requires a separate itemized exception with run and job URLs, the result or missing receipt, and rationale. +Do not ask for the release confirmation phrase until each required result has successful evidence or its own exception. + +Tag the confirmed release commit with `vX.Y.Z`. +Let the workflow move `latest`, carry open work forward, and delete the released label. +Prepare the Announcement draft for the maintainer to post. ## Step 5: Confirm and Share After the tag is cut and release notes are drafted or posted by the maintainer, present the final summary: - **Tag**: `v0.0.8` at commit `abc1234` -- **Pre-tag E2E evidence**: 12/13 tests green for the candidate SHA; 1 itemized maintainer exception +- **Pre-tag E2E evidence**: 12/13 tests and exact Brev Launchable qualification passing for the candidate SHA; 1 itemized maintainer exception - **Release notes draft**: `../nemoclaw-release-v0.0.8/release-note-draft.md` - **Shipped**: 4 items (#1234, #1235, #1236, #1237) - **Moved to v0.0.9**: 1 item (#1238 — still needs CI fix) diff --git a/.agents/skills/nemoclaw-maintainer-policies/references/release-train.md b/.agents/skills/nemoclaw-maintainer-policies/references/release-train.md index c4ca47c378f..41337484f95 100644 --- a/.agents/skills/nemoclaw-maintainer-policies/references/release-train.md +++ b/.agents/skills/nemoclaw-maintainer-policies/references/release-train.md @@ -51,14 +51,20 @@ The release candidate is the full `origin/main` commit SHA captured by the gener Before asking for the release confirmation phrase, build and show an evidence ledger for that SHA: +- Run `nemoclaw-maintainer-e2e` in full mode when no applicable full-mode run exists for the candidate SHA. +- Require one workflow run for the candidate SHA that includes the default-enabled suite and a successful `Exact staging Brev Launchable` job. +- Require the trusted dispatch receipt to bind that run and attempt to empty selectors and `include_staging_brev_launchable=true`. +- Require the qualification receipt to identify the candidate SHA in the repository and provision records. +- Require the cleanup receipt to identify the qualified workspace and report `ABSENT`. - Every E2E test execution declared by the workflow must have at least one completed, successful execution for the candidate SHA. This includes tests that require explicit selection and every expanded matrix execution. - Treat each expanded matrix execution as a separate ledger entry. Use its matrix `id`, or all distinguishing matrix dimensions when no single ID exists, in the test identifier so results for distinct expansions are never collapsed under the parent job. - Green evidence may accumulate across multiple workflow runs, selective runs, reruns, and attempts. A later failure does not erase an earlier successful execution for the same test and SHA. - Skipped, unexecuted, queued, in-progress, cancelled, and failing results are not green evidence. - Map each test with green evidence to its successful run or job URL and attempt number. -- If a test has no successful execution, the tag may still proceed at maintainer discretion only with an itemized maintainer exception that records the test identifier, relevant run links or available evidence, the current result or failure summary, and the rationale for proceeding. +- Each test without a successful execution requires its own itemized maintainer exception. Record the test identifier, relevant run links or available evidence, the current result or failure summary, and the rationale. +- Missing or invalid exact Brev Launchable qualification requires a separate itemized maintainer exception. Record the run and job URLs, the current result or missing receipt, and the rationale. -Every test must have either green evidence or an itemized maintainer exception before the release confirmation is requested. If the candidate SHA changes, discard the ledger and its exceptions, regenerate the release plan, and repeat the review for the new SHA. +Each test and the exact Brev Launchable qualification must have successful evidence or its own itemized maintainer exception before release confirmation. If the candidate SHA changes, discard the ledger and its exceptions, including qualification evidence. Regenerate the release plan and repeat the review for the new SHA. No release-note-only delta exception is currently defined. ## Carry Forward diff --git a/.agents/skills/nemoclaw-skills-guide/SKILL.md b/.agents/skills/nemoclaw-skills-guide/SKILL.md index 3401fcf02cc..1d22e67f500 100644 --- a/.agents/skills/nemoclaw-skills-guide/SKILL.md +++ b/.agents/skills/nemoclaw-skills-guide/SKILL.md @@ -25,10 +25,10 @@ The prefix in each skill name indicates who it is for. For end users operating a NemoClaw sandbox. Covers routing human users' AI agents to the canonical NemoClaw Markdown documentation. -### `nemoclaw-maintainer-*` (14 skills) +### `nemoclaw-maintainer-*` (15 skills) For project maintainers. -Covers the daily maintainer cadence (morning standup, daytime loop, evening handoff), workflow policy reference, documentation information-architecture refactors, cutting releases, drafting release notes, finding PRs to review, comparing PRs, cross-issue sweeps, triage, normalizing issue and PR title tags, performing security code reviews, and verifying whether stale bug reports still reproduce on the latest release. +Covers the daily maintainer cadence, trusted E2E dispatch, workflow policy, documentation refactors, releases, review selection, comparison, triage, security review, and stale bug verification. ### `nemoclaw-contributor-*` (5 skills) @@ -58,6 +58,7 @@ documentation updates, and onboarding new messaging channels. | `nemoclaw-maintainer-day` | Run one daytime maintainer pass for the release version. Select a merge, salvage, security, test, conflict, or sequencing workflow. Designed for `/loop`. | | `nemoclaw-maintainer-evening` | End-of-day handoff: require the pre-tag dated changelog PR, check version progress, identify stragglers, generate a QA handoff summary, cut the release tag, carry stragglers forward, retire the released label, and hand off the Announcement. | | `nemoclaw-maintainer-cut-release-tag` | Verify the dated changelog entry, cut an annotated semver tag on a maintainer-confirmed `origin/main` commit, wait for workflow-managed `latest`, carry remaining open items forward, and delete the released label; `lkg` stays manual. | +| `nemoclaw-maintainer-e2e` | Dispatch ordinary or full trusted GitHub Actions E2E and verify exact-candidate Brev Launchable qualification evidence. | | `nemoclaw-maintainer-release-notes` | Draft the post-tag Announcement from live tag/compare data, with the three-paragraph narrative, categorized change list, and external-only contributor thanks. | | `nemoclaw-maintainer-find-review-pr` | Find open security PRs with Urgent or High Project Priority. Link each PR to its issue and identify competing PRs. | | `nemoclaw-maintainer-pr-comparator` | Compare open PRs for the same issue. Apply gates and score the eligible PRs before you recommend one to merge. | @@ -90,6 +91,6 @@ Skills are cumulative. Each role includes the skills from the roles above it: |------|----------------|-------|------------| | User | `nemoclaw-user-*` | 1 | `nemoclaw-user-guide` | | Contributor | `nemoclaw-user-*` + `nemoclaw-contributor-*` | 6 | `nemoclaw-contributor-onboard` | -| Maintainer | All skills | 20 | `nemoclaw-maintainer-morning` | +| Maintainer | All skills | 21 | `nemoclaw-maintainer-morning` | After identifying the role, present the applicable skills from the Skill Catalog above and recommend the starting skill. diff --git a/.github/workflows/e2e.yaml b/.github/workflows/e2e.yaml index 8f5f9adbd97..bcd7720424e 100644 --- a/.github/workflows/e2e.yaml +++ b/.github/workflows/e2e.yaml @@ -2,7 +2,7 @@ # SPDX-License-Identifier: Apache-2.0 name: E2E -run-name: "${{ inputs.checkout_sha != '' && format('E2E PR #{0} ({1})', inputs.pr_number, inputs.correlation_id) || format('E2E {0}', github.ref_name) }}" +run-name: "${{ inputs.checkout_sha != '' && format('E2E PR #{0} ({1})', inputs.pr_number, inputs.correlation_id) || inputs.correlation_id != '' && format('E2E {0} ({1})', github.ref_name, inputs.correlation_id) || format('E2E {0}', github.ref_name) }}" on: schedule: @@ -15,10 +15,15 @@ on: default: "" type: string jobs: - description: "Optional comma-separated E2E test IDs. Empty runs default-enabled tests only when targets is also empty; explicit-only tests openshell-gateway-auth-contract, mcp-bridge-dev, hermes-gpu-startup, sandbox-rlimits-connect, and jetson-nvmap-gpu are skipped unless selected." + description: "Optional comma-separated E2E test IDs. Empty runs default-enabled tests only when targets is also empty; explicit-only tests openshell-gateway-auth-contract, mcp-bridge-dev, hermes-gpu-startup, sandbox-rlimits-connect, jetson-nvmap-gpu, and staging-brev-launchable are skipped unless selected." required: false default: "" type: string + include_staging_brev_launchable: + description: "Include Exact staging Brev Launchable qualification when jobs and targets are empty. Requires the persistent repository readiness gate." + required: false + default: false + type: boolean inference_mode: description: "Inference adapter mode for compatible Vitest E2E jobs: mock, internal-nvidia, or public-nvidia." required: false @@ -263,7 +268,7 @@ jobs: staging-brev-launchable: name: Exact staging Brev Launchable needs: generate-matrix - if: ${{ vars.NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED == 'true' && github.repository == 'NVIDIA/NemoClaw' && github.ref == 'refs/heads/main' && (github.event_name == 'schedule' || (github.event_name == 'workflow_dispatch' && contains(format(',{0},', inputs.jobs), ',staging-brev-launchable,'))) }} + if: ${{ vars.NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED == 'true' && github.repository == 'NVIDIA/NemoClaw' && github.ref == 'refs/heads/main' && (github.event_name == 'schedule' || (github.event_name == 'workflow_dispatch' && (contains(format(',{0},', inputs.jobs), ',staging-brev-launchable,') || (inputs.include_staging_brev_launchable && inputs.jobs == '' && inputs.targets == '')))) }} runs-on: ubuntu-latest timeout-minutes: 180 environment: @@ -307,6 +312,37 @@ jobs: brev login --api-key "$BREV_API_KEY" --org-id "$BREV_ORG_ID" printf 'work_dir=%s\n' "$work_dir" >> "$GITHUB_OUTPUT" + - name: Record E2E dispatch identity + env: + CANDIDATE_SHA: ${{ env.CANDIDATE_SHA }} + DISPATCH_JOBS: ${{ inputs.jobs }} + DISPATCH_TARGETS: ${{ inputs.targets }} + EVENT_NAME: ${{ github.event_name }} + INCLUDE_STAGING_BREV_LAUNCHABLE: ${{ inputs.include_staging_brev_launchable && 'true' || 'false' }} + RUN_ATTEMPT: ${{ github.run_attempt }} + RUN_ID: ${{ github.run_id }} + WORK_DIR: ${{ steps.workspace.outputs.work_dir }} + run: | + jq -n \ + --arg candidateSha "$CANDIDATE_SHA" \ + --arg eventName "$EVENT_NAME" \ + --arg jobs "$DISPATCH_JOBS" \ + --arg targets "$DISPATCH_TARGETS" \ + --arg workflowRunId "$RUN_ID" \ + --argjson includeStagingBrevLaunchable "$INCLUDE_STAGING_BREV_LAUNCHABLE" \ + --argjson workflowRunAttempt "$RUN_ATTEMPT" \ + '{ + kind: "nemoclaw-e2e-dispatch-v1", + candidateSha: $candidateSha, + eventName: $eventName, + workflowRunId: $workflowRunId, + workflowRunAttempt: $workflowRunAttempt, + jobs: $jobs, + targets: $targets, + includeStagingBrevLaunchable: $includeStagingBrevLaunchable, + defaultSuiteSelected: ($jobs == "" and $targets == "") + }' >"$WORK_DIR/dispatch.json" + - name: Build, deploy, verify, test, and clean up env: BREV_LAUNCHABLE_ID: ${{ vars.NEMOCLAW_STAGING_LAUNCHABLE_ID }} @@ -322,10 +358,24 @@ jobs: name: staging-brev-launchable-${{ env.CANDIDATE_SHA }}-${{ github.run_id }} path: | ${{ steps.workspace.outputs.work_dir }}/lane.log + ${{ steps.workspace.outputs.work_dir }}/dispatch.json ${{ steps.workspace.outputs.work_dir }}/qualification.json ${{ steps.workspace.outputs.work_dir }}/full-e2e.log ${{ steps.workspace.outputs.work_dir }}/cleanup.json + staging-brev-launchable-readiness: + name: Require staging Brev Launchable readiness + needs: generate-matrix + if: ${{ github.event_name == 'workflow_dispatch' && inputs.include_staging_brev_launchable && inputs.jobs == '' && inputs.targets == '' && (github.repository != 'NVIDIA/NemoClaw' || github.ref != 'refs/heads/main' || vars.NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED != 'true') }} + runs-on: ubuntu-latest + permissions: + contents: read + steps: + - name: Fail when staging Brev Launchable is not ready + run: | + echo "::error::Full E2E must run from NVIDIA/NemoClaw main and requires NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED=true. A release administrator must enable it after the protected environment, secrets, staging Launchable ID, and ownership are configured." + exit 1 + live: needs: generate-matrix if: ${{ needs.generate-matrix.outputs.matrix != '[]' }} @@ -5633,6 +5683,7 @@ jobs: base-image-publication, generate-matrix, staging-brev-launchable, + staging-brev-launchable-readiness, live, shared-e2e, openshell-gateway-auth-contract, diff --git a/test/e2e/support/e2e-workflow.test.ts b/test/e2e/support/e2e-workflow.test.ts index af290259c9a..acb40831559 100644 --- a/test/e2e/support/e2e-workflow.test.ts +++ b/test/e2e/support/e2e-workflow.test.ts @@ -9,6 +9,7 @@ import { describe, expect, it } from "vitest"; import YAML from "yaml"; import { evaluateE2eWorkflowDispatchSelectors, + evaluateStagingBrevLaunchableDispatch, focusedE2eJobsForChangedFiles, readFreeStandingJobsInventory, validateE2eWorkflow, @@ -62,6 +63,99 @@ describe("e2e workflow boundary", () => { ); }); + it("keeps ordinary, full, selective, scheduled, and disabled-readiness dispatches distinct (#7487)", () => { + expect( + evaluateStagingBrevLaunchableDispatch({ + eventName: "workflow_dispatch", + readinessEnabled: true, + }), + ).toEqual({ failReadiness: false, runQualification: false }); + expect( + evaluateStagingBrevLaunchableDispatch({ + eventName: "workflow_dispatch", + includeStagingBrevLaunchable: true, + readinessEnabled: true, + }), + ).toEqual({ failReadiness: false, runQualification: true }); + expect( + evaluateStagingBrevLaunchableDispatch({ + eventName: "workflow_dispatch", + includeStagingBrevLaunchable: true, + jobs: "hermes-e2e", + readinessEnabled: true, + }), + ).toEqual({ failReadiness: false, runQualification: false }); + expect( + evaluateStagingBrevLaunchableDispatch({ + eventName: "workflow_dispatch", + jobs: "staging-brev-launchable", + readinessEnabled: true, + }), + ).toEqual({ failReadiness: false, runQualification: true }); + expect( + evaluateStagingBrevLaunchableDispatch({ + eventName: "schedule", + readinessEnabled: true, + }), + ).toEqual({ failReadiness: false, runQualification: true }); + expect( + evaluateStagingBrevLaunchableDispatch({ + eventName: "workflow_dispatch", + includeStagingBrevLaunchable: true, + readinessEnabled: false, + }), + ).toEqual({ failReadiness: true, runQualification: false }); + expect( + evaluateStagingBrevLaunchableDispatch({ + eventName: "workflow_dispatch", + includeStagingBrevLaunchable: true, + readinessEnabled: true, + trustedMain: false, + }), + ).toEqual({ failReadiness: true, runQualification: false }); + }); + + it("rejects full-dispatch input, correlation, selector, and readiness drift (#7487)", () => { + const workflow = readWorkflow() as { + "run-name": string; + on: { + workflow_dispatch: { + inputs: Record; + }; + }; + jobs: Record< + string, + { + if?: string; + steps?: Array<{ env?: Record; name?: string; run?: string }>; + } + >; + }; + workflow["run-name"] = "E2E"; + workflow.on.workflow_dispatch.inputs.include_staging_brev_launchable.default = true; + workflow.jobs["staging-brev-launchable"]!.if = "${{ github.event_name == 'schedule' }}"; + workflow.jobs["staging-brev-launchable-readiness"]!.if = "${{ false }}"; + const dispatchIdentity = workflow.jobs["staging-brev-launchable"]!.steps!.find( + (step) => step.name === "Record E2E dispatch identity", + )!; + delete dispatchIdentity.env!.DISPATCH_JOBS; + dispatchIdentity.run = dispatchIdentity.run!.replace( + 'kind: "nemoclaw-e2e-dispatch-v1"', + 'kind: "untrusted"', + ); + + expect(validateE2eWorkflow(workflow)).toEqual( + expect.arrayContaining([ + "workflow run-name must expose the unique manual-dispatch correlation ID", + "workflow_dispatch include_staging_brev_launchable input must be boolean and default to false", + "staging-brev-launchable must run for schedules, explicit selection, or an empty-selector full dispatch", + "staging-brev-launchable-readiness must fail only full dispatches with disabled readiness", + "staging-brev-launchable dispatch identity must bind DISPATCH_JOBS", + `step 'Record E2E dispatch identity' run script must include kind: "nemoclaw-e2e-dispatch-v1"`, + ]), + ); + }); + it("keeps network-policy scenarios isolated with cleanup reserve", () => { const workflow = readWorkflow() as { jobs: Record< diff --git a/test/maintainer-e2e-skill.test.ts b/test/maintainer-e2e-skill.test.ts new file mode 100644 index 00000000000..1e6cb251605 --- /dev/null +++ b/test/maintainer-e2e-skill.test.ts @@ -0,0 +1,178 @@ +// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. All rights reserved. +// SPDX-License-Identifier: Apache-2.0 + +import fs from "node:fs"; +import path from "node:path"; +import { describe, expect, it } from "vitest"; +import { validateFullE2eEvidence } from "../.agents/skills/nemoclaw-maintainer-e2e/scripts/validate-full-e2e-evidence.mts"; + +const candidateSha = "a".repeat(40); + +function validEvidence() { + return { + candidateSha, + cleanup: { + status: "ABSENT", + verifiedAt: "2026-07-24T12:00:00Z", + workspaceId: "workspace-123", + workspaceName: "nclaw-e2e-100-2", + }, + dispatch: { + candidateSha, + defaultSuiteSelected: true, + eventName: "workflow_dispatch", + includeStagingBrevLaunchable: true, + jobs: "", + kind: "nemoclaw-e2e-dispatch-v1", + targets: "", + workflowRunAttempt: 2, + workflowRunId: "100", + }, + jobs: { + jobs: [ + { + conclusion: "success", + html_url: "https://github.com/NVIDIA/NemoClaw/actions/runs/100/job/200", + name: "Exact staging Brev Launchable", + run_attempt: 2, + status: "completed", + }, + ], + }, + qualification: { + boot: { + provisionSha: candidateSha, + repoClean: true, + repoSha: candidateSha, + }, + candidateSha, + fullE2e: "passed", + producer: { runId: "99", status: "success" }, + workspace: { id: "workspace-123", name: "nclaw-e2e-100-2" }, + }, + run: { + conclusion: "success", + event: "workflow_dispatch", + head_branch: "main", + head_sha: candidateSha, + html_url: "https://github.com/NVIDIA/NemoClaw/actions/runs/100", + id: 100, + path: ".github/workflows/e2e.yaml", + run_attempt: 2, + status: "completed", + }, + }; +} + +describe("nemoclaw-maintainer-e2e evidence validation", () => { + it("returns exact-candidate job, qualification, and cleanup evidence (#7487)", () => { + expect(validateFullE2eEvidence(validEvidence())).toEqual({ + attempt: 2, + candidateSha, + cleanup: { + status: "ABSENT", + verifiedAt: "2026-07-24T12:00:00Z", + workspaceId: "workspace-123", + workspaceName: "nclaw-e2e-100-2", + }, + dispatch: { + defaultSuiteSelected: true, + includeStagingBrevLaunchable: true, + }, + jobUrl: "https://github.com/NVIDIA/NemoClaw/actions/runs/100/job/200", + qualification: { + fullE2e: "passed", + producerRunId: "99", + provisionSha: candidateSha, + repoClean: true, + repoSha: candidateSha, + }, + runUrl: "https://github.com/NVIDIA/NemoClaw/actions/runs/100", + }); + }); + + it.each([ + [ + "a run for another SHA", + (evidence: ReturnType) => { + evidence.run.head_sha = "b".repeat(40); + }, + "run.head_sha", + ], + [ + "a selective qualification dispatch", + (evidence: ReturnType) => { + evidence.dispatch.defaultSuiteSelected = false; + evidence.dispatch.includeStagingBrevLaunchable = false; + evidence.dispatch.jobs = "staging-brev-launchable"; + }, + "dispatch.jobs", + ], + [ + "a skipped qualification job", + (evidence: ReturnType) => { + evidence.jobs.jobs[0]!.conclusion = "skipped"; + }, + "Exact staging Brev Launchable conclusion", + ], + [ + "a qualification receipt for another SHA", + (evidence: ReturnType) => { + evidence.qualification.boot.repoSha = "b".repeat(40); + }, + "qualification.boot.repoSha", + ], + [ + "a cleanup receipt without verified absence", + (evidence: ReturnType) => { + evidence.cleanup.status = "PRESENT"; + }, + "cleanup.status", + ], + [ + "a job from another attempt", + (evidence: ReturnType) => { + evidence.jobs.jobs[0]!.run_attempt = 1; + }, + "Exact staging Brev Launchable run_attempt", + ], + ])("rejects %s (#7487)", (_name, mutate, message) => { + const evidence = validEvidence(); + mutate(evidence); + + expect(() => validateFullE2eEvidence(evidence)).toThrow(message); + }); +}); + +describe("nemoclaw-maintainer-e2e workflow routing", () => { + const skill = fs.readFileSync( + path.join(process.cwd(), ".agents", "skills", "nemoclaw-maintainer-e2e", "SKILL.md"), + "utf8", + ); + + it("keeps ordinary and billable full requests distinct (#7487)", () => { + expect(skill).toContain("Run the E2E suite"); + expect(skill).toContain("include_staging_brev_launchable=false"); + expect(skill).toContain("Run the full E2E suite"); + expect(skill).toContain("include_staging_brev_launchable=true"); + expect(skill).toContain("deploy pre-release full E2E"); + expect(skill).toContain("run pre-tag full E2E"); + expect(skill).toContain("run release-candidate E2E"); + expect(skill).toContain("must not authorize the protected Brev path"); + expect(skill).not.toMatch(/variable (?:set|delete) NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED/u); + }); + + it("binds dispatch, evidence, invalidation, and release handoff to one SHA (#7487)", () => { + expect(skill).toContain("git rev-parse origin/main"); + expect(skill).toContain("correlation_id=${CORRELATION_ID}"); + expect(skill).toContain("head_sha"); + expect(skill).toContain("Exact staging Brev Launchable"); + expect(skill).toContain("qualification.json"); + expect(skill).toContain("cleanup.json"); + expect(skill).toContain("dispatch.json"); + expect(skill).toContain("validate-full-e2e-evidence.mts"); + expect(skill).toContain("provisional release evidence"); + expect(skill).toContain("If the release candidate SHA changes"); + expect(skill).toContain("nemoclaw-maintainer-cut-release-tag"); + }); +}); diff --git a/test/maintainer-skills-policy.test.ts b/test/maintainer-skills-policy.test.ts index 72b28e10416..5c0d36d90fd 100644 --- a/test/maintainer-skills-policy.test.ts +++ b/test/maintainer-skills-policy.test.ts @@ -155,14 +155,56 @@ describe("maintainer skills follow canonical workflow policy", () => { ); expect(evidenceSummary).toBeGreaterThanOrEqual(0); expect(evidenceSummary).toBeLessThan(confirmationPrompt); - expect(evening).toContain("every test has green evidence"); - expect(evening).toContain("explicit itemized maintainer exception"); - expect(evening).toContain("tag the confirmed release commit with `vX.Y.Z`"); + expect(evening).toContain( + "Each missing test result requires its own itemized maintainer exception", + ); + expect(evening).toContain("Missing or invalid qualification requires a separate"); + expect(evening).toContain("Tag the confirmed release commit with `vX.Y.Z`"); expect(evening).not.toContain("tag `main`"); expect(dailyFlow).toContain("freeze the candidate SHA and review every E2E test"); expect(priorities).toContain("Record the release SHA and required E2E evidence"); }); + it("requires full-mode exact Brev Launchable evidence before release confirmation (#7487)", () => { + const e2e = read(".agents/skills/nemoclaw-maintainer-e2e/SKILL.md"); + const evening = read(".agents/skills/nemoclaw-maintainer-evening/SKILL.md"); + const release = read(".agents/skills/nemoclaw-maintainer-cut-release-tag/SKILL.md"); + const policy = read(".agents/skills/nemoclaw-maintainer-policies/references/release-train.md"); + const skillsGuide = read(".agents/skills/nemoclaw-skills-guide/SKILL.md"); + + expect(e2e).toContain("include_staging_brev_launchable=true"); + expect(e2e).toContain("Exact staging Brev Launchable"); + expect(e2e).toContain("qualification.json"); + expect(e2e).toContain("cleanup.json"); + expect(e2e).toContain("dispatch.json"); + expect(e2e).toContain("If the release candidate SHA changes"); + expect(release).toContain("load `nemoclaw-maintainer-e2e` and dispatch full mode"); + expect(release).toContain("Treat a skipped job as missing evidence"); + expect(release).toContain("include_staging_brev_launchable=true"); + expect(release).toContain("cleanup evidence that reports the qualified workspace as `ABSENT`"); + expect(release).toContain("a separate itemized maintainer exception for each test"); + expect(release).toContain( + "a separate itemized maintainer exception for missing or invalid exact Brev Launchable qualification", + ); + expect(release.indexOf("load `nemoclaw-maintainer-e2e` and dispatch full mode")).toBeLessThan( + release.indexOf("Ask the maintainer to paste this phrase"), + ); + expect(evening).toContain("load `nemoclaw-maintainer-e2e`"); + expect(evening).toContain("Run full mode"); + expect(policy).toContain( + "Require one workflow run for the candidate SHA that includes the default-enabled suite", + ); + expect(policy).toContain("successful `Exact staging Brev Launchable` job"); + expect(policy).toContain("cleanup receipt"); + expect(policy).toContain("trusted dispatch receipt"); + expect(policy).toContain("Each test without a successful execution requires its own"); + expect(policy).toContain( + "Missing or invalid exact Brev Launchable qualification requires a separate", + ); + expect(policy).toContain("No release-note-only delta exception is currently defined"); + expect(skillsGuide).toContain("`nemoclaw-maintainer-e2e`"); + }); + it("runs release-prep docs before generating the final release plan", () => { const updateDocs = read(".agents/skills/nemoclaw-contributor-update-docs/SKILL.md"); const createPr = read(".agents/skills/nemoclaw-contributor-create-pr/SKILL.md"); diff --git a/tools/e2e/upload-e2e-artifacts-workflow-boundary.mts b/tools/e2e/upload-e2e-artifacts-workflow-boundary.mts index 9f503c2f629..fabf57a4c92 100644 --- a/tools/e2e/upload-e2e-artifacts-workflow-boundary.mts +++ b/tools/e2e/upload-e2e-artifacts-workflow-boundary.mts @@ -62,6 +62,7 @@ const EXPLICIT_UPLOAD_CONTRACTS = new Map([ name: "staging-brev-launchable-${{ env.CANDIDATE_SHA }}-${{ github.run_id }}", path: [ "${{ steps.workspace.outputs.work_dir }}/lane.log", + "${{ steps.workspace.outputs.work_dir }}/dispatch.json", "${{ steps.workspace.outputs.work_dir }}/qualification.json", "${{ steps.workspace.outputs.work_dir }}/full-e2e.log", "${{ steps.workspace.outputs.work_dir }}/cleanup.json", diff --git a/tools/e2e/workflow-boundary.mts b/tools/e2e/workflow-boundary.mts index c383a0de2a5..ec95cd6e4e7 100644 --- a/tools/e2e/workflow-boundary.mts +++ b/tools/e2e/workflow-boundary.mts @@ -102,6 +102,11 @@ export interface FocusedE2eJob { matchedFiles: string[]; } +export interface StagingBrevLaunchableDispatchEvaluation { + failReadiness: boolean; + runQualification: boolean; +} + type CachedFreeStandingJobsInventory = { mtimeMs: number; size: number; @@ -670,6 +675,33 @@ export function evaluateE2eWorkflowDispatchSelectors(input: { }; } +export function evaluateStagingBrevLaunchableDispatch(input: { + eventName: "schedule" | "workflow_dispatch"; + includeStagingBrevLaunchable?: boolean; + jobs?: string; + readinessEnabled?: boolean; + targets?: string; + trustedMain?: boolean; +}): StagingBrevLaunchableDispatchEvaluation { + const jobs = input.jobs ?? ""; + const targets = input.targets ?? ""; + const fullDispatch = + input.eventName === "workflow_dispatch" && + input.includeStagingBrevLaunchable === true && + jobs === "" && + targets === ""; + const explicitlySelected = + input.eventName === "workflow_dispatch" && + splitSelector(jobs).includes("staging-brev-launchable"); + const requested = input.eventName === "schedule" || fullDispatch || explicitlySelected; + const trustedMain = input.trustedMain !== false; + + return { + failReadiness: fullDispatch && (input.readinessEnabled !== true || !trustedMain), + runQualification: requested && input.readinessEnabled === true && trustedMain, + }; +} + function namedStep(steps: readonly WorkflowStep[], name: string): WorkflowStep | undefined { return steps.find((step) => step.name === name); } @@ -4094,11 +4126,64 @@ function validateStagingBrevLaunchableJob(errors: string[], jobs: WorkflowRecord ) { errors.push("staging-brev-launchable must allow only protected trusted-main dispatches"); } + const expectedSelector = + "${{ vars.NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED == 'true' && github.repository == 'NVIDIA/NemoClaw' && github.ref == 'refs/heads/main' && (github.event_name == 'schedule' || (github.event_name == 'workflow_dispatch' && (contains(format(',{0},', inputs.jobs), ',staging-brev-launchable,') || (inputs.include_staging_brev_launchable && inputs.jobs == '' && inputs.targets == '')))) }}"; + if (job.if !== expectedSelector) { + errors.push( + "staging-brev-launchable must run for schedules, explicit selection, or an empty-selector full dispatch", + ); + } const steps = asSteps(job.steps); - const prepareEnv = asRecord(requireStep(errors, steps, "Prepare the trusted lane")?.env); - const runEnv = asRecord( - requireStep(errors, steps, "Build, deploy, verify, test, and clean up")?.env, - ); + const prepare = requireStep(errors, steps, "Prepare the trusted lane"); + const prepareEnv = asRecord(prepare?.env); + const dispatchIdentity = requireStep(errors, steps, "Record E2E dispatch identity"); + const dispatchEnv = asRecord(dispatchIdentity?.env); + for (const [key, expected] of [ + ["CANDIDATE_SHA", "${{ env.CANDIDATE_SHA }}"], + ["DISPATCH_JOBS", "${{ inputs.jobs }}"], + ["DISPATCH_TARGETS", "${{ inputs.targets }}"], + ["EVENT_NAME", "${{ github.event_name }}"], + [ + "INCLUDE_STAGING_BREV_LAUNCHABLE", + "${{ inputs.include_staging_brev_launchable && 'true' || 'false' }}", + ], + ["RUN_ATTEMPT", "${{ github.run_attempt }}"], + ["RUN_ID", "${{ github.run_id }}"], + ["WORK_DIR", "${{ steps.workspace.outputs.work_dir }}"], + ] as const) { + if (dispatchEnv[key] !== expected) { + errors.push(`staging-brev-launchable dispatch identity must bind ${key}`); + } + } + for (const required of [ + 'kind: "nemoclaw-e2e-dispatch-v1"', + "candidateSha: $candidateSha", + "eventName: $eventName", + "workflowRunId: $workflowRunId", + "workflowRunAttempt: $workflowRunAttempt", + "jobs: $jobs", + "targets: $targets", + "includeStagingBrevLaunchable: $includeStagingBrevLaunchable", + 'defaultSuiteSelected: ($jobs == "" and $targets == "")', + '>"$WORK_DIR/dispatch.json"', + ]) { + requireRunContains(errors, dispatchIdentity, required); + } + const run = requireStep(errors, steps, "Build, deploy, verify, test, and clean up"); + if ( + prepare && + dispatchIdentity && + run && + !( + steps.indexOf(prepare) < steps.indexOf(dispatchIdentity) && + steps.indexOf(dispatchIdentity) < steps.indexOf(run) + ) + ) { + errors.push( + "staging-brev-launchable must record dispatch identity after preparation and before qualification", + ); + } + const runEnv = asRecord(run?.env); for (const [env, key, secret] of [ [prepareEnv, "BREV_API_KEY", "BREV_API_KEY"], [prepareEnv, "BREV_ORG_ID", "BREV_ORG_ID"], @@ -4112,6 +4197,51 @@ function validateStagingBrevLaunchableJob(errors: string[], jobs: WorkflowRecord } } +function validateStagingBrevLaunchableInput( + errors: string[], + dispatchInputs: WorkflowRecord, +): void { + const input = requireInput(errors, dispatchInputs, "include_staging_brev_launchable"); + if (input.type !== "boolean" || input.default !== false) { + errors.push( + "workflow_dispatch include_staging_brev_launchable input must be boolean and default to false", + ); + } + const description = stringValue(input.description); + if ( + !description.includes("Exact staging Brev Launchable") || + !description.includes("jobs and targets are empty") || + !description.includes("persistent repository readiness gate") + ) { + errors.push( + "workflow_dispatch include_staging_brev_launchable input must document qualification scope and readiness", + ); + } +} + +function validateStagingBrevLaunchableReadinessJob(errors: string[], jobs: WorkflowRecord): void { + const job = asRecord(jobs["staging-brev-launchable-readiness"]); + if (job.needs !== "generate-matrix") { + errors.push("staging-brev-launchable-readiness must depend on generate-matrix"); + } + const expectedIf = + "${{ github.event_name == 'workflow_dispatch' && inputs.include_staging_brev_launchable && inputs.jobs == '' && inputs.targets == '' && (github.repository != 'NVIDIA/NemoClaw' || github.ref != 'refs/heads/main' || vars.NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED != 'true') }}"; + if (job.if !== expectedIf) { + errors.push( + "staging-brev-launchable-readiness must fail only full dispatches with disabled readiness", + ); + } + if (job["runs-on"] !== "ubuntu-latest") { + errors.push("staging-brev-launchable-readiness must run on ubuntu-latest"); + } + const steps = asSteps(job.steps); + const fail = namedStep(steps, "Fail when staging Brev Launchable is not ready"); + requireRunContains(errors, fail, "NEMOCLAW_BREV_LAUNCHABLE_E2E_ENABLED=true"); + requireRunContains(errors, fail, "protected environment"); + requireRunContains(errors, fail, "staging Launchable ID"); + requireRunContains(errors, fail, "exit 1"); +} + export function validateE2eWorkflow(workflowValue: unknown): string[] { const workflow = asRecord(workflowValue); const errors: string[] = []; @@ -4144,6 +4274,7 @@ export function validateE2eWorkflow(workflowValue: unknown): string[] { const dispatchInputs = asRecord(workflowDispatch.inputs); requireInput(errors, dispatchInputs, "targets"); + validateStagingBrevLaunchableInput(errors, dispatchInputs); validateInferenceModeInput(errors, workflow, dispatchInputs); const jobsInput = requireInput(errors, dispatchInputs, "jobs"); const jobsDescription = stringValue(jobsInput.description); @@ -4165,6 +4296,11 @@ export function validateE2eWorkflow(workflowValue: unknown): string[] { if (permissions.contents !== "read") errors.push("workflow permissions.contents must be read"); const jobs = asRecord(workflow.jobs); + const expectedRunName = + "${{ inputs.checkout_sha != '' && format('E2E PR #{0} ({1})', inputs.pr_number, inputs.correlation_id) || inputs.correlation_id != '' && format('E2E {0} ({1})', github.ref_name, inputs.correlation_id) || format('E2E {0}', github.ref_name) }}"; + if (workflow["run-name"] !== expectedRunName) { + errors.push("workflow run-name must expose the unique manual-dispatch correlation ID"); + } errors.push(...validateJetsonRunnerDispatchBoundary(workflow)); const { errors: inventoryErrors, inventory: freeStandingInventory } = deriveFreeStandingJobsInventoryFromJobs(jobs); @@ -4611,6 +4747,7 @@ export function validateE2eWorkflow(workflowValue: unknown): string[] { validateSharedE2eJob(errors, jobs); validateStagingBrevLaunchableJob(errors, jobs); + validateStagingBrevLaunchableReadinessJob(errors, jobs); validateSkillAgentJob(errors, jobs); validateFreeStandingJobSelector(errors, jobs, "credential-migration", "credential-migration"); validateFreeStandingJobSelector(errors, jobs, "sessions-agents-cli", "sessions-agents-cli"); From b62382ade7340e58b81270bf2d8df5d07af2d8c9 Mon Sep 17 00:00:00 2001 From: "J. Yaunches" Date: Fri, 24 Jul 2026 14:31:30 -0400 Subject: [PATCH 2/4] test(e2e): reject malformed qualification evidence Signed-off-by: J. Yaunches --- test/maintainer-e2e-skill.test.ts | 20 ++++++++++++++++++++ 1 file changed, 20 insertions(+) diff --git a/test/maintainer-e2e-skill.test.ts b/test/maintainer-e2e-skill.test.ts index 1e6cb251605..b4cf5656596 100644 --- a/test/maintainer-e2e-skill.test.ts +++ b/test/maintainer-e2e-skill.test.ts @@ -142,6 +142,26 @@ describe("nemoclaw-maintainer-e2e evidence validation", () => { expect(() => validateFullE2eEvidence(evidence)).toThrow(message); }); + + it.each([ + [ + "a missing cleanup receipt", + (evidence: ReturnType) => ({ ...evidence, cleanup: undefined }), + "cleanup must be an object", + ], + [ + "a non-object dispatch receipt", + (evidence: ReturnType) => ({ ...evidence, dispatch: "invalid" }), + "dispatch must be an object", + ], + [ + "a non-object jobs response", + (evidence: ReturnType) => ({ ...evidence, jobs: [] }), + "jobs response must be an object", + ], + ])("rejects %s (#7487)", (_name, malformedEvidence, message) => { + expect(() => validateFullE2eEvidence(malformedEvidence(validEvidence()))).toThrow(message); + }); }); describe("nemoclaw-maintainer-e2e workflow routing", () => { From db2dc732bb02c949e31b49e1dc541a28b1a030ce Mon Sep 17 00:00:00 2001 From: Carlos Villela Date: Sun, 26 Jul 2026 07:33:38 -0700 Subject: [PATCH 3/4] fix(e2e): preserve pending release qualification Isolate each empty-selector full dispatch with its server-assigned run ID. Queue protected Brev qualification jobs so newer runs cannot replace pending release evidence. Signed-off-by: Carlos Villela --- .github/workflows/e2e.yaml | 3 ++- test/e2e/README.md | 7 +++++++ test/e2e/support/e2e-workflow.test.ts | 17 +++++++++++++++++ tools/e2e/workflow-boundary.mts | 23 +++++++++++++++++++++++ 4 files changed, 49 insertions(+), 1 deletion(-) diff --git a/.github/workflows/e2e.yaml b/.github/workflows/e2e.yaml index bcd7720424e..147225c51b9 100644 --- a/.github/workflows/e2e.yaml +++ b/.github/workflows/e2e.yaml @@ -79,7 +79,7 @@ permissions: pull-requests: read concurrency: - group: e2e-${{ github.ref }}-${{ inputs.checkout_sha != '' && format('pr-{0}', inputs.pr_number) || inputs.targets || 'supported' }}-${{ inputs.checkout_sha != '' && 'pr-gate' || inputs.jobs || 'all-jobs' }} + group: e2e-${{ github.ref }}-${{ inputs.checkout_sha != '' && format('pr-{0}', inputs.pr_number) || (inputs.include_staging_brev_launchable && inputs.jobs == '' && inputs.targets == '' && format('full-{0}', github.run_id)) || inputs.targets || 'supported' }}-${{ inputs.checkout_sha != '' && 'pr-gate' || inputs.jobs || 'all-jobs' }} cancel-in-progress: ${{ inputs.checkout_sha != '' }} env: @@ -278,6 +278,7 @@ jobs: contents: read concurrency: group: staging-brev-launchable-cpu + queue: max cancel-in-progress: false env: CANDIDATE_SHA: ${{ inputs.checkout_sha || github.sha }} diff --git a/test/e2e/README.md b/test/e2e/README.md index 587467d28ae..5b4f28b08d8 100644 --- a/test/e2e/README.md +++ b/test/e2e/README.md @@ -155,6 +155,13 @@ graph as the live targets: `post_to_slack=true`, which uses the preview Slack route. Branch-dispatched runs never receive Slack webhook secrets. +A manual run with `include_staging_brev_launchable=true` and empty `jobs` and +`targets` selectors is a full dispatch. Each full dispatch uses `github.run_id` +in its workflow concurrency identity, so another full dispatch cannot supersede +it while it waits. The protected `staging-brev-launchable` job uses the +non-cancelling `staging-brev-launchable-cpu` group with `queue: max`, so pending +qualifications remain queued instead of replacing one another. + ### Runner comparison telemetry Trusted `main` runs without an alternate checkout SHA record runner-comparison diff --git a/test/e2e/support/e2e-workflow.test.ts b/test/e2e/support/e2e-workflow.test.ts index acb40831559..aec89d6b8b2 100644 --- a/test/e2e/support/e2e-workflow.test.ts +++ b/test/e2e/support/e2e-workflow.test.ts @@ -156,6 +156,23 @@ describe("e2e workflow boundary", () => { ); }); + it("rejects superseding full-dispatch and qualification concurrency drift (#7487)", () => { + const workflow = readWorkflow() as { + concurrency: Record; + jobs: Record }>; + }; + workflow.concurrency.group = + "e2e-${{ github.ref }}-${{ inputs.checkout_sha != '' && format('pr-{0}', inputs.pr_number) || inputs.targets || 'supported' }}-${{ inputs.checkout_sha != '' && 'pr-gate' || inputs.jobs || 'all-jobs' }}"; + delete workflow.jobs["staging-brev-launchable"]!.concurrency!.queue; + + expect(validateE2eWorkflow(workflow)).toEqual( + expect.arrayContaining([ + "workflow concurrency must isolate each full dispatch with github.run_id", + "staging-brev-launchable concurrency must queue all pending qualifications without cancellation", + ]), + ); + }); + it("keeps network-policy scenarios isolated with cleanup reserve", () => { const workflow = readWorkflow() as { jobs: Record< diff --git a/tools/e2e/workflow-boundary.mts b/tools/e2e/workflow-boundary.mts index ec95cd6e4e7..3c05cb59abb 100644 --- a/tools/e2e/workflow-boundary.mts +++ b/tools/e2e/workflow-boundary.mts @@ -4113,6 +4113,18 @@ function validateInferenceModeGeneration( requireRunContains(errors, step, "--ci-output"); } +function validateFullE2eConcurrency(errors: string[], workflow: WorkflowRecord): void { + const concurrency = asRecord(workflow.concurrency); + const expectedGroup = + "e2e-${{ github.ref }}-${{ inputs.checkout_sha != '' && format('pr-{0}', inputs.pr_number) || (inputs.include_staging_brev_launchable && inputs.jobs == '' && inputs.targets == '' && format('full-{0}', github.run_id)) || inputs.targets || 'supported' }}-${{ inputs.checkout_sha != '' && 'pr-gate' || inputs.jobs || 'all-jobs' }}"; + if (concurrency.group !== expectedGroup) { + errors.push("workflow concurrency must isolate each full dispatch with github.run_id"); + } + if (concurrency["cancel-in-progress"] !== "${{ inputs.checkout_sha != '' }}") { + errors.push("workflow concurrency must cancel only superseded PR gate runs"); + } +} + function validateStagingBrevLaunchableJob(errors: string[], jobs: WorkflowRecord): void { const job = asRecord(jobs["staging-brev-launchable"]); const environment = asRecord(job.environment); @@ -4133,6 +4145,16 @@ function validateStagingBrevLaunchableJob(errors: string[], jobs: WorkflowRecord "staging-brev-launchable must run for schedules, explicit selection, or an empty-selector full dispatch", ); } + const concurrency = asRecord(job.concurrency); + if ( + concurrency.group !== "staging-brev-launchable-cpu" || + concurrency.queue !== "max" || + concurrency["cancel-in-progress"] !== false + ) { + errors.push( + "staging-brev-launchable concurrency must queue all pending qualifications without cancellation", + ); + } const steps = asSteps(job.steps); const prepare = requireStep(errors, steps, "Prepare the trusted lane"); const prepareEnv = asRecord(prepare?.env); @@ -4274,6 +4296,7 @@ export function validateE2eWorkflow(workflowValue: unknown): string[] { const dispatchInputs = asRecord(workflowDispatch.inputs); requireInput(errors, dispatchInputs, "targets"); + validateFullE2eConcurrency(errors, workflow); validateStagingBrevLaunchableInput(errors, dispatchInputs); validateInferenceModeInput(errors, workflow, dispatchInputs); const jobsInput = requireInput(errors, dispatchInputs, "jobs"); From 57b73834d85cf37d42526ba78918e92b89a92a25 Mon Sep 17 00:00:00 2001 From: Carlos Villela Date: Sun, 26 Jul 2026 11:53:47 -0700 Subject: [PATCH 4/4] test(e2e): update recovery run identity Signed-off-by: Carlos Villela --- test/hosted-runner-recovery-workflow.test.ts | 3 ++- 1 file changed, 2 insertions(+), 1 deletion(-) diff --git a/test/hosted-runner-recovery-workflow.test.ts b/test/hosted-runner-recovery-workflow.test.ts index 2f3f75be074..8ef9464766f 100644 --- a/test/hosted-runner-recovery-workflow.test.ts +++ b/test/hosted-runner-recovery-workflow.test.ts @@ -12,7 +12,7 @@ const PLATFORM_WORKFLOW_PATH = ".github/workflows/platform-vitest-main.yaml"; const TRUSTED_CHECKOUT = "actions/checkout@df4cb1c069e1874edd31b4311f1884172cec0e10"; const TRUSTED_SETUP_NODE = "actions/setup-node@820762786026740c76f36085b0efc47a31fe5020"; const E2E_RUN_NAME = - "${{ inputs.checkout_sha != '' && format('E2E PR #{0} ({1})', inputs.pr_number, inputs.correlation_id) || format('E2E {0}', github.ref_name) }}"; + "${{ inputs.checkout_sha != '' && format('E2E PR #{0} ({1})', inputs.pr_number, inputs.correlation_id) || inputs.correlation_id != '' && format('E2E {0} ({1})', github.ref_name, inputs.correlation_id) || format('E2E {0}', github.ref_name) }}"; type RecoveryWorkflow = { name: string; @@ -86,6 +86,7 @@ describe("hosted-runner recovery workflow boundary", () => { const platform = sourceWorkflow(PLATFORM_WORKFLOW_PATH); expect(e2e).toMatchObject({ name: "E2E", "run-name": E2E_RUN_NAME }); + expect(E2E_RUN_NAME).toContain("inputs.correlation_id != ''"); expect(E2E_RUN_NAME).toContain("format('E2E {0}', github.ref_name)"); expect([wsl.name, macos.name, platform.name]).toEqual([ "E2E / WSL",