diff --git a/.github/workflows/enterprise-dates.yml b/.github/workflows/enterprise-dates.yml index 2bc66b8a6cf6..83b9c63afdee 100644 --- a/.github/workflows/enterprise-dates.yml +++ b/.github/workflows/enterprise-dates.yml @@ -1,17 +1,12 @@ name: Enterprise date updater -# **What it does**: Runs on a schedule to update -# src/ghes-releases/lib/enterprise-dates.json. -# **Why we have it**: The src/ghes-releases/lib/enterprise-dates.json -# file needs to be up-to-date for the -# Used to display deprecation banner dates and as a reference -# for all past server release numbers. -# **Who does it impact**: Docs engineering, docs content. +# Keep src/ghes-releases/lib/enterprise-dates.json current for deprecation banner dates +# and historical GitHub Enterprise Server release references. on: workflow_dispatch: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' permissions: contents: write @@ -36,10 +31,11 @@ jobs: id: create-pull-request uses: peter-evans/create-pull-request@98357b18bf14b5342f975ff684046ec3b2a07725 # pin @v8.0.0 env: - # Disable pre-commit hooks; they don't play nicely here + # Create-pull-request runs git commands itself, so disable hooks that can + # change the worktree. HUSKY: '0' with: - # need to use a token with repo and workflow scopes for this step + # This PR update requires repo and workflow scopes. token: ${{ secrets.DOCS_BOT_PAT_BASE }} commit-message: '🤖 ran src/ghes-releases/scripts/update-enterprise-dates.ts' title: 🤖 src/ghes-releases/lib/enterprise-dates.json update diff --git a/.github/workflows/enterprise-release-issue.yml b/.github/workflows/enterprise-release-issue.yml index ac929c5f06a6..a6308e611252 100644 --- a/.github/workflows/enterprise-release-issue.yml +++ b/.github/workflows/enterprise-release-issue.yml @@ -1,13 +1,12 @@ name: Open Enterprise release or deprecation issue -# **What it does**: Checks if there is an Enterprise release or deprecation upcoming, and if so, opens an issue with the tasks to be completed. -# **Why we have it**: GHES releases and deprecations run on a predictable schedule, so we can automate some of the project management aspects. -# **Who does it impact**: Docs engineering, docs content. +# GHES releases and deprecations follow a predictable schedule, so this workflow opens +# their planning issues automatically. on: workflow_dispatch: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' permissions: contents: read diff --git a/.github/workflows/expertise-required-label-message.yml b/.github/workflows/expertise-required-label-message.yml index a3e3613cec43..153d4f040f04 100644 --- a/.github/workflows/expertise-required-label-message.yml +++ b/.github/workflows/expertise-required-label-message.yml @@ -1,8 +1,6 @@ name: Expertise Required label message -# **What it does**: Adds a bot comment stating a certain level of expertise is required to a docs-content issue when the `contributor-expertise-required` label is applied -# **Why we have it**: We need a method to surface a message denoting if an issue requires a certain level of expertise in order to be resolved -# **Who does it impact**: Open Source and Hubbers +# Surface contributor expertise requirements directly on labeled github/docs issues. on: issues: diff --git a/.github/workflows/feedback-prompt.yml b/.github/workflows/feedback-prompt.yml index dace8f90a87f..9d335eebd14f 100644 --- a/.github/workflows/feedback-prompt.yml +++ b/.github/workflows/feedback-prompt.yml @@ -10,9 +10,8 @@ permissions: jobs: comment-on-pr: - # This workflow should only run on the 'github/docs-internal' repository because it posts a feedback request - # to non-Docs team contributors when their PR is merged into the main branch. - # The feedback request asks contributors to leave feedback on their contributing experience in Slack. + # Ask merged main-branch pull request authors for feedback when they are outside + # the docs-content team. if: github.repository == 'github/docs-internal' && github.event.pull_request.merged == true && github.event.pull_request.base.ref == 'main' runs-on: ubuntu-latest @@ -26,16 +25,13 @@ jobs: script: | try { const pr = context.payload.pull_request; - // Team is addressed by numeric ID (org github = 9919, team docs-content = 2796154) - // because IDs survive team renames and slugs do not. + // Numeric IDs survive team renames; slugs do not. GitHub org 9919, docs-content team 2796154. await github.request('GET /organizations/{org_id}/team/{team_id}/memberships/{username}', { org_id: 9919, team_id: 2796154, username: pr.user.login, }); - // Author is in the team. Do nothing! } catch(err) { - // Author not in team core.exportVariable('NON_DOCS_HUBBER', 'true'); } @@ -66,7 +62,6 @@ jobs: " - Thanks for your contribution! " + "If you think something could be improved about the contributor experience, please post in `#docs-contributor-feedback` on Slack."; } else { - // nobody to mention! commentBody = "👋 Thanks for your contribution! " + "If you think something could be improved about the contributor experience, please post in `#docs-contributor-feedback` on Slack."; diff --git a/.github/workflows/first-responder-v2-prs-collect.yml b/.github/workflows/first-responder-v2-prs-collect.yml index ff0d2aba47c7..2325865e6577 100644 --- a/.github/workflows/first-responder-v2-prs-collect.yml +++ b/.github/workflows/first-responder-v2-prs-collect.yml @@ -1,8 +1,6 @@ name: Add maintenance PRs to the docs-content FR project v2 -# **What it does**: Adds docs-internal pull requests authored by docs-bot to the docs-content FR project v2 -# **Why we have it**: So we don't lose track of maintenance pull requests for docs-content to review -# **Who does it impact**: Docs content +# The docs-content FR project is the review queue for docs-bot maintenance pull requests. on: pull_request: @@ -25,10 +23,6 @@ jobs: steps: - name: Checkout repository uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 - - # Add to the FR project - # and set type to "Maintenance" - # and set date to now - name: Triage to docs-content FR project env: GITHUB_TOKEN: ${{ secrets.DOCS_BOT_PAT_BASE }} diff --git a/.github/workflows/generate-code-scanning-query-lists.yml b/.github/workflows/generate-code-scanning-query-lists.yml index 0558abd14e33..d1c447c287fc 100644 --- a/.github/workflows/generate-code-scanning-query-lists.yml +++ b/.github/workflows/generate-code-scanning-query-lists.yml @@ -1,11 +1,7 @@ name: Generate code scanning query lists -# **What it does**: This workflow is currently run manually approximately every two weeks as part -# of the release process for the CodeQL CLI. We hope to automate this in the future -# When run, this workflow generates updated query lists with data from the codeql -# repository, and creates a pull request if there are updates. -# **Why we have it**: So we can automate CodeQL query tables and show code scanning users the built in queries. -# **Who does it impact**: Anyone making CodeQL query suite changes in `github/codeql`, and wanting to get them published on the docs site. +# Manual CodeQL CLI release runs, about every two weeks, generate query table reusables +# from github/codeql and open a pull request when they change. on: workflow_dispatch: @@ -53,16 +49,14 @@ jobs: echo "Copied files from github/codeql repo. Commit SHA: $OPENAPI_COMMIT_SHA" - name: Download CodeQL CLI - # Look under the `codeql` directory, as this is where we checked out the `github/codeql` repo + # fetch-codeql lives in the checked-out github/codeql repository. uses: ./codeql/.github/actions/fetch-codeql - name: Test CodeQL CLI Download shell: bash run: codeql --version - # "Server for running multiple commands while avoiding repeated JVM initialization." - # Having started this should speed up the execution of the various - # CLI calls of the executable. + # Start the CodeQL CLI server once so later CodeQL commands avoid repeated JVM initialization. - name: Start CodeQL CLI server in the background shell: bash run: | @@ -72,10 +66,8 @@ jobs: - uses: ./.github/actions/install-cocofix with: - # The Docs Engineering Bot app cannot read the org-scoped - # @github/cocofix package (its Packages permission is repo-level - # only), so this step keeps using the PAT until the app is granted - # organization package read access. + # The Docs Engineering Bot app has repo-level Packages permission and cannot + # read org-scoped packages, so cocofix installation requires the PAT. token: ${{ secrets.DOCS_BOT_PAT_BASE }} - name: Build code scanning security query lists @@ -123,16 +115,14 @@ jobs: echo "Copied files from github/codeql repo. Commit SHA: $OPENAPI_COMMIT_SHA" - name: Download CodeQL CLI - # Look under the `codeql` directory, as this is where we checked out the `github/codeql` repo + # fetch-codeql lives in the checked-out github/codeql repository. uses: ./codeql/.github/actions/fetch-codeql - name: Test CodeQL CLI Download shell: bash run: codeql --version - # "Server for running multiple commands while avoiding repeated JVM initialization." - # Having started this should speed up the execution of the various - # CLI calls of the executable. + # Start the CodeQL CLI server once so later CodeQL commands avoid repeated JVM initialization. - name: Start CodeQL CLI server in the background shell: bash run: | @@ -210,12 +200,10 @@ jobs: shell: bash run: | - # When we started, we downloaded the CodeQL CLI here in this workflow. - # We have no intention of checking that in but we also don't want - # `git status ...` to show it as an untracked file. + # Git status must only report generated query tables, so remove the + # checked-out CodeQL repository. rm -fr ./codeql - # If nothing to commit, exit now. It's fine. No orphans. changes=$(git diff --name-only | wc -l) untracked=$(git status --untracked-files --short | wc -l) if [[ $changes -eq 0 ]] && [[ $untracked -eq 0 ]]; then @@ -228,12 +216,10 @@ jobs: branchname=codeql-query-tables-${{ steps.codeql.outputs.OPENAPI_COMMIT_SHA }} - # Exit if the branch already exists. Since the actions/checkout fetch-depth is 1, - # it doesn't "know" about branches locally, so we need to manually list them. + # Query the remote because actions/checkout with fetch-depth 1 omits other remote-tracking branches. branchExists=$(git ls-remote --heads origin refs/heads/$branchname | wc -l) - # When run on a pull_request, we're just testing the tooling. - # Exit before it actually pushes the possible changes. + # Pull request runs validate generated files without pushing branches. if [ "$DRY_RUN" = "true" ]; then echo "Dry-run mode when run in a pull request" echo "See the 'Insight into diff' step for the changes it would create PR about." diff --git a/.github/workflows/headless-tests.yml b/.github/workflows/headless-tests.yml index 54e844c36906..a000570dadad 100644 --- a/.github/workflows/headless-tests.yml +++ b/.github/workflows/headless-tests.yml @@ -1,10 +1,6 @@ name: Headless Tests -# **What it does**: This runs our browser tests to test things that depend -# on client-side JavaScript. -# **Why we have it**: Because most automated vitest tests only test static -# input and outputs. -# **Who does it impact**: Docs engineering, open-source engineering contributors. +# Browser tests cover client-side JavaScript behavior that static Vitest tests miss. on: workflow_dispatch: @@ -14,7 +10,7 @@ on: permissions: contents: read -# This allows a subsequently queued workflow run to interrupt previous runs +# Cancel older runs for the same ref to save runner time. concurrency: group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}' cancel-in-progress: true @@ -27,8 +23,7 @@ jobs: if: github.repository == 'github/docs-internal' || github.repository == 'github/docs' runs-on: ubuntu-latest strategy: - # When we're comfortable a11y tests aren't generating false positives and helping, - # let's remove the matrix and just run playwright in a single job. + # Keep Playwright variants as separate checks while a11y false-positive rates settle. matrix: node: - playwright-rendering @@ -58,11 +53,9 @@ jobs: - name: Run Playwright tests env: PLAYWRIGHT_WORKERS: ${{ fromJSON('[1, 4]')[github.repository == 'github/docs-internal'] }} - # workaround for https://github.com/nodejs/node/issues/59364 as of 22.18.0 + # Supported Node runtimes can hit https://github.com/nodejs/node/issues/59364 without this flag. NODE_OPTIONS: '--no-experimental-strip-types' PLAYWRIGHT_TIMEOUT: ${{ matrix.node == 'playwright-a11y' && '60000' || '' }} - # Run playwright rendering tests and a11y tests (axe scans) as distinct checks - # so that we can run them without blocking merges until we can be confident - # results for a11y tests are meaningul and scenarios we're testing are correct. + # Keep a11y scans in a distinct check so branch protection can leave them non-blocking. run: npm run playwright-test -- ${{ matrix.node }} --reporter list diff --git a/.github/workflows/hubber-contribution-help.yml b/.github/workflows/hubber-contribution-help.yml index 45eec0641fde..3f49c8fce110 100644 --- a/.github/workflows/hubber-contribution-help.yml +++ b/.github/workflows/hubber-contribution-help.yml @@ -1,8 +1,6 @@ name: Hubber contribution help -# **What it does**: When a PR is opened by a non-Docs team Hubber, adds a bot comment with helpful links -# **Why we have it**: To help non–Docs Hubbers navigate how to get a PR reviewed by the Docs team -# **Who does it impact**: docs-internal contributors +# Help non-Docs Hubbers route content pull requests to the Docs Content review board. on: pull_request: @@ -31,8 +29,7 @@ jobs: github-token: ${{ secrets.DOCS_BOT_PAT_BASE }} script: | try { - // Team is addressed by numeric ID (org github = 9919, team docs = 325922) - // because IDs survive team renames and slugs do not. + // Numeric IDs survive team renames; slugs do not. GitHub org 9919, docs team 325922. await github.request('GET /organizations/{org_id}/team/{team_id}/memberships/{username}', { org_id: 9919, team_id: 325922, diff --git a/.github/workflows/index-autocomplete-search.yml b/.github/workflows/index-autocomplete-search.yml index d9c4d96418ad..5494b4e3eb1e 100644 --- a/.github/workflows/index-autocomplete-search.yml +++ b/.github/workflows/index-autocomplete-search.yml @@ -1,13 +1,11 @@ name: Index autocomplete search in Elasticsearch -# **What it does**: Indexes AI search autocomplete data into Elasticsearch. -# **Why we have it**: So we can power the APIs for AI search autocomplete. -# **Who does it impact**: docs-engineering +# Keep Elasticsearch autocomplete data current for AI search APIs. on: workflow_dispatch: schedule: - - cron: '20 16 * * 1-5' # Run Mon-Fri at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1-5' pull_request: paths: - .github/workflows/index-autocomplete-search.yml diff --git a/.github/workflows/index-general-search-pr.yml b/.github/workflows/index-general-search-pr.yml index ca386f5af510..26be1f8aabcd 100644 --- a/.github/workflows/index-general-search-pr.yml +++ b/.github/workflows/index-general-search-pr.yml @@ -1,9 +1,7 @@ name: Index general search in Elasticsearch on PR -# **What it does**: This does what `index-general-search-elasticsearch.yml` does but -# with a localhost Elasticsearch and only for English. -# **Why we have it**: To test that the script works and the popular pages json is valid. -# **Who does it impact**: Docs engineering +# Test general search indexing against local Elasticsearch and validate popular pages JSON +# before merge. on: workflow_dispatch: @@ -11,23 +9,22 @@ on: paths: - 'src/search/**' - 'package*.json' - # For debugging this workflow + # Debugging changes to this workflow need the same PR index test. - .github/workflows/index-general-search-pr.yml - # Make sure we run this if the composite action changes + # Setup changes can break the local Elasticsearch path this workflow tests. - .github/actions/setup-elasticsearch/action.yml permissions: contents: read -# This allows a subsequently queued workflow run to interrupt previous runs +# Cancel older runs for the same ref to save runner time. concurrency: group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}' cancel-in-progress: true env: ELASTICSEARCH_URL: http://localhost:9200 - # Since we'll run in NDOE_ENV=production, we need to be explicit that - # we don't want Hydro configured. + # Empty Hydro credentials keep production-mode test runs from sending analytics. HYDRO_ENDPOINT: '' HYDRO_SECRET: '' @@ -43,7 +40,7 @@ jobs: uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: repository: github/docs-internal-data - # This works because user `docs-bot` has read access to that private repo. + # docs-bot can read the private github/docs-internal-data repository. token: ${{ secrets.DOCS_BOT_PAT_BASE }} path: docs-internal-data @@ -62,7 +59,6 @@ jobs: run: | npm run general-search-scrape-server > /tmp/stdout.log 2> /tmp/stderr.log & - # first sleep to give it a chance to start sleep 6 curl --retry-connrefused --retry 6 -I http://localhost:4002/ @@ -76,13 +72,9 @@ jobs: - name: Scrape records into a temp directory env: - # If a reusable, or anything in the `data/*` directory is deleted - # you might get a - # - # RenderError: Can't find the key 'site.data.reusables...' in the scope - # - # But that'll get fixed in the next translation pipeline. For now, - # let's just accept an empty string instead. + # Deleting a reusable or data/* entry can leave translations pointing at missing + # site.data keys and trigger RenderError. Accept empty strings until the next + # translation pipeline removes those references. THROW_ON_EMPTY: false DOCS_INTERNAL_DATA: docs-internal-data diff --git a/.github/workflows/index-general-search.yml b/.github/workflows/index-general-search.yml index 3e7809b10f5e..99a0261f6af8 100644 --- a/.github/workflows/index-general-search.yml +++ b/.github/workflows/index-general-search.yml @@ -1,9 +1,7 @@ name: Index general search in Elasticsearch -# **What it does**: It scrapes the whole site and dumps the records in a -# temp directory. Then it indexes that into Elasticsearch. -# **Why we have it**: We want our search indexes kept up to date. -# **Who does it impact**: Anyone using search on docs. +# Keep general search indexes current by scraping docs content and indexing the records +# into Elasticsearch. on: workflow_dispatch: @@ -17,7 +15,7 @@ on: required: false default: '' schedule: - - cron: '20 16 * * 1-5' # Run Mon-Fri at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1-5' workflow_run: workflows: ['Purge Fastly'] types: @@ -26,24 +24,20 @@ on: permissions: contents: read -# This allows a subsequently queued workflow run to cancel previous runs. -# Include the triggering workflow's conclusion in the group so that runs triggered -# by skipped Purge Fastly workflows don't cancel runs triggered by successful ones. +# Include the Purge Fastly conclusion so skipped purge runs cannot cancel valid indexing runs. concurrency: group: '${{ github.workflow }} @ ${{ github.head_ref }} ${{ github.event_name }} ${{ github.event.workflow_run.conclusion }}' cancel-in-progress: true env: ELASTICSEARCH_URL: ${{ secrets.ELASTICSEARCH_URL }} - # Since we'll run in NODE_ENV=production, we need to be explicit that - # we don't want Hydro configured. + # Empty Hydro credentials keep production-mode indexing runs from sending analytics. HYDRO_ENDPOINT: '' HYDRO_SECRET: '' jobs: figureOutMatrix: - # Skip immediately if triggered by a non-successful Purge Fastly run. - # This prevents skipped runs from canceling valid indexing runs via concurrency. + # Skip non-successful Purge Fastly workflow_run events. if: ${{ github.repository == 'github/docs-internal' && (github.event_name != 'workflow_run' || github.event.workflow_run.conclusion == 'success') }} runs-on: ubuntu-latest outputs: @@ -53,18 +47,15 @@ jobs: id: set-matrix with: script: | - // Edit this list for the definitive list of languages - // (other than English) we want to index in Elasticsearch. + // allNonEnglish is the source of truth for non-English Elasticsearch languages. const allNonEnglish = 'es,ja,pt,zh,ru,fr,ko,de'.split(',') const allPossible = ["en", ...allNonEnglish] if (context.eventName === "workflow_run") { - // Job-level `if` already ensures we only get here for successful runs, - // but keep this as a safety check. + // The job filter keeps non-successful workflow_run events out, so treat this as a guard. if (context.payload.workflow_run.conclusion === "success") { return ["en"] } - // This shouldn't happen due to job-level filter, but handle gracefully. console.warn(`Unexpected: workflow_run with conclusion '${context.payload.workflow_run.conclusion}'`) return [] } @@ -113,15 +104,9 @@ jobs: runs-on: ubuntu-latest strategy: fail-fast: false - # When it's only English (i.e. a simple array of ['en']), this value - # does not matter. If it's ALL the languages, then we know we can - # be patient because it's a daily scheduled run and it's run by bots - # while humans are asleep. So there's no rush and no need to finish - # the whole job fast. - # As of June 2023, it takes about 10+ minutes to index one whole - # language and we have 8 non-English languages. - # As of May 2025, we index so many pages that we are being rate-limited by - # Elasticsearch. So we are shrinking this value to 2, down from 3 + # English-only runs have one matrix item, so max-parallel does not matter. + # Full runs index eight non-English languages, each taking more than 10 minutes. + # Elasticsearch rate-limits higher concurrency, so keep this at 2. max-parallel: 2 matrix: language: ${{ fromJSON(needs.figureOutMatrix.outputs.matrix) }} @@ -132,7 +117,7 @@ jobs: uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: repository: github/docs-internal-data - # This works because user `docs-bot` has read access to that private repo. + # docs-bot can read the private github/docs-internal-data repository. token: ${{ secrets.DOCS_BOT_PAT_BASE }} path: docs-internal-data @@ -155,7 +140,6 @@ jobs: run: | npm run general-search-scrape-server > /tmp/stdout.log 2> /tmp/stderr.log & - # first sleep to give it a chance to start sleep 6 curl --retry-connrefused --retry 6 -I http://localhost:4002/ @@ -169,17 +153,12 @@ jobs: - name: Scrape records into a temp directory env: - # If a reusable, or anything in the `data/*` directory is deleted - # you might get a - # - # RenderError: Can't find the key 'site.data.reusables...' in the scope - # - # But that'll get fixed in the next translation pipeline. For now, - # let's just accept an empty string instead. + # Deleting a reusable or data/* entry can leave translations pointing at missing + # site.data keys and trigger RenderError. Accept empty strings until the next + # translation pipeline removes those references. THROW_ON_EMPTY: false - # Note that by default, this is '' (empty string) and that means - # the same as not set within the script. + # Empty VERSION makes the scrape script use its default version selection. VERSION: ${{ inputs.version }} DOCS_INTERNAL_DATA: docs-internal-data @@ -210,9 +189,7 @@ jobs: - name: Index into Elasticsearch env: - # Must match what we used when scraping (npm run general-search-scrape) - # otherwise the script will seek other versions from disk that might - # not exist. + # Match the scrape version so the indexer reads the directory the scraper wrote. VERSION: ${{ inputs.version }} run: | npm run index-general-search -- /tmp/records \ @@ -222,9 +199,7 @@ jobs: - name: Check created indexes and aliases run: | - # Not using `--fail` here because I've observed that it can fail - # with a rather cryptic 404 error when it should, if anything, be - # a 200 OK with a list of no indices. + # Avoid --fail because an empty index list can return a 404 instead of 200 OK. curl --retry-connrefused --retry 5 ${{ env.ELASTICSEARCH_URL }}/_cat/indices?v curl --retry-connrefused --retry 5 ${{ env.ELASTICSEARCH_URL }}/_cat/indices?v @@ -302,8 +277,7 @@ jobs: FILE_URL: ${{ github.server_url }}/${{ github.repository }}/blob/main/.github/workflows/index-general-search.yml WORKFLOW_NAME: ${{ github.workflow }} run: | - # Reuse the oldest open scraping-failures issue if one exists, - # to keep the noise down. Otherwise open a new one. + # Reuse the oldest open scraping-failures issue to keep the issue list quiet. existing_issue=$(gh issue list \ --repo github/technical-content \ --label "search-scraping-failures" \ diff --git a/.github/workflows/keep-caches-warm.yml b/.github/workflows/keep-caches-warm.yml index d405d2454e96..ad3420362004 100644 --- a/.github/workflows/keep-caches-warm.yml +++ b/.github/workflows/keep-caches-warm.yml @@ -1,18 +1,7 @@ name: Keep caches warm -# **What it does**: -# Makes sure the caching of ./node_modules and ./.next is kept warm -# for making other pull requests faster. -# We also use this workflow to precompute other things so that the -# actions cache is warmed up with data available during deployment -# actions. When you use actions/cache within a run on `main` -# what gets saved can be used by other pull requests. But it's -# also so that when we make production deployments, -# we can just rely on the cache to already be warmed up. -# **Why we have it**: -# A PR workflow that depends on caching can't reuse a -# cached artifact acorss PRs unless it also runs on `main`. -# **Who does it impact**: Docs engineering, open-source engineering contributors. +# Main-branch runs warm node_modules, Next.js, remote JSON, and pageinfo caches that +# pull request workflows and production deployments can reuse. on: workflow_dispatch: diff --git a/.github/workflows/line-endings.yml b/.github/workflows/line-endings.yml index dc52c09925f9..b325dd9017dd 100644 --- a/.github/workflows/line-endings.yml +++ b/.github/workflows/line-endings.yml @@ -1,14 +1,7 @@ name: Line endings -# **What it does**: Fails if any committed file violates the repo's `.gitattributes` -# line-ending rules (for example, CRLF in a file declared `eol=lf`). -# **Why we have it**: `.gitattributes` (`*.md text eol=lf`, `* text=auto`) only -# normalizes line endings during a local `git add`/checkout. A server-side -# squash-merge can commit CRLF blobs that bypass it. A file with mixed line endings -# then shows as permanently "modified" on Linux runners, which breaks PR-creating -# sync workflows that use peter-evans/create-pull-request. See -# github/docs-engineering#6657 and github/docs-engineering#6656. -# **Who does it impact**: Docs engineering, open-source engineering contributors. +# .gitattributes (*.md text eol=lf, * text=auto) normalizes local adds and checkouts, +# but CRLF blobs from server-side squash merges look modified and break peter-evans/create-pull-request. on: workflow_dispatch: @@ -18,7 +11,7 @@ on: permissions: contents: read -# This allows a subsequently queued workflow run to interrupt previous runs +# Cancel older runs for the same ref to save runner time. concurrency: group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}' cancel-in-progress: true @@ -32,8 +25,7 @@ jobs: uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 - name: Check line endings - # Re-stage every file through the `.gitattributes` filters and fail if any - # committed blob disagrees with them (usually CRLF where LF is required). + # Re-stage through .gitattributes so git can detect blobs that check out differently. run: | git add --renormalize . if ! git diff --cached --quiet; then diff --git a/.github/workflows/link-check-external.yml b/.github/workflows/link-check-external.yml index 36e3ef3d6561..376f6c8e5d5c 100644 --- a/.github/workflows/link-check-external.yml +++ b/.github/workflows/link-check-external.yml @@ -2,7 +2,7 @@ name: 'Link Check: External' on: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' workflow_dispatch: inputs: max_urls: @@ -17,9 +17,9 @@ jobs: check-external-links: if: github.repository == 'github/docs-internal' runs-on: ubuntu-latest - timeout-minutes: 180 # 3 hours for external checks - # Serialize publishing so two overlapping runs can't both create a - # "rolling" issue, or write their results out of order. + timeout-minutes: 180 # External checks can take hours over all URLs. + # Serialize the full external link check so overlapping runs cannot publish duplicate + # rolling issues or write results out of order. concurrency: group: broken-external-links-report cancel-in-progress: false @@ -77,8 +77,7 @@ jobs: const repo = 'docs-content' const label = 'broken link report' - // GitHub rejects issue bodies over 65536 characters with a 422. - // Truncate and point at the run artifact for the full contents. + // GitHub rejects issue bodies over 65536 characters with a 422, so truncate and point at the run artifact. const MAX_BODY_SIZE = 60000 const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}` let body = fs.readFileSync('artifacts/external-link-report.md', 'utf8') @@ -88,9 +87,7 @@ jobs: core.warning(`Report exceeded ${MAX_BODY_SIZE} characters, so it was truncated.`) } - // Reuse a single rolling issue instead of opening a new one every - // week, which floods the first responders' board. Find the open - // report issues (newest first). + // Reuse one rolling issue so weekly failures do not flood the first responders' board; sort newest first. const open = await github.paginate(github.rest.issues.listForRepo, { owner, repo, @@ -114,8 +111,7 @@ jobs: return } - // Refresh the newest open report in place and close any older - // duplicates so exactly one canonical issue remains. + // Refresh the newest open report and close duplicates so one canonical issue remains. const [canonical, ...superseded] = reportIssues await github.rest.issues.update({ owner, @@ -126,8 +122,7 @@ jobs: }) core.info(`Updated rolling report issue: ${canonical.html_url}`) - // Attempt every duplicate even if one fails, so a single transient - // API error doesn't leave the rest open. + // Close duplicates independently so one transient API error cannot leave the rest open. const results = await Promise.allSettled( superseded.map(async (issue) => { await github.rest.issues.createComment({ @@ -169,8 +164,7 @@ jobs: const repo = 'docs-content' const label = 'broken link report' - // A clean run means the open report is stale. Leaving it open would - // keep fixed failures on the first responders' board. + // Close stale reports so fixed failures leave the first responders' board. const open = await github.paginate(github.rest.issues.listForRepo, { owner, repo, diff --git a/.github/workflows/link-check-github-github.yml b/.github/workflows/link-check-github-github.yml index 7f6e14c92c9d..730d85f11192 100644 --- a/.github/workflows/link-check-github-github.yml +++ b/.github/workflows/link-check-github-github.yml @@ -1,13 +1,11 @@ name: 'Link Check: github/github' -# **What it does**: This checks for any broken docs.github.com links in github/github -# **Why we have it**: Make sure all docs in github/github are up to date -# **Who does it impact**: Docs engineering, people on GitHub +# Check docs.github.com links in github/github so product documentation stays current. on: workflow_dispatch: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' permissions: contents: read @@ -24,7 +22,7 @@ jobs: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: - # To prevent issues with cloning early access content later + # Do not persist the default token, so later git operations use their own credentials. persist-credentials: 'false' - uses: ./.github/actions/node-npm-setup @@ -47,15 +45,13 @@ jobs: - name: Run broken github/github link check env: - # Needs a token with access to github/github; the app token is scoped to it above + # The link checker requires a token that can read github/github. GITHUB_TOKEN: ${{ secrets.DOCS_BOT_PAT_BASE }} run: | npm run check-github-github-links -- broken_github_github_links.md - name: Get title for issue - # If the file 'broken_github_github_links.md' got created, - # the hash of it will not be an empty string. That means if found - # broken links, we want to create an issue. + # The link checker creates broken_github_github_links.md only when it finds broken links. if: ${{ hashFiles('broken_github_github_links.md') != '' }} id: check run: echo "title=$(head -1 broken_github_github_links.md)" >> $GITHUB_OUTPUT diff --git a/.github/workflows/link-check-internal.yml b/.github/workflows/link-check-internal.yml index 482948c6ab28..fb14cf9115de 100644 --- a/.github/workflows/link-check-internal.yml +++ b/.github/workflows/link-check-internal.yml @@ -2,7 +2,7 @@ name: 'Link Check: Internal' on: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' workflow_dispatch: inputs: version: @@ -29,7 +29,6 @@ permissions: contents: read jobs: - # Determine which version/language combos to run setup-matrix: if: github.repository == 'github/docs-internal' runs-on: ubuntu-latest @@ -44,13 +43,10 @@ jobs: id: set-matrix run: | if [[ "${EVENT_NAME}" == "workflow_dispatch" ]]; then - # Manual run: use the provided version and language echo "matrix={\"include\":[{\"version\":\"${INPUT_VERSION}\",\"language\":\"${INPUT_LANGUAGE}\"}]}" >> $GITHUB_OUTPUT else - # Scheduled run: every published version, in English. A link can be broken in - # one version and fine in another, so checking two of eight left most of the - # site unchecked. The report job merges the results, so this does not multiply - # the size of the issue. + # Scheduled runs check every published version because links can break in one + # version and work in another; the report job merges the results into one issue. MATRIX=$(npx tsx -e "import { allVersions } from './src/versions/lib/all-versions'; console.log(JSON.stringify({ include: Object.keys(allVersions).map((version) => ({ version, language: 'en' })) }))") echo "matrix=${MATRIX}" >> $GITHUB_OUTPUT fi @@ -79,14 +75,13 @@ jobs: fail-fast: false matrix: ${{ fromJson(needs.setup-matrix.outputs.matrix) }} env: - # Disable Elasticsearch for faster warmServer + # Link checking does not need search data, so leave Elasticsearch unconfigured. ELASTICSEARCH_URL: '' steps: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 - uses: ./.github/actions/node-npm-setup - # Clone translations if not English - name: Clone translations if: matrix.language != 'en' uses: ./.github/actions/clone-translations @@ -112,6 +107,8 @@ jobs: retention-days: 5 if-no-files-found: ignore + # The REST API agent_assignment field starts Copilot cloud agent sessions. + # Details: https://docs.github.com/en/copilot/how-tos/use-copilot-agents/cloud-agent/start-copilot-sessions#using-the-rest-api - name: Create Copilot redirect issue if: inputs.create_copilot_issue uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 @@ -169,8 +166,6 @@ jobs: body = body.slice(0, lastNewline > 0 ? lastNewline : truncatedLength) + artifactNote } - // Use the REST API with agent_assignment to properly trigger Copilot cloud agent. - // See: https://docs.github.com/en/copilot/how-tos/use-copilot-agents/cloud-agent/start-copilot-sessions#using-the-rest-api const issue = await github.request('POST /repos/{owner}/{repo}/issues', { owner: 'github', repo: 'docs-content', @@ -199,13 +194,12 @@ jobs: slack_token: ${{ secrets.SLACK_DOCS_BOT_TOKEN }} issue_url: ${{ steps.create-failure-issue.outputs.issue_url }} - # Create combined report after all matrix jobs complete create-report: if: always() && github.repository == 'github/docs-internal' needs: [setup-matrix, check-internal-links] runs-on: ubuntu-latest - # Serialize publishing so two overlapping runs can't both create a - # "rolling" issue, or write their results out of order. + # Serialize publishing so overlapping runs cannot create duplicate rolling issues + # or write results out of order. concurrency: group: broken-internal-links-report cancel-in-progress: false @@ -228,13 +222,11 @@ jobs: id: combine env: ACTION_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} - # A version with no broken links uploads no report, so the files on disk undercount - # what was checked. Pass the matrix so the report can say "broken in all versions" - # and mean it. + # Pass the full matrix because versions with no broken links upload no report. + # The report needs those versions to claim a link breaks in all versions. MATRIX: ${{ needs.setup-matrix.outputs.matrix }} run: | - # Merge the per-version JSON rather than concatenating the rendered Markdown. - # A link broken in every version is one problem, not one per version. + # Merge per-version JSON so a link broken in every version stays one problem. if ls reports/*.json 1> /dev/null 2>&1; then echo "has_reports=true" >> $GITHUB_OUTPUT VERSIONS=$(echo "$MATRIX" | jq -r '[.include[] | "\(.version) \(.language)"] | join(",")') @@ -252,8 +244,7 @@ jobs: if: steps.combine.outputs.has_reports == 'true' uses: actions/upload-artifact@bbbca2ddaa5d8feaa63e36b76fdaad77386f024f # v7.0.0 with: - # The issue body caps every long section, and the notes there point at - # "the report attached to the workflow run". Upload it so that is true. + # Upload the artifact because issue notes point readers there when sections are capped. name: combined-link-report path: combined-report.md retention-days: 5 @@ -274,18 +265,14 @@ jobs: const repo = 'docs-content' const label = 'broken link report' - // GitHub rejects issue bodies over 65536 characters with a 422. The - // internal report routinely exceeds that, so truncate and point at - // the run artifact for the full contents. + // GitHub rejects issue bodies over 65536 characters with a 422, so truncate and point at the artifact. const MAX_BODY_SIZE = 60000 const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${context.runId}` let body = fs.readFileSync('combined-report.md', 'utf8') if (body.length > MAX_BODY_SIZE) { const notice = `\n\n---\n\n*Report truncated. Download the full report from the [workflow run artifacts](${runUrl}).*` let cut = body.slice(0, MAX_BODY_SIZE - notice.length) - // Cut at a line boundary so the last thing a reader sees is not half - // a table row, and close any `
` the cut left open, since an - // unclosed one swallows everything after it. + // Cut at a full line and close unbalanced
tags because one open tag hides later content. cut = cut.slice(0, cut.lastIndexOf('\n')) const opened = (cut.match(/
/g) || []).length const closed = (cut.match(/<\/details>/g) || []).length @@ -294,9 +281,7 @@ jobs: core.warning(`Report exceeded ${MAX_BODY_SIZE} characters, so it was truncated.`) } - // Reuse a single rolling issue instead of opening a new one every - // week, which floods the first responders' board. Find the open - // report issues (newest first). + // Reuse one rolling issue so weekly failures do not flood the first responders' board; sort newest first. const open = await github.paginate(github.rest.issues.listForRepo, { owner, repo, @@ -320,8 +305,7 @@ jobs: return } - // Refresh the newest open report in place and close any older - // duplicates so exactly one canonical issue remains. + // Refresh the newest open report and close duplicates so one canonical issue remains. const [canonical, ...superseded] = reportIssues await github.rest.issues.update({ owner, @@ -332,8 +316,7 @@ jobs: }) core.info(`Updated rolling report issue: ${canonical.html_url}`) - // Attempt every duplicate even if one fails, so a single transient - // API error doesn't leave the rest open. + // Close duplicates independently so one transient API error cannot leave the rest open. const results = await Promise.allSettled( superseded.map(async (issue) => { await github.rest.issues.createComment({ @@ -374,8 +357,7 @@ jobs: const repo = 'docs-content' const label = 'broken link report' - // A clean run means the open report is stale. Leaving it open would - // keep fixed failures on the first responders' board. + // Close stale reports so fixed failures leave the first responders' board. const open = await github.paginate(github.rest.issues.listForRepo, { owner, repo, diff --git a/.github/workflows/link-check-on-pr.yml b/.github/workflows/link-check-on-pr.yml index 39fa97ade145..09f6d87d3452 100644 --- a/.github/workflows/link-check-on-pr.yml +++ b/.github/workflows/link-check-on-pr.yml @@ -1,12 +1,9 @@ name: 'Link Check: On PR' -# **What it does**: Checks internal links in changed content files. -# **Why we have it**: To catch broken links before they're merged. -# **Who does it impact**: Docs content. +# Catch broken internal links in changed content files before merge. on: workflow_dispatch: - # merge_group: pull_request: types: [opened, synchronize, reopened] @@ -15,7 +12,7 @@ permissions: pull-requests: write issues: write -# Cancel in-progress runs for the same PR +# Cancel older runs for the same ref to save runner time. concurrency: group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}' cancel-in-progress: true @@ -29,7 +26,7 @@ jobs: - name: Checkout uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: - # Fetch 2 commits so tj-actions/changed-files can diff without extra API calls + # Fetch 2 commits so tj-actions/changed-files can diff without extra API calls. fetch-depth: 2 - uses: ./.github/actions/node-npm-setup @@ -55,9 +52,8 @@ jobs: ACTION_RUN_URL: ${{ github.server_url }}/${{ github.repository }}/actions/runs/${{ github.run_id }} SHOULD_COMMENT: ${{ secrets.DOCS_BOT_PAT_BASE != '' }} FAIL_ON_FLAW: true - # Cross-page anchor checking is on, but non-blocking during rollout: broken - # anchors are reported in the PR comment without failing the build. Flip - # FAIL_ON_ANCHOR_FLAW to true once false-positive/perf rates look clean. + # Cross-page anchor failures appear in PR comments, but rollout keeps them + # from failing builds while false-positive and performance rates settle. CHECK_ANCHORS: true FAIL_ON_ANCHOR_FLAW: false ENABLED_LANGUAGES: en diff --git a/.github/workflows/lint-code.yml b/.github/workflows/lint-code.yml index 3600a02cbeb3..e12a70a11536 100644 --- a/.github/workflows/lint-code.yml +++ b/.github/workflows/lint-code.yml @@ -1,8 +1,6 @@ name: Lint code -# **What it does**: Lints our code to ensure the code matches the specified code style. -# **Why we have it**: We want some level of consistency to our code. -# **Who does it impact**: Docs engineering, open-source engineering contributors. +# Catch lint, formatting, and type-check failures before merge. on: workflow_dispatch: @@ -12,7 +10,7 @@ on: permissions: contents: read -# This allows a subsequently queued workflow run to interrupt previous runs +# Cancel older runs for the same ref to save runner time. concurrency: group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}' cancel-in-progress: true diff --git a/.github/workflows/lint-entire-content-data-markdown.yml b/.github/workflows/lint-entire-content-data-markdown.yml index 9817a99675fa..11abcb189cd7 100644 --- a/.github/workflows/lint-entire-content-data-markdown.yml +++ b/.github/workflows/lint-entire-content-data-markdown.yml @@ -1,13 +1,12 @@ name: 'Lint entire content and data markdown files' -# **What it does**: Lints our content markdown weekly to ensure the content matches the specified styleguide. If errors or warnings exist, it opens an issue for the Docs content team to review. -# **Why we have it**: Extra precaution to run linter on the entire content/data directories. -# **Who does it impact**: Docs content. +# Run the content linter across all content and data Markdown so recurring issues get +# reported to the Docs content team. on: workflow_dispatch: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' permissions: contents: read diff --git a/.github/workflows/local-dev.yml b/.github/workflows/local-dev.yml index 048f6f2629cc..7bde34d55989 100644 --- a/.github/workflows/local-dev.yml +++ b/.github/workflows/local-dev.yml @@ -1,8 +1,6 @@ name: Local development -# **What it does**: Basic smoke test to ensure local dev server starts and serves content -# **Why we have it**: Catch catastrophic "npm start is completely broken" scenarios -# **Who does it impact**: Engineers, Contributors. +# Smoke-test npm start so local development failures block pull requests. on: merge_group: @@ -31,11 +29,9 @@ jobs: - name: Start server and basic smoke test run: | - # Start server in background npm start > /tmp/stdout.log 2> /tmp/stderr.log & SERVER_PID=$! - # Wait for server to be ready and test homepage if curl --fail --retry-connrefused --retry 10 --retry-delay 2 http://localhost:4000/; then echo "✅ Local dev server started successfully and serves homepage" kill $SERVER_PID 2>/dev/null || true diff --git a/.github/workflows/merged-notification.yml b/.github/workflows/merged-notification.yml index c650e765ee94..1cb35a15791c 100644 --- a/.github/workflows/merged-notification.yml +++ b/.github/workflows/merged-notification.yml @@ -1,11 +1,9 @@ name: Merged notification -# **What it does**: When we merge an open-source pull request, we want to set expectations that deployment may take awhile. -# **Why we have it**: We deploy to production from docs-internal, not docs. -# **Who does it impact**: Open-source contributors. +# docs-internal deploys docs, so merged public PRs need a production timing notice. on: - # Needed in lieu of `pull_request` so that the notification comment is posted to a PR from a fork. + # pull_request_target lets this workflow comment on pull requests from forks. pull_request_target: types: - 'closed' diff --git a/.github/workflows/moda-allowed-ips.yml b/.github/workflows/moda-allowed-ips.yml index 98e736cf6265..a2a80d243e3b 100644 --- a/.github/workflows/moda-allowed-ips.yml +++ b/.github/workflows/moda-allowed-ips.yml @@ -1,12 +1,10 @@ name: Update Moda allowed IPs -# **What it does**: Make sure that the allowed IPs in Moda are up to date. -# **Why we have it**: The IP ranges from Fastly can change. -# **Who does it impact**: Docs engineering. +# Fastly IP ranges can change, so this workflow opens PRs when Moda allowed IPs drift. on: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' workflow_dispatch: permissions: diff --git a/.github/workflows/moda-ci.yaml b/.github/workflows/moda-ci.yaml index b168608203b8..c21b934fea03 100644 --- a/.github/workflows/moda-ci.yaml +++ b/.github/workflows/moda-ci.yaml @@ -1,6 +1,6 @@ name: docs-internal Moda CI -# More info on CI actions setup can be found here: +# Moda CI follows this CI Actions setup: # https://github.com/github/ops/blob/master/docs/playbooks/build-systems/moving-moda-apps-from-bp-to-actions.md on: @@ -14,9 +14,6 @@ on: permissions: {} jobs: - ########################## - # Generate Vault keys - ########################## set-vault-keys: permissions: {} runs-on: ubuntu-latest @@ -29,17 +26,13 @@ jobs: VAULT_KEYS: ${{ vars.VAULT_KEYS }} run: | if [ -z "$VAULT_KEYS" ]; then - # We want to add the DOCS_BOT_PAT_BASE to the list of keys - # so that builds fetch the secret from the docs-internal vault - # where --environment is "ci" + # Add DOCS_BOT_PAT_BASE so builds using --environment ci + # fetch it from the docs-internal Vault. echo "modified=DOCS_BOT_PAT_BASE" >> "$GITHUB_OUTPUT" else echo "modified=${VAULT_KEYS},DOCS_BOT_PAT_BASE" >> "$GITHUB_OUTPUT" fi - ############# - # Moda jobs - ############# moda-config-bundle: if: ${{ github.repository == 'github/docs-internal' }} name: ${{ matrix.ci_job.job }} @@ -63,9 +56,6 @@ jobs: dx-bot-token: ${{ secrets.INTERNAL_ACTIONS_DX_BOT_ACCOUNT_TOKEN }} datadog-api-key: ${{ secrets.DATADOG_API_KEY }} - ############# - # Docker Image jobs - ############# docker-image: if: ${{ github.repository == 'github/docs-internal' }} name: ${{ matrix.ci_job.job }} @@ -85,16 +75,13 @@ jobs: with: ci-formatted-job-name: ${{ matrix.ci_job.job }} vault-keys: ${{ needs.set-vault-keys.outputs.modified_vault_keys }} - # Passes 'DOCS_BOT_PAT_BASE' secret from Vault to docker as --secret id=DOCS_BOT_PAT_BASE,src= + # docker-build-env-secrets passes DOCS_BOT_PAT_BASE from Vault as a Docker build secret. attest: true docker-build-env-secrets: 'DOCS_BOT_PAT_BASE' secrets: dx-bot-token: ${{ secrets.INTERNAL_ACTIONS_DX_BOT_ACCOUNT_TOKEN }} datadog-api-key: ${{ secrets.DATADOG_API_KEY }} - ############# - # Docker Security jobs - ############# docker-security: if: ${{ github.repository == 'github/docs-internal' }} name: ${{ matrix.ci_job.job }} @@ -113,7 +100,7 @@ jobs: with: ci-formatted-job-name: ${{ matrix.ci_job.job }} vault-keys: ${{ needs.set-vault-keys.outputs.modified_vault_keys }} - # Passes 'DOCS_BOT_PAT_BASE' secret from Vault to docker as --secret id=DOCS_BOT_PAT_BASE,src= + # docker-build-env-secrets passes DOCS_BOT_PAT_BASE from Vault as a Docker build secret. docker-build-env-secrets: 'DOCS_BOT_PAT_BASE' secrets: dx-bot-token: ${{ secrets.INTERNAL_ACTIONS_DX_BOT_ACCOUNT_TOKEN }} diff --git a/.github/workflows/move-content.yml b/.github/workflows/move-content.yml index bfd9147230ab..0cf5ba6d4035 100644 --- a/.github/workflows/move-content.yml +++ b/.github/workflows/move-content.yml @@ -1,8 +1,6 @@ name: Move content script test -# **What it does**: Tests the `npm run move-content` script -# **Why we have it**: To be sure it continues to work as expected -# **Who does it impact**: Docs team. +# This workflow protects npm run move-content from regressions. on: pull_request: @@ -11,7 +9,7 @@ on: - src/content-render/scripts/test-move-content.ts - 'src/frame/lib/**/*.js' - .github/workflows/move-content.yml - # In case any of the dependencies affect the script + # Dependency changes can break the move-content script. - 'package*.json' - src/fixtures/fixtures/content/get-started/ - src/fixtures/fixtures/content/code-security/ @@ -31,9 +29,7 @@ jobs: - name: Set up a dummy git user run: | - # These must be set to something before running the move-content - # script because it depends on executing `git mv ...` - # and `git commit ...` + # move-content runs git mv and git commit, so Git needs an identity. git config --global user.name "docs-bot" git config --global user.email "77750099+docs-bot@users.noreply.github.com" @@ -48,8 +44,6 @@ jobs: npm run test-moved-content -- \ src/fixtures/fixtures/content/get-started/start-your-journey/hello-world.md \ src/fixtures/fixtures/content/get-started/start-your-journey/hello-wurld.md - - # TODO: Add tests that inspects the git log git log | head -n 100 - name: Move code-security/getting-started to code-security/got-started @@ -63,6 +57,4 @@ jobs: npm run test-moved-content -- \ src/fixtures/fixtures/content/code-security/getting-started \ src/fixtures/fixtures/content/code-security/got-started - - # TODO: Add tests that inspects the git log git log | head -n 100 diff --git a/.github/workflows/move-ready-to-merge-pr.yaml b/.github/workflows/move-ready-to-merge-pr.yaml index 97d03d17b14b..3384511dadf1 100644 --- a/.github/workflows/move-ready-to-merge-pr.yaml +++ b/.github/workflows/move-ready-to-merge-pr.yaml @@ -1,11 +1,10 @@ name: Move and unlabel ready to merge PRs -# **What it does**: When a PR in the open source repo is labeled "ready to merge," the "waiting for review" label is removed and the PR is moved to the "Triage" column. -# **Why we have it**: To help with managing our project boards. -# **Who does it impact**: Open source contributors, open-source maintainers. +# Public PRs labeled ready to merge leave waiting for review and move to Triage +# on the open source board. on: - # Needed in lieu of `pull_request` so that the a PR from a fork can trigger the project board and label automation. + # pull_request_target lets forked PRs trigger project-board and label automation. pull_request_target: types: - labeled diff --git a/.github/workflows/move-reopened-issues-to-triage.yaml b/.github/workflows/move-reopened-issues-to-triage.yaml index bc2bd0ae34fa..a55c8ad6eaf7 100644 --- a/.github/workflows/move-reopened-issues-to-triage.yaml +++ b/.github/workflows/move-reopened-issues-to-triage.yaml @@ -1,8 +1,6 @@ name: Move Reopened Issues to Triage -# **What it does**: Moves issues that are reopened from the Done column to the Triage column. -# **Why we have it**: To prevent having to do this manually. -# **Who does it impact**: Open-source. +# Reopened public issues move from Done to Triage without maintainers moving cards manually. on: issues: diff --git a/.github/workflows/needs-sme-stale-check.yaml b/.github/workflows/needs-sme-stale-check.yaml index 5de7e7046ac8..abbcea63e086 100644 --- a/.github/workflows/needs-sme-stale-check.yaml +++ b/.github/workflows/needs-sme-stale-check.yaml @@ -1,12 +1,10 @@ name: Stale check for issues or PRs with "needs SME" label -# **What it does**: Runs only in the OS repository to provide stale checks on issues/PRs that need SME(subject matter expert) review. -# **Why we have it**: In the open repo, we want we want to check on issues/PRs that are waiting on SME review. -# **Who does it impact**: Anyone working in the open repo. +# Public issues and PRs waiting on subject matter expert review get stale reminders without automatic closure. on: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' permissions: contents: read @@ -23,13 +21,13 @@ jobs: id: stale with: only-labels: needs SME - days-before-stale: 28 # adds stale label if no activity for 7 days - temporarily changed to 28 days as we work through the backlog + days-before-stale: 28 # SME backlog volume needs a slower reminder cadence. stale-issue-message: 'This is a gentle reminder for the Technical Content team that this issue is waiting for technical review by a subject matter expert (SME).' stale-issue-label: 'Waiting on SME review' - days-before-issue-close: -1 # never close + days-before-issue-close: -1 # Keep issues open. stale-pr-message: 'This is a gentle reminder for the Technical Content team that this PR is waiting for technical review by a subject matter expert.' stale-pr-label: 'Waiting on SME review' - days-before-pr-close: -1 # never close + days-before-pr-close: -1 # Keep PRs open. - name: Print outputs env: diff --git a/.github/workflows/needs-sme-workflow.yml b/.github/workflows/needs-sme-workflow.yml index f8e6fb7d9ba3..973f29a0d3b9 100644 --- a/.github/workflows/needs-sme-workflow.yml +++ b/.github/workflows/needs-sme-workflow.yml @@ -1,13 +1,11 @@ name: Comment on adding "needs SME" label -# **What it does**: Comment on issues and pull requests when a "needs SME" label is added. SME = subject matter expert. -# **Why we have it**: We want to manage our queue of issues and pull requests that need sme review. -# **Who does it impact**: Everyone that works on docs or docs-internal. +# Public issues and PRs labeled needs SME get an automated comment for subject matter expert review. on: issues: types: [labeled] - # Needed in lieu of `pull_request` so that PRs from a fork can be labeled. + # pull_request_target lets forked PRs receive the triage comment. pull_request_target: types: [labeled] diff --git a/.github/workflows/no-response.yaml b/.github/workflows/no-response.yaml index 2c43c14787e4..bffeb590ad4e 100644 --- a/.github/workflows/no-response.yaml +++ b/.github/workflows/no-response.yaml @@ -1,18 +1,14 @@ name: Stale check for no response from author -# **What it does**: Runs only in the OS repository to close issues that don't have enough information to be -# actionable. -# **Why we have it**: To remove the need for maintainers to remember to check -# back on issues periodically to see if contributors have -# responded. -# **Who does it impact**: Everyone that works in the docs repository. +# Public issues and PRs with no author response close after stale reminders +# so maintainers do not recheck them manually. on: issue_comment: types: [created] schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' permissions: contents: read @@ -22,11 +18,8 @@ permissions: jobs: noResponse: runs-on: ubuntu-latest - # Only run in the OS repository, and skip bot-authored events. On failure the - # create-workflow-failure-issue step below posts a comment (as a bot); that - # comment is itself an issue_comment event that would re-trigger this workflow. - # During a transient failure that loops, so guard against bot actors to keep - # one failure from producing a flood of runs and Slack alerts. + # Only run in the public repo, and skip automated comments. + # Bot-authored issue_comment events are workflow chatter, not author responses. if: >- github.repository == 'github/docs' && github.actor != 'docs-bot' && @@ -39,9 +32,8 @@ jobs: repo-token: ${{ secrets.GITHUB_TOKEN }} only-labels: 'more-information-needed' - # Define behavior for issues days-before-issue-stale: 21 - days-before-issue-close: 1 # close after 1 day if the issue is not updated + days-before-issue-close: 1 stale-issue-label: 'Waiting on contributor' close-issue-message: > This issue has been automatically closed because there has been no response @@ -52,9 +44,8 @@ jobs: importance of repro steps](https://www.lee-dohm.com/2015/01/04/writing-good-bug-reports/) for more information about the kind of information that may be helpful. - # Define behavior for pull requests days-before-pr-stale: 21 - days-before-pr-close: 1 # close after a day if no activity is detected + days-before-pr-close: 1 stale-pr-label: 'Waiting on contributor' close-pr-message: > This PR has been automatically closed because there has been no response to diff --git a/.github/workflows/notify-about-deployment.yml b/.github/workflows/notify-about-deployment.yml index 51edd5ae4f55..c025debd6f09 100644 --- a/.github/workflows/notify-about-deployment.yml +++ b/.github/workflows/notify-about-deployment.yml @@ -1,11 +1,7 @@ name: Notify about production deployment -# **What it does**: Posts a comment on every PR in the deploy that got into -# production. The merge queue can batch several PRs into one -# deploy, so it walks back from the deployed commit to find -# all of them. -# **Why we have it**: So that the PR author can be informed when their merged PR is in production. -# **Who does it impact**: Writers +# The merge queue can batch PRs into one deploy, so this workflow walks back +# from the deployed commit and tells every PR author when their change reaches production. on: workflow_dispatch: @@ -32,11 +28,8 @@ jobs: uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 - uses: ./.github/actions/node-npm-setup - # The "Purge Fastly" action takes about 6 minutes to purge all - # languages. First does the language agnostic URLs, then English, - # then all the other languages. - # So it takes about ~30 seconds until it has sent the purge for - # all English docs. + # The production purge targets changed English page keys. + # Give Fastly a short propagation window before finding built PRs. - name: Sleep a little to give Fastly Purge a chance run: sleep 30 diff --git a/.github/workflows/notify-release-pms.yml b/.github/workflows/notify-release-pms.yml index 4dc122f42e47..590b622fac3d 100644 --- a/.github/workflows/notify-release-pms.yml +++ b/.github/workflows/notify-release-pms.yml @@ -1,10 +1,7 @@ name: Notify release PMs -# **What it does**: Posts review notification comments on release issues -# in github/releases for generated GHES release notes. -# **Why we have it**: So comments are always posted by docs-bot, without -# needing to distribute a PAT to individual team members. -# **Who does it impact**: Docs content (GHES release DRIs). +# Docs-bot posts generated GHES release-note review comments on github/releases issues +# so PMs do not need a shared PAT. on: workflow_dispatch: diff --git a/.github/workflows/notify-when-maintainers-cannot-edit.yaml b/.github/workflows/notify-when-maintainers-cannot-edit.yaml index 69a2298edb6b..4fec4efb0c94 100644 --- a/.github/workflows/notify-when-maintainers-cannot-edit.yaml +++ b/.github/workflows/notify-when-maintainers-cannot-edit.yaml @@ -1,11 +1,10 @@ name: Notify When Maintainers Cannot Edit -# **What it does**: Notifies the author of a PR when their PR does not allow maintainers to edit it. -# **Why we have it**: To prevent having to do this manually. -# **Who does it impact**: Open-source. +# Public PR authors get a maintainer-edit notice without maintainers checking +# the fork setting manually. on: - # Needed in lieu of `pull_request` so that PRs from a fork can be notified. + # pull_request_target lets forked PRs receive the maintainer-edit notice. pull_request_target: types: - opened diff --git a/.github/workflows/orphaned-features-check.yml b/.github/workflows/orphaned-features-check.yml index e9a00ae73602..0fa5fb4604f9 100644 --- a/.github/workflows/orphaned-features-check.yml +++ b/.github/workflows/orphaned-features-check.yml @@ -1,17 +1,16 @@ name: 'Orphaned features check' -# **What it does**: Finds any data/features that are no longer used in the repo. -# **Why we have it**: To avoid orphans into the repo. -# **Who does it impact**: Docs content. +# This workflow opens deletion PRs for unused data/features entries so stale entries +# do not stay in the repo. on: workflow_dispatch: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' pull_request: paths: - .github/workflows/orphaned-features-check.yml - # In case any of the dependencies affect the script + # Dependency changes can break the orphaned-features script. - 'package*.json' - 'src/data-directory/scripts/find-orphaned-features/**' - .github/actions/clone-translations/action.yml @@ -28,13 +27,10 @@ jobs: - name: Checkout English repo uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: - # Using a PAT is necessary so that the new commit will trigger the - # CI in the PR. (Events from GITHUB_TOKEN don't trigger new workflows.) + # A PAT triggers CI on the PR; GITHUB_TOKEN events do not start workflows. token: ${{ secrets.DOCS_BOT_PAT_BASE }} - # It's important because translations are often a bit behind. - # So if a translation is a bit behind, it might still be referencing - # a feature even though none of the English content does. + # Translations can lag English, so clone them before deciding that a feature is unused. - name: Clone all translations uses: ./.github/actions/clone-translations with: @@ -44,7 +40,7 @@ jobs: - name: Check for orphaned features env: - # Needed for gh + # The gh CLI reads GITHUB_TOKEN. GITHUB_TOKEN: ${{ secrets.DOCS_BOT_PAT_BASE }} DRY_RUN: ${{ github.event_name == 'pull_request'}} run: | @@ -64,14 +60,12 @@ jobs: git status - # When run on a pull_request, we're just testing the tooling. - # Exit before it actually pushes the possible changes. + # Pull request runs exercise the tooling without pushing deletion PRs. if [ "$DRY_RUN" = "true" ]; then echo "Dry-run mode when run in a pull request" exit 0 fi - # Replicated from the translation pipeline PR-maker Action git config --global user.name "docs-bot" git config --global user.email "77750099+docs-bot@users.noreply.github.com" diff --git a/.github/workflows/orphaned-files-check.yml b/.github/workflows/orphaned-files-check.yml index 9ba23f362176..3dc7fdbfdfbe 100644 --- a/.github/workflows/orphaned-files-check.yml +++ b/.github/workflows/orphaned-files-check.yml @@ -1,18 +1,17 @@ name: 'Orphaned files check' -# **What it does**: Checks that there are no files in ./assets/, ./data/reusables, or ./data/tables that aren't mentioned in any source file. -# **Why we have it**: To avoid orphans into the repo. -# **Who does it impact**: Docs content. +# This workflow opens deletion PRs for unused assets, reusables, and tables +# so orphaned files leave the repo. on: workflow_dispatch: schedule: - - cron: '20 16 * * 1' # Run every Monday at 16:20 UTC / 8:20 PST + - cron: '20 16 * * 1' pull_request: paths: - .github/workflows/orphaned-assets-check.yml - .github/workflows/orphaned-files-check.yml - # In case any of the dependencies affect the script + # Dependency changes can break the orphaned-files scripts. - 'package*.json' - src/assets/scripts/find-orphaned-assets.ts - src/content-render/scripts/reusables-cli/find/unused.ts @@ -33,13 +32,10 @@ jobs: - name: Checkout English repo uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: - # Using a PAT is necessary so that the new commit will trigger the - # CI in the PR. (Events from GITHUB_TOKEN don't trigger new workflows.) + # A PAT triggers CI on the PR; GITHUB_TOKEN events do not start workflows. token: ${{ secrets.DOCS_BOT_PAT_BASE }} - # It's important because translations are often a bit behind. - # So if a translation is a bit behind, it might still be referencing - # an asset even though none of the English content does. + # Translations can lag English, so clone them before deciding that a file is unused. - name: Clone all translations uses: ./.github/actions/clone-translations with: @@ -49,14 +45,13 @@ jobs: - name: Check for orphaned assets and reusables env: - # Needed for gh + # The gh CLI reads GITHUB_TOKEN. GITHUB_TOKEN: ${{ secrets.DOCS_BOT_PAT_BASE }} DRY_RUN: ${{ github.event_name == 'pull_request'}} run: | set -e - # The `-s` is to make npm run silent and not print verbose - # information about the npm script alias. + # The -s flag keeps npm script-alias noise out of the file lists. assetFilesToRemove=$(npm run -s find-orphaned-assets) reusableFilesToRemove=$(npm run -s reusables -- find unused | grep '^data/reusables' || true) tableFilesToRemove=$(npm run -s find-orphaned-tables) @@ -74,17 +69,14 @@ jobs: git status - # If nothing to commit, exit now. It's fine. No orphans. git status -- ':!translations*' | grep 'nothing to commit' && exit 0 - # When run on a pull_request, we're just testing the tooling. - # Exit before it actually pushes the possible changes. + # Pull request runs exercise the tooling without pushing deletion PRs. if [ "$DRY_RUN" = "true" ]; then echo "Dry-run mode when run in a pull request" exit 0 fi - # Replicated from the translation pipeline PR-maker Action git config --global user.name "docs-bot" git config --global user.email "77750099+docs-bot@users.noreply.github.com" diff --git a/.github/workflows/os-ready-for-review.yml b/.github/workflows/os-ready-for-review.yml index 3c3cd37aff82..47c0ebf39597 100644 --- a/.github/workflows/os-ready-for-review.yml +++ b/.github/workflows/os-ready-for-review.yml @@ -1,10 +1,9 @@ name: OS Ready for review -# **What it does**: Adds pull requests and issues in the docs repository to the docs-content review board when the "waiting for review" label is added -# **Why we have it**: So that contributors in the OS repo can easily get reviews from the docs-content team, and so that writers can see when a PR is ready for review -# **Who does it impact**: Writers working in the docs repository +# Public issues and PRs labeled waiting for review go to the docs-content review board +# when a docs team member labels them. on: - # Needed in lieu of `pull_request` so that PRs from a fork can be triaged to the proper project board. + # pull_request_target lets forked PRs reach project-board triage. pull_request_target: types: [labeled] issues: @@ -30,8 +29,8 @@ jobs: result-encoding: string script: | const triggerer_login = context.payload.sender.login - // Team is addressed by numeric ID (org github = 9919, team docs = 325922) - // because IDs survive team renames and slugs do not. + // Numeric org and team IDs survive team renames; org github is 9919. + // Team technical-content is 325922. const teamMembers = await github.request( `/organizations/9919/team/325922/members?per_page=100` ) diff --git a/.github/workflows/package-lock-lint.yml b/.github/workflows/package-lock-lint.yml index 502e5d6ac05f..8cf827c7cfa2 100644 --- a/.github/workflows/package-lock-lint.yml +++ b/.github/workflows/package-lock-lint.yml @@ -1,8 +1,6 @@ name: Package lock lint -# **What it does**: Makes sure package.json and package-lock.json is in sync -# **Why we have it**: Accidental manual edits of the dependencies directly in package.json -# **Who does it impact**: Docs engineering/writers/contributors. +# This workflow catches manual package.json edits that leave package-lock.json out of sync. on: pull_request: @@ -14,7 +12,7 @@ on: permissions: contents: read -# This allows a subsequently queued workflow run to interrupt previous runs +# Cancel older runs for the same PR because this check only depends on the latest commit. concurrency: group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }}' cancel-in-progress: true @@ -37,23 +35,17 @@ jobs: run: | npm --version - # Save the current top-level dependencies from package-lock.json node -e "console.log(JSON.stringify(require('./package-lock.json').packages['']))" > /tmp/before.json - # From https://docs.npmjs.com/cli/v7/commands/npm-install - # - # The --package-lock-only argument will only update the - # package-lock.json, instead of checking node_modules and - # downloading dependencies. - # + # npm install --package-lock-only updates package-lock.json + # without checking node_modules or downloading packages. + # See https://docs.npmjs.com/cli/v7/commands/npm-install. npm install --package-lock-only --ignore-scripts --include=optional - # Extract the top-level dependencies after regeneration node -e "console.log(JSON.stringify(require('./package-lock.json').packages['']))" > /tmp/after.json - # Compare only the top-level package dependencies - # This ignores platform-specific differences in nested dependency resolution - # (like "peer" flags) that don't affect actual installed versions + # Compare only top-level package dependencies because platform-specific nested dependency + # metadata, such as peer flags, does not affect actual installed versions. if ! diff /tmp/before.json /tmp/after.json; then echo "ERROR: Top-level dependencies in package-lock.json are out of sync with package.json" echo "Please run 'npm install' locally and commit the updated package-lock.json" diff --git a/.github/workflows/purge-fastly.yml b/.github/workflows/purge-fastly.yml index 4949dd1775a0..a911270c2168 100644 --- a/.github/workflows/purge-fastly.yml +++ b/.github/workflows/purge-fastly.yml @@ -1,12 +1,7 @@ name: Purge Fastly -# **What it does**: -# On production deploy, hard-purge the changed English content pages by key. -# On demand, soft or hard purge language keys, a single key, or entire cache. -# **Why we have it**: So a just-deployed change is visible right away, and so -# docs engineering can clear a bad cache state without the Fastly UI. -# **Who does it impact**: Writers and engineers. A full purge impacts all readers -# and spikes origin traffic while the cache refills, so it's gated below. +# Production deploys hard-purge changed English content, and manual runs clear bad Fastly cache state. +# Full-cache purges affect all readers and spike origin traffic, so they stay gated below. on: deployment_status: @@ -30,9 +25,8 @@ permissions: contents: read deployments: read -# Serialize full-cache purges so two can't overlap and leave the cache in an -# unknown state. Every other run (per-deploy, per-language) gets a unique group -# so those never block each other. +# Serialize full-cache purges so concurrent runs cannot leave the cache in an unknown state. +# Per-deploy and per-language runs get unique groups so they never block each other. concurrency: group: ${{ (inputs.everything == 'purge everything' && 'purge-fastly-all') || format('purge-fastly-{0}', github.run_id) }} cancel-in-progress: false @@ -43,11 +37,8 @@ env: jobs: send-purges: - # Run when workflow_dispatch is the event - # or when deployment_status is the event and it's a successful production deploy. - # NOTE: This workflow triggers on all deployment_status events, - # including staging, but only runs for production. - # Non-production deploys will show as "skipped" - this is expected behavior. + # This workflow sees every deployment_status event, but only production successes purge. + # Non-production deployments skip this job. if: >- ${{ github.repository == 'github/docs-internal' && @@ -62,11 +53,8 @@ jobs: - uses: ./.github/actions/node-npm-setup - name: Validate confirmation input - # A full-cache purge only triggers on the exact string "purge everything". - # Any other non-empty value (e.g. a typo) would otherwise be silently - # ignored and fall through to a normal soft purge that finishes green, so - # an operator could think they evicted the whole cache when they didn't. - # Fail loudly instead. + # A full-cache purge requires the exact string "purge everything"; typos must fail + # instead of falling through to a green soft purge. env: EVERYTHING_INPUT: ${{ inputs.everything }} run: | @@ -80,9 +68,8 @@ jobs: run: npm run wait-for-build - name: Purge Fastly (manual) - # Raw inputs are passed through the environment and quoted, never spliced - # into the command string, so a value like `en' --everything` can't break - # out of its argument and inject another flag. + # Raw inputs pass through the environment as quoted arguments. + # A value like en' --everything cannot break out and inject another flag. if: ${{ github.event_name == 'workflow_dispatch' }} env: LANGUAGES_INPUT: ${{ inputs.languages }} @@ -102,8 +89,7 @@ jobs: npm run purge-fastly -- "${args[@]}" - name: Hard-purge changed English content pages - # On prod deploys, evict the surrogate keys of the English content pages - # whose content/ files changed in this deploy. + # Production deploys purge surrogate keys for English pages whose content/ files changed. if: ${{ github.event_name == 'deployment_status' }} env: GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }} diff --git a/.github/workflows/ready-for-doc-review.yml b/.github/workflows/ready-for-doc-review.yml index ad3cdf537114..552ecc7fade4 100644 --- a/.github/workflows/ready-for-doc-review.yml +++ b/.github/workflows/ready-for-doc-review.yml @@ -1,8 +1,7 @@ name: Ready for docs-content review -# **What it does**: Adds pull requests in the docs-internal repository to the docs-content review board when the "ready-for-doc-review" label is added or when a review by docs-content or docs-reviewers is requested. This workflow is also called as a reusable workflow from other repos including docs-content, docs-strategy, docs-early-access, and github. -# **Why we have it**: So that other GitHub teams can easily request reviews from the docs-content team, and so that writers can see when a PR is ready for review -# **Who does it impact**: Writers who need to review docs-related PRs +# GitHub repos other than docs can add PRs to the docs-content review board by label, +# team review request, or reusable workflow, so writers can find work ready for review. on: pull_request: diff --git a/.github/workflows/remove-fr-label-remove-from-fr-v2.yml b/.github/workflows/remove-fr-label-remove-from-fr-v2.yml index 76a0b9bed897..b802a753e6fe 100644 --- a/.github/workflows/remove-fr-label-remove-from-fr-v2.yml +++ b/.github/workflows/remove-fr-label-remove-from-fr-v2.yml @@ -1,8 +1,7 @@ name: Remove PRs from FR project v2 when FR label is removed -# **What it does**: When the `docs-content-fr` label is removed from a pull request, this workflow removes the PR from the FR project v2 project. -# **Why we have it**: Reduce busy work for the first responder. -# **Who does it impact**: docs-content first responder. +# Removing docs-content-fr archives the PR from the FR v2 project +# so first responders do not do it manually. on: pull_request: diff --git a/.github/workflows/repo-sync.yml b/.github/workflows/repo-sync.yml index 4b4e76de0dd3..8057b2770d3b 100644 --- a/.github/workflows/repo-sync.yml +++ b/.github/workflows/repo-sync.yml @@ -1,16 +1,12 @@ name: Repo Sync -# **What it does**: GitHub Docs has two repositories: github/docs (public) and github/docs-internal (private). -# This GitHub Actions workflow keeps the `main` branch of those two repos in sync. -# **Why we have it**: To keep the open-source repository up-to-date -# while still having an internal repository for sensitive work. -# **Who does it impact**: Open-source. -# For more details, see https://github.com/repo-sync/repo-sync#how-it-works +# Syncs main between github/docs and github/docs-internal so public docs stay current +# and sensitive work stays private. Mechanics: https://github.com/repo-sync/repo-sync#how-it-works on: workflow_dispatch: schedule: - - cron: '20 14-23/3 * * 1-5' # Mon-Fri 6:20a, 9:20a, 12:20p, 3:20p PST + - cron: '20 14-23/3 * * 1-5' permissions: contents: write @@ -69,7 +65,7 @@ jobs: pull_number: prNumber, state: 'closed' }) - // Error loud here, so no try/catch + // Let close failures fail the workflow. console.log('Closed pull request', prNumber) } @@ -120,8 +116,7 @@ jobs: pull_number = pull.number console.log('Created pull request successfully', pull.html_url) } catch (err) { - // Don't error/alert if there's no commits to sync - // Don't throw if > 100 pulls with same head_sha issue + // Don't alert on "No commits" or on the "same head_sha" error from over 100 PRs with one head. if (err.message?.includes('No commits') || err.message?.includes('same head_sha')) { console.log(err.message) return @@ -139,7 +134,7 @@ jobs: console.log('Locked the pull request to prevent spam') } catch (error) { console.error('Failed to lock the pull request.', error) - // Don't fail the workflow + // Lock failures must not fail the sync. } console.log('Counting files changed') @@ -161,10 +156,9 @@ jobs: console.log('No detected merge conflicts') console.log('Merging the pull request') - // Admin merge pull request to avoid squash - // Retry once per minute for up to 15 minutes to wait for required checks (e.g. CodeQL) + // merge_method: merge keeps merge commits; retry each minute for up to 15 minutes for checks. const maxAttempts = 15 - const delay = 60_000 // 1 minute + const delay = 60_000 for (let attempt = 1; attempt <= maxAttempts; attempt++) { try { await github.rest.pulls.merge({ diff --git a/.github/workflows/restrict-merge-queue.yml b/.github/workflows/restrict-merge-queue.yml index bbaec5ac9f21..02bb0d8cb53f 100644 --- a/.github/workflows/restrict-merge-queue.yml +++ b/.github/workflows/restrict-merge-queue.yml @@ -1,30 +1,10 @@ name: Restrict who can queue merges -# **What it does**: -# On a `merge_group` event, checks whether the person who put the pull -# request into the merge queue is on github/technical-content. If they -# are not, comments on the pull request saying so and fails, which -# ejects the entry from the queue. -# **Why we have it**: -# Classic branch protection used to restrict who could push to `main`, -# but that rule was swept org-wide on 2026-06-22, so today anyone with -# write access can merge. Rebuilding it means also enabling a merge -# queue on the same rule, which could collide with the merge queue on -# our ruleset and lock the branch for everyone. This does the same job -# with machinery we own outright. -# **Who does it impact**: Anyone merging to `main`. - -# Two things to know before changing this: -# -# 1. `merge-queue-restriction` has to be a required status check on the ruleset targeting `refs/heads/main`, or this -# enforces nothing. Add it there only after this workflow is on `main` and reporting. The other order makes the -# check required before it has ever reported, which blocks every pull request. To turn enforcement off again, -# remove it from the ruleset. This workflow keeps running and keeps passing. -# -# 2. The `pull_request` runs do no work. They exist so the required check reports a passing context on the pull -# request itself. Drop them and the check sits pending forever and nothing can ever be enqueued. +# Checks merge-queue enqueuers against github/technical-content and ejects failures. +# This avoids a branch-protection rule with its own merge queue, which could lock the branch. on: + # pull_request runs do no work, but without their passing context PRs can never enter the queue. pull_request: types: [opened, reopened, synchronize, ready_for_review] merge_group: @@ -33,17 +13,20 @@ permissions: contents: read pull-requests: write -# Keyed on head SHA rather than pull request number. Webhook delivery order is not guaranteed, and keying on the pull -# request would let a late event for an old SHA cancel the run for a newer one. +# Key concurrency on head SHA because webhook delivery order can lag. +# A late event for an old SHA must not cancel the newer run. concurrency: group: ${{ github.workflow }}-${{ github.event.pull_request.head.sha || github.event.merge_group.head_sha }} cancel-in-progress: true jobs: - # The job id doubles as the check run name because this job deliberately has no `name:` key. Renaming this job renames - # the required status check, which silently stops enforcing anything. + # This check enforces nothing unless the refs/heads/main ruleset requires it. + # Make it required only after it reports on main, or PRs block on a check that never ran. + # To disable enforcement, remove the required check from the ruleset. + # This job omits name so its id becomes the required check name. + # Renaming it leaves the ruleset waiting for the old check, so PRs stay blocked. merge-queue-restriction: - # This repository syncs a subset of files to the public github/docs, including workflows. Nothing here applies there. + # Some workflows sync to public github/docs; this restriction only applies in docs-internal. if: github.repository == 'github/docs-internal' runs-on: ubuntu-latest steps: @@ -51,12 +34,12 @@ jobs: if: github.event_name == 'merge_group' uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0 with: - # Reading org team membership needs `read:org`, which GITHUB_TOKEN does not have. `merge_group` always runs in - # the base repository, so this secret is always available here, including for pull requests from forks. + # Reading team membership needs read:org, which GITHUB_TOKEN lacks. + # merge_group runs in the base repo, so this secret is available for forked PRs. github-token: ${{ secrets.DOCS_BOT_PAT_BASE }} script: | - // Addressed by numeric ID (org github = 9919, team technical-content = 325922) because IDs survive renames - // and slugs do not. This team was called `docs` until recently and the rename broke a pile of automation. + // Numeric org and team IDs survive team renames; org github is 9919. + // Team technical-content is 325922. const ORG_ID = 9919 const TEAM_ID = 325922 const TEAM = 'github/technical-content' @@ -65,13 +48,14 @@ jobs: const CONTENT_SLACK_CHANNEL = 'C0E9DK082' const MAX_ATTEMPTS = 3 - // `github.actor` is the person who enqueued. On a re-run it stays the original actor, unlike - // `github.triggering_actor`, so re-running cannot launder a failing check into a passing one. + // github.actor stays the original enqueuer on re-runs. + // github.triggering_actor changes to the rerunner. + // Re-runs cannot turn a failing check into a passing one. const actor = context.actor core.info(`This merge group was queued by @${actor}.`) - // A GitHub App actor always ends in `[bot]`, and `[` is not a valid character in a username, so nobody can - // impersonate one. `docs-bot` is a plain User account and has to be named explicitly. + // GitHub App actors always end in [bot], and [ cannot appear in a username. + // docs-bot is a User account, so name it explicitly. core.info(`Checking whether @${actor} is an automation account...`) if (actor.endsWith('[bot]') || EXEMPT_USERS.includes(actor)) { core.info(`Checked: @${actor} is an automation account. Allowing the merge.`) @@ -79,10 +63,10 @@ jobs: } core.info(`Checked: @${actor} is a person, so they need to be on the team.`) - // Every request retries transient failures before giving up, then fails closed. Failing closed is safe - // here: github/technical-content is an `always` bypass actor on the ruleset, so a broken check stops - // non-Docs merges but never stops Docs. A 404 comes back as null data rather than as an error, because on - // both of the endpoints below it is an answer rather than a failure. + // Retry transient failures before failing closed. + // github/technical-content bypasses the ruleset, + // so a broken check blocks non-Docs merges only. + // A 404 returns null data instead of an error because both endpoints use it as a valid answer. async function ask(description, route, params) { for (let attempt = 1; attempt <= MAX_ATTEMPTS; attempt++) { core.info(`${description} (attempt ${attempt} of ${MAX_ATTEMPTS})...`) @@ -100,15 +84,16 @@ jobs: } } - // Deliberately says nothing on the pull request. We only comment when we know the answer, and here we - // do not. + // Only comment on the pull request when membership is known. function giveUp(detail) { core.setFailed(`${detail} Failing closed, so this stays out of the queue. Ask in #docs-content.`) } - // Checking the parent team is enough. Every member of every child team (docs-content, docs-engineering, - // docs-localization, docs-content-systems, docs-product-managers, docs-open-source, docs-design, - // docs-content-design, copilot-docs) also resolves as a member of the parent. + // Checking the parent team covers every child team: docs-content, + // docs-engineering, docs-localization, docs-content-systems, + // docs-product-managers, docs-open-source, docs-design, + // docs-content-design, and copilot-docs. + // GitHub resolves child-team members as members of the parent. const membershipResult = await ask( `Asking the API whether @${actor} is on ${TEAM}`, 'GET /organizations/{org_id}/team/{team_id}/memberships/{username}', @@ -130,9 +115,9 @@ jobs: } else { core.info(`Asked: the API reports no membership for @${actor} on ${TEAM} (HTTP 404).`) - // That 404 is ambiguous. It is byte for byte the same response for "not a member", "team no longer - // exists", and "the token lost visibility into the org". Read the team back before believing it, - // otherwise a deleted team or a downgraded token would blame every single person who tries to merge. + // The 404 response is identical for a non-member, deleted team, + // and token without org visibility. + // Read the team before blocking the merge, or infrastructure failures would blame every enqueuer. const teamResult = await ask( `A 404 is ambiguous, so reading ${TEAM} back to confirm it is still visible`, 'GET /organizations/{org_id}/team/{team_id}', @@ -161,9 +146,7 @@ jobs: : `@${actor} is not a member of ${TEAM}.` core.info(`Blocking the merge. ${reason}`) - // A failed check on a `gh-readonly-queue` ref is not something anyone goes looking for, so say why on the - // pull request itself. The merge group ref is `refs/heads/gh-readonly-queue//pr--`, and - // the pull request it names is the one this actor just enqueued, so the number and the actor correspond. + // Queue ref refs/heads/gh-readonly-queue//pr-- names this actor's PR. const ref = context.payload.merge_group?.head_ref ?? context.ref core.info(`Working out which pull request this merge group is for, from "${ref}"...`) const number = Number(ref.match(/\/pr-(\d+)-[0-9a-f]+$/)?.[1]) @@ -173,8 +156,8 @@ jobs: } else { core.info(`Worked it out: this merge group is for #${number}.`) - // Only comment once. Someone who tries to enqueue again already has the explanation, and repeating it - // turns a useful comment into noise. + // Comment once; repeated attempts already have the explanation, + // and repeating it adds noise. core.info(`Reading the existing comments on #${number}...`) const comments = await github.paginate(github.rest.issues.listComments, { owner: context.repo.owner, @@ -193,9 +176,9 @@ jobs: repo: context.repo.repo, issue_number: number, body: [ - // The marker has to be on its own line. GitHub parses ``, so anything sharing that - // line renders as literal text: no code spans, no links. + // Put the marker on its own line. + // GitHub treats , so shared-line content renders as literal text. MARKER, [ `👋 Hi @${actor}, this pull request was removed from the merge queue.`, diff --git a/.github/workflows/review-comment.yml b/.github/workflows/review-comment.yml index 5c6542dea0b9..29cf95c58e2b 100644 --- a/.github/workflows/review-comment.yml +++ b/.github/workflows/review-comment.yml @@ -1,18 +1,17 @@ name: Review comment -# **What it does**: When a PR is opened in docs-internal or docs containing code, it comments with instructions on how to deploy and review the changes. it adds the staging review and live article links in a Content Directory Changes table in a comment. -# **Why we have it**: To help Docs contributors understand how to review their changes. -# **Who does it impact**: docs-internal and docs maintainers and contributors +# Eligible PRs get review guidance, and internal PRs also get staging guidance. +# Content changes also get the review-links table. on: - # Required in lieu of `pull_request` so that the comment can be posted to PRs opened from a fork. + # pull_request_target lets forked PRs receive the review comment. pull_request_target: types: - opened - synchronize paths-ignore: - '.github/workflows/review-comment.yml' - # For reviewing changes to this workflow + # pull_request runs only when this workflow changes, so edits can be tested safely. pull_request: types: - opened @@ -24,7 +23,7 @@ permissions: contents: read pull-requests: write -# This allows a subsequently queued workflow run to interrupt previous runs +# Cancel older runs for the same PR because the comment must match the latest commit. concurrency: group: '${{ github.workflow }} @ ${{ github.event.pull_request.head.label || github.head_ref || github.ref }} x ${{ github.event_name }}' cancel-in-progress: true diff --git a/.github/workflows/reviewers-content-systems.yml b/.github/workflows/reviewers-content-systems.yml index 47d773b43482..4f80de61eb9c 100644 --- a/.github/workflows/reviewers-content-systems.yml +++ b/.github/workflows/reviewers-content-systems.yml @@ -1,8 +1,7 @@ name: Reviewers - Content Systems -# **What it does**: Automatically add reviewers based on paths, but only for the docs-internal repo. -# **Why we have it**: So we can have reviewers automatically without getting open source notifications. -# **Who does it impact**: Docs team. +# Internal content-systems paths request docs-content-systems review. +# Restricting to docs-internal avoids public-repo reviewer notifications. on: pull_request: diff --git a/.github/workflows/reviewers-dependabot.yml b/.github/workflows/reviewers-dependabot.yml index 56ec3af50009..dc3b615c89a6 100644 --- a/.github/workflows/reviewers-dependabot.yml +++ b/.github/workflows/reviewers-dependabot.yml @@ -1,8 +1,7 @@ name: Reviewers - Dependabot -# **What it does**: Automatically add reviewers based on paths, for docs-internal and docs repos. -# **Why we have it**: So dependabot maintainers can be notified about relevant pull requests. -# **Who does it impact**: dependabot-updates-reviewers. +# Dependabot docs paths request dependabot-updates-reviewers +# so maintainers see relevant PRs. on: pull_request: diff --git a/.github/workflows/reviewers-docs-engineering.yml b/.github/workflows/reviewers-docs-engineering.yml index 9b3b92a6e780..1bd5ab675ba3 100644 --- a/.github/workflows/reviewers-docs-engineering.yml +++ b/.github/workflows/reviewers-docs-engineering.yml @@ -16,9 +16,9 @@ on: - '**.tsx' - '**.scss' - 'src/**' - - '!src/**.json' # Docs Engineering does not triage automated pipeline data PRs. - - '!src/**.yml' # Docs Engineering does not triage automated pipeline data PRs. - - '!src/**.sha' # Docs Engineering does not triage automated pipeline data PRs. + - '!src/**.json' # Skip automated pipeline data PRs. + - '!src/**.yml' # Skip automated pipeline data PRs. + - '!src/**.sha' # Skip automated pipeline data PRs. - '.github/**' - 'config/**' - '.devcontainer/**' @@ -48,15 +48,13 @@ jobs: - name: Checkout repository uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 - # Detect PRs that only changed package-lock.json (no engineering source files). - # These are usually cross-platform `npm install` churn from contributors - # editing content. We comment with reset instructions instead of adding - # them to the engineering review board. - # - # Dependabot is exempt. Its security updates for transitive dependencies - # change only the lockfile, because the dependency is not in package.json. - # Those PRs are intentional, so the reset instructions are wrong and - # keeping them off the board leaves them without engineering triage. + # Detect PRs that only change package-lock.json. Contributors often create + # cross-platform npm install churn while editing content. + # Comment with reset instructions instead of adding them to the engineering review board. + # Dependabot security updates can change only the lockfile when package.json + # does not name the transitive dependency. + # Those PRs are intentional, so reset instructions would be wrong and + # keeping them off the board would leave them without engineering triage. - name: Detect lockfile-only churn id: detect env: diff --git a/.github/workflows/reviewers-legal.yml b/.github/workflows/reviewers-legal.yml index bef7f702d45c..ff5d93a822df 100644 --- a/.github/workflows/reviewers-legal.yml +++ b/.github/workflows/reviewers-legal.yml @@ -1,8 +1,7 @@ name: Reviewers - Legal -# **What it does**: Enforces reviews of Responsible AI (RAI) content by the GitHub legal team. Because RAI content can live anywhere in the content directory, it becomes a maintenance problem to use CODEOWNERS to enforce review on each article. -# **Why we have it**: RAI content must be reviewed by the GitHub legal team. -# **Who does it impact**: Content writers and the GitHub legal team. +# Responsible AI content can live anywhere under content, so CODEOWNERS path matching cannot select it. +# Matching PRs must request GitHub legal review. on: workflow_dispatch: @@ -34,7 +33,7 @@ jobs: - name: Checkout repository uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0 with: - # Fetch 2 commits so tj-actions/changed-files can diff without extra API calls + # tj-actions/changed-files needs two commits to diff without extra API calls. fetch-depth: 2 - name: Get changed files diff --git a/Dockerfile b/Dockerfile index 1d49692f2726..94766ce91d10 100644 --- a/Dockerfile +++ b/Dockerfile @@ -30,8 +30,8 @@ RUN --mount=type=secret,id=apt-auth-conf,target=/etc/apt/auth.conf.d/apt_auth.co && apt-get install -y nodejs \ && node --version -# Create the node user and home directory -ARG APP_HOME="/home/node/app" # Define in base so all child stages inherit it +# Stages built FROM base inherit this ARG, so every later stage can use APP_HOME. +ARG APP_HOME="/home/node/app" RUN useradd -ms /bin/bash node \ && mkdir -p $APP_HOME && chown -R node:node $APP_HOME diff --git a/content/actions/reference/workflows-and-actions/workflow-syntax.md b/content/actions/reference/workflows-and-actions/workflow-syntax.md index 5f916e52265e..0953709ba987 100644 --- a/content/actions/reference/workflows-and-actions/workflow-syntax.md +++ b/content/actions/reference/workflows-and-actions/workflow-syntax.md @@ -313,6 +313,10 @@ A boolean specifying whether the secret must be supplied. {% data reusables.actions.workflows.section-specifying-branches %} +## `on.workflow_run.workflows` + +{% data reusables.actions.workflows.section-specifying-workflows %} + ## `on.workflow_dispatch` {% data reusables.actions.workflow-dispatch %} @@ -1639,7 +1643,7 @@ Allowed expression contexts: `github`, `needs`, and `secrets`. ## Filter pattern cheat sheet -You can use special characters in path, branch, and tag filters. +You can use special characters in path, branch, tag, and workflow name filters. * `*`: Matches zero or more characters, but does not match the `/` character. For example, `Octo*` matches `Octocat`. * `**`: Matches zero or more of any character. diff --git a/content/admin/concepts/enterprise-best-practices/organize-work.md b/content/admin/concepts/enterprise-best-practices/organize-work.md index c7625310d5cf..0b30cf98c7f4 100644 --- a/content/admin/concepts/enterprise-best-practices/organize-work.md +++ b/content/admin/concepts/enterprise-best-practices/organize-work.md @@ -17,6 +17,8 @@ redirect_from: allowTitleToDifferFromFilename: true category: - Get started with GitHub Enterprise +docsTeamMetrics: + - enterprise-onboarding --- ## Use organizations for work or governance diff --git a/content/admin/concepts/enterprise-best-practices/use-innersource.md b/content/admin/concepts/enterprise-best-practices/use-innersource.md index 0592f8d190fb..736934c1b4f8 100644 --- a/content/admin/concepts/enterprise-best-practices/use-innersource.md +++ b/content/admin/concepts/enterprise-best-practices/use-innersource.md @@ -11,6 +11,8 @@ redirect_from: allowTitleToDifferFromFilename: true category: - Get started with GitHub Enterprise +docsTeamMetrics: + - enterprise-onboarding --- You can use innersource practices to drive collaboration and productivity in your enterprise. Innersource makes it easy for all employees to discover and reuse work. This allows development teams to learn from each other's work, share their expertise, and avoid duplicating effort to recreate common services. diff --git a/content/admin/concepts/enterprise-fundamentals/automations-in-your-enterprise.md b/content/admin/concepts/enterprise-fundamentals/automations-in-your-enterprise.md index f62d3c83a040..9747fbc30d1c 100644 --- a/content/admin/concepts/enterprise-fundamentals/automations-in-your-enterprise.md +++ b/content/admin/concepts/enterprise-fundamentals/automations-in-your-enterprise.md @@ -9,6 +9,8 @@ redirect_from: - /enterprise-onboarding/github-apps/automations-in-your-enterprise category: - Get started with GitHub Enterprise +docsTeamMetrics: + - enterprise-onboarding --- Automation on {% data variables.product.github %} typically involves multiple components working together. The most important {% data variables.product.github %} native components are: diff --git a/content/admin/concepts/enterprise-fundamentals/roles-in-an-enterprise.md b/content/admin/concepts/enterprise-fundamentals/roles-in-an-enterprise.md index daea04b8d6bc..9dcee8dec56b 100644 --- a/content/admin/concepts/enterprise-fundamentals/roles-in-an-enterprise.md +++ b/content/admin/concepts/enterprise-fundamentals/roles-in-an-enterprise.md @@ -11,6 +11,8 @@ redirect_from: contentType: concepts category: - Get started with GitHub Enterprise +docsTeamMetrics: + - enterprise-onboarding --- ## What are roles? diff --git a/content/admin/concepts/enterprise-fundamentals/teams-in-an-enterprise.md b/content/admin/concepts/enterprise-fundamentals/teams-in-an-enterprise.md index fd5029dc296c..573f0fea7e2a 100644 --- a/content/admin/concepts/enterprise-fundamentals/teams-in-an-enterprise.md +++ b/content/admin/concepts/enterprise-fundamentals/teams-in-an-enterprise.md @@ -11,6 +11,8 @@ redirect_from: contentType: concepts category: - Get started with GitHub Enterprise +docsTeamMetrics: + - enterprise-onboarding --- ## What are teams? diff --git a/content/admin/concepts/security-and-compliance/audit-log-for-an-enterprise.md b/content/admin/concepts/security-and-compliance/audit-log-for-an-enterprise.md index feae403c0956..6fff72d8760a 100644 --- a/content/admin/concepts/security-and-compliance/audit-log-for-an-enterprise.md +++ b/content/admin/concepts/security-and-compliance/audit-log-for-an-enterprise.md @@ -20,6 +20,8 @@ versions: contentType: concepts category: - Secure and govern your enterprise +docsTeamMetrics: + - enterprise-onboarding --- ## What are audit logs? diff --git a/content/admin/concepts/security-and-compliance/enterprise-policies.md b/content/admin/concepts/security-and-compliance/enterprise-policies.md index a0ae3313a179..e49e3021b117 100644 --- a/content/admin/concepts/security-and-compliance/enterprise-policies.md +++ b/content/admin/concepts/security-and-compliance/enterprise-policies.md @@ -12,6 +12,8 @@ redirect_from: - /enterprise-onboarding/govern-people-and-repositories/about-enterprise-policies category: - Secure and govern your enterprise +docsTeamMetrics: + - enterprise-onboarding --- ## What are enterprise policies and why are they important? diff --git a/content/admin/data-residency/github-copilot-with-data-residency.md b/content/admin/data-residency/github-copilot-with-data-residency.md index cd8001e1ebac..410ebc2814a9 100644 --- a/content/admin/data-residency/github-copilot-with-data-residency.md +++ b/content/admin/data-residency/github-copilot-with-data-residency.md @@ -69,7 +69,9 @@ The models available for {% data variables.product.prodname_copilot_short %} var * {% data variables.copilot.copilot_claude_opus_47 %} * {% data variables.copilot.copilot_claude_opus_48 %} * {% data variables.copilot.copilot_claude_opus_5 %} +* {% data variables.copilot.copilot_claude_opus_55 %} * {% data variables.copilot.copilot_claude_sonnet_5 %} +* {% data variables.copilot.copilot_claude_sonnet_55 %} * {% data variables.copilot.copilot_gemini_35_flash %} ## Pricing changes diff --git a/content/admin/managing-accounts-and-repositories/managing-organizations-in-your-enterprise/adding-organizations-to-your-enterprise.md b/content/admin/managing-accounts-and-repositories/managing-organizations-in-your-enterprise/adding-organizations-to-your-enterprise.md index a1799d3f5e6a..80a0936f2b1f 100644 --- a/content/admin/managing-accounts-and-repositories/managing-organizations-in-your-enterprise/adding-organizations-to-your-enterprise.md +++ b/content/admin/managing-accounts-and-repositories/managing-organizations-in-your-enterprise/adding-organizations-to-your-enterprise.md @@ -16,6 +16,8 @@ permissions: Enterprise owners contentType: how-tos category: - Manage accounts and repositories +docsTeamMetrics: + - enterprise-onboarding --- There are three ways to add organizations to your enterprise. diff --git a/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/governing-how-people-use-repositories-in-your-enterprise.md b/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/governing-how-people-use-repositories-in-your-enterprise.md index 7a032ed68348..6bde518a2d44 100644 --- a/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/governing-how-people-use-repositories-in-your-enterprise.md +++ b/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/governing-how-people-use-repositories-in-your-enterprise.md @@ -10,6 +10,8 @@ category: - Manage accounts and repositories redirect_from: - /enterprise-onboarding/govern-people-and-repositories/create-repository-policies +docsTeamMetrics: + - enterprise-onboarding --- {% data reusables.enterprise.repo-policy-rules-preview %} diff --git a/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/managing-custom-properties-for-repositories-in-your-enterprise.md b/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/managing-custom-properties-for-repositories-in-your-enterprise.md index 5d58c3ea2459..327dfd5dce42 100644 --- a/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/managing-custom-properties-for-repositories-in-your-enterprise.md +++ b/content/admin/managing-accounts-and-repositories/managing-repositories-in-your-enterprise/managing-custom-properties-for-repositories-in-your-enterprise.md @@ -9,6 +9,8 @@ versions: shortTitle: Custom properties category: - Manage accounts and repositories +docsTeamMetrics: + - enterprise-onboarding --- Custom properties allow you to decorate your repositories with information such as compliance frameworks, data sensitivity, or project details. Custom properties are private and can only be viewed by people with read permissions to the repository. An enterprise can have up to 100 property definitions. An allowed value list can have up to 200 items. diff --git a/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/assign-roles.md b/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/assign-roles.md index 0c1007fbbd22..c080b577f541 100644 --- a/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/assign-roles.md +++ b/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/assign-roles.md @@ -10,6 +10,8 @@ redirect_from: contentType: how-tos category: - Manage accounts and repositories +docsTeamMetrics: + - enterprise-onboarding --- Enterprise owners can assign custom and predefined **enterprise roles** to users and teams. Some roles can be assigned to enterprise teams, whereas other roles are only available for individual users. Find the section below for the role you want to assign. diff --git a/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/create-custom-roles.md b/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/create-custom-roles.md index 4621d3d6bfe7..06994e7b64b1 100644 --- a/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/create-custom-roles.md +++ b/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/create-custom-roles.md @@ -10,6 +10,8 @@ redirect_from: contentType: how-tos category: - Manage accounts and repositories +docsTeamMetrics: + - enterprise-onboarding --- To tailor access management to your company's needs, you can create custom roles for your{% ifversion enterprise-custom-roles %} enterprise account and{% endif %} organizations. diff --git a/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/identify-role-requirements.md b/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/identify-role-requirements.md index a420b8cc431e..45a200f08d23 100644 --- a/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/identify-role-requirements.md +++ b/content/admin/managing-accounts-and-repositories/managing-roles-in-your-enterprise/identify-role-requirements.md @@ -10,6 +10,8 @@ redirect_from: - /enterprise-onboarding/setting-up-organizations-and-teams/identify-role-requirements category: - Manage accounts and repositories +docsTeamMetrics: + - enterprise-onboarding --- Roles control people's access to settings and resources in your enterprise and organizations. For an introduction to roles, see [AUTOTITLE](/admin/concepts/enterprise-fundamentals/roles-in-an-enterprise). diff --git a/content/admin/managing-accounts-and-repositories/managing-users-in-your-enterprise/create-enterprise-teams.md b/content/admin/managing-accounts-and-repositories/managing-users-in-your-enterprise/create-enterprise-teams.md index 78808134811c..7d0c728f6001 100644 --- a/content/admin/managing-accounts-and-repositories/managing-users-in-your-enterprise/create-enterprise-teams.md +++ b/content/admin/managing-accounts-and-repositories/managing-users-in-your-enterprise/create-enterprise-teams.md @@ -12,6 +12,8 @@ redirect_from: contentType: how-tos category: - Manage accounts and repositories +docsTeamMetrics: + - enterprise-onboarding --- To simplify administration at scale, you can create enterprise teams. {% data reusables.enterprise.enterprise-teams-can %} diff --git a/content/admin/managing-github-apps-for-your-enterprise/creating-github-apps-for-your-enterprise.md b/content/admin/managing-github-apps-for-your-enterprise/creating-github-apps-for-your-enterprise.md index 94a66e9e5fba..22e92af1f47c 100644 --- a/content/admin/managing-github-apps-for-your-enterprise/creating-github-apps-for-your-enterprise.md +++ b/content/admin/managing-github-apps-for-your-enterprise/creating-github-apps-for-your-enterprise.md @@ -11,6 +11,8 @@ redirect_from: contentType: how-tos category: - Enable GitHub features for your enterprise +docsTeamMetrics: + - enterprise-onboarding --- You can create a {% data variables.product.prodname_github_app %} under your enterprise account. The app can only be installed on{% ifversion enterprise-installed-apps %} your enterprise or{% endif %} organizations within your enterprise, and can only be authorized by members of your enterprise. The app can't be installed on user accounts. diff --git a/content/apps/using-github-apps/installing-a-github-app-on-your-enterprise.md b/content/apps/using-github-apps/installing-a-github-app-on-your-enterprise.md index 696084ab2d33..4e39cfaf76b6 100644 --- a/content/apps/using-github-apps/installing-a-github-app-on-your-enterprise.md +++ b/content/apps/using-github-apps/installing-a-github-app-on-your-enterprise.md @@ -9,6 +9,8 @@ redirect_from: permissions: 'Enterprise owners can install {% data variables.product.prodname_github_apps %} on their enterprise. App managers cannot install apps at the enterprise level.' category: - Install and authorize apps +docsTeamMetrics: + - enterprise-onboarding --- > [!NOTE] diff --git a/content/code-security/how-tos/secure-at-scale/configure-organization-security/manage-usage-and-access/giving-org-access-private-registries.md b/content/code-security/how-tos/secure-at-scale/configure-organization-security/manage-usage-and-access/giving-org-access-private-registries.md index efe3f68b28bf..9395fde9ac23 100644 --- a/content/code-security/how-tos/secure-at-scale/configure-organization-security/manage-usage-and-access/giving-org-access-private-registries.md +++ b/content/code-security/how-tos/secure-at-scale/configure-organization-security/manage-usage-and-access/giving-org-access-private-registries.md @@ -108,9 +108,6 @@ See [AUTOTITLE](/code-security/how-tos/secure-your-supply-chain/manage-your-depe OIDC (OpenID Connect) authentication allows {% data variables.product.prodname_dependabot %} to use short-lived credentials from your cloud identity provider to access private registries, eliminating the need to store long-lived secrets. With OIDC, credentials are generated dynamically for each {% data variables.product.prodname_dependabot %} update job. You must configure a trust relationship between your cloud provider and {% data variables.product.github %} before {% data variables.product.prodname_dependabot %} can authenticate. -> [!NOTE] -> OIDC authentication for organization-level private registries is currently supported by {% data variables.product.prodname_dependabot %}. It is not supported by {% data variables.product.prodname_code_scanning %} default setup. - When you select **OIDC** as the authentication method for a private registry, choose one of the supported providers and fill in the required fields: * **Azure**: Enter the **Tenant ID** (Azure AD tenant ID) and **Client ID** (Azure AD application client ID). You must configure a federated credential in Azure AD that trusts {% data variables.product.github %}'s OIDC provider. diff --git a/content/code-security/reference/code-scanning/codeql/codeql-cli-manual/resolve-library-paths.md b/content/code-security/reference/code-scanning/codeql/codeql-cli-manual/resolve-library-paths.md index 48a015357a33..5324fd264652 100644 --- a/content/code-security/reference/code-scanning/codeql/codeql-cli-manual/resolve-library-paths.md +++ b/content/code-security/reference/code-scanning/codeql/codeql-cli-manual/resolve-library-paths.md @@ -9,6 +9,9 @@ product: '{% data reusables.gated-features.codeql %}' category: - Find CodeQL CLI commands autogenerated: codeql-cli +intro: |- + [Deep plumbing] Determine QL library paths and dbschemes for multiple + queries. --- diff --git a/content/codespaces/about-codespaces/deep-dive.md b/content/codespaces/about-codespaces/deep-dive.md index feeeefb1cd0e..b714a48b6113 100644 --- a/content/codespaces/about-codespaces/deep-dive.md +++ b/content/codespaces/about-codespaces/deep-dive.md @@ -82,7 +82,7 @@ If you work on codespaces in {% data variables.product.prodname_vscode %}, you c ### Closing or stopping your codespace -Your codespace will keep running while you are using it, but will time out after a period of inactivity. File changes from the editor and terminal output are counted as activity, so your codespace will not time out if terminal output is continuing. The default inactivity timeout period is 30 minutes. You can define your personal timeout setting for codespaces you create, but this may be overruled by an organization timeout policy. For more information, see [AUTOTITLE](/codespaces/setting-your-user-preferences/setting-your-timeout-period-for-github-codespaces). +Your codespace will keep running while you are using it, up to a maximum lifetime of 12 hours, but will time out after a period of inactivity. File changes from the editor and terminal output are counted as activity, so your codespace will not time out if terminal output is continuing. The default inactivity timeout period is 30 minutes. You can define your personal timeout setting for codespaces you create, but this may be overruled by an organization timeout policy. For more information, see [AUTOTITLE](/codespaces/setting-your-user-preferences/setting-your-timeout-period-for-github-codespaces). For more information about the maximum lifetime, see [AUTOTITLE](/codespaces/about-codespaces/understanding-the-codespace-lifecycle#maximum-lifetime-of-a-codespace). If a codespace times out it will stop running, but you can restart it from the browser tab (if you were using the codespace in the browser), from within {% data variables.product.prodname_vscode_shortname %}, or from your list of codespaces at [https://github.com/codespaces](https://github.com/codespaces). @@ -152,4 +152,3 @@ If you want to make changes to your codespace that will be more robust over rebu * [AUTOTITLE](/codespaces/managing-codespaces-for-your-organization/enabling-or-disabling-github-codespaces-for-your-organization) * [AUTOTITLE](/codespaces/managing-codespaces-for-your-organization/managing-the-cost-of-github-codespaces-in-your-organization) * [AUTOTITLE](/codespaces/setting-up-your-project-for-codespaces/adding-a-dev-container-configuration) -* [AUTOTITLE](/codespaces/about-codespaces/understanding-the-codespace-lifecycle) diff --git a/content/codespaces/about-codespaces/understanding-the-codespace-lifecycle.md b/content/codespaces/about-codespaces/understanding-the-codespace-lifecycle.md index cf0aca59023b..015ae64d4502 100644 --- a/content/codespaces/about-codespaces/understanding-the-codespace-lifecycle.md +++ b/content/codespaces/about-codespaces/understanding-the-codespace-lifecycle.md @@ -44,6 +44,16 @@ If you leave your codespace running without interaction, or if you exit your cod When a codespace times out, your data is preserved from the last time your changes were saved. For more information, see [Saving changes in a codespace](#saving-changes-in-a-codespace). +## Maximum lifetime of a codespace + +A codespace has a maximum lifetime of 12 hours, regardless of your idle timeout policy or settings. This limit applies even while you are actively using the codespace. + +As a codespace approaches this 12-hour limit, you will see the following warning: + +> Your codespace must be stopped soon. Stop and then reconnect to your codespace to keep working. + +Your data is saved, then the codespace automatically stops. To continue working, restart the codespace. For more information, see [AUTOTITLE](/codespaces/developing-in-a-codespace/stopping-and-starting-a-codespace#restarting-a-codespace). + ## Rebuilding a codespace You can rebuild your codespace to implement changes you've made to your dev container configuration. For most uses, you can create a new codespace as an alternative to rebuilding a codespace. By default, when you rebuild your codespace, {% data variables.product.prodname_github_codespaces %} will reuse images from your cache to speed up the rebuild process. Alternatively, you can perform a full rebuild, which clears your cache and rebuilds the container with fresh images. diff --git a/content/codespaces/developing-in-a-codespace/stopping-and-starting-a-codespace.md b/content/codespaces/developing-in-a-codespace/stopping-and-starting-a-codespace.md index 3fdac1a8134c..d793581a774a 100644 --- a/content/codespaces/developing-in-a-codespace/stopping-and-starting-a-codespace.md +++ b/content/codespaces/developing-in-a-codespace/stopping-and-starting-a-codespace.md @@ -16,6 +16,9 @@ category: {% data reusables.codespaces.stopping-a-codespace %} +> [!NOTE] +> A codespace also has a maximum lifetime of 12 hours, regardless of its idle timeout setting. For more information, see [AUTOTITLE](/codespaces/about-codespaces/understanding-the-codespace-lifecycle#maximum-lifetime-of-a-codespace). + Regardless of where you created or access your codespaces, you can view and manage them in your browser at https://github.com/codespaces. ## Stopping a codespace @@ -86,7 +89,3 @@ When you restart a codespace you can choose to open it in {% data variables.prod 1. In the list of codespaces, select the codespace you want to restart. {% endvscode %} - -## Further reading - -* [AUTOTITLE](/codespaces/about-codespaces/understanding-the-codespace-lifecycle) diff --git a/content/codespaces/setting-your-user-preferences/setting-your-timeout-period-for-github-codespaces.md b/content/codespaces/setting-your-user-preferences/setting-your-timeout-period-for-github-codespaces.md index c240ac25a588..bd6b036dfbb6 100644 --- a/content/codespaces/setting-your-user-preferences/setting-your-timeout-period-for-github-codespaces.md +++ b/content/codespaces/setting-your-user-preferences/setting-your-timeout-period-for-github-codespaces.md @@ -17,6 +17,8 @@ category: A codespace will stop running after a period of inactivity. By default this period is 30 minutes, but you can specify a longer or shorter default timeout period in your personal settings on {% data variables.product.prodname_dotcom %}. The updated setting will apply to any new codespaces you create. You can also specify a timeout when you use {% data variables.product.prodname_cli %} to create a codespace. +Regardless of your idle timeout setting, a codespace has a maximum lifetime of 12 hours. For more information, see [AUTOTITLE](/codespaces/about-codespaces/understanding-the-codespace-lifecycle#maximum-lifetime-of-a-codespace). + > [!WARNING] > Codespaces compute usage is billed for the duration for which a codespace is active. If you're not using a codespace but it remains running, and hasn't yet timed out, you are billed for the total time that the codespace was active, irrespective of whether you were using it. For more information, see [AUTOTITLE](/billing/concepts/product-billing/github-codespaces#pricing). diff --git a/data/reusables/actions/azure-vnet-supported-regions.md b/data/reusables/actions/azure-vnet-supported-regions.md index 854542dd9549..bd9b6060a826 100644 --- a/data/reusables/actions/azure-vnet-supported-regions.md +++ b/data/reusables/actions/azure-vnet-supported-regions.md @@ -14,11 +14,9 @@ The following regions are supported on {% data variables.product.prodname_dotcom
  • EastUs
  • EastUs2
  • FranceCentral
  • -
  • GermanyWestCentral
  • JapanWest
  • KoreaCentral
  • NorthCentralUs
  • -
  • NorthEurope
  • NorwayEast
  • SouthCentralUs
  • SoutheastAsia
  • diff --git a/data/reusables/actions/workflows/section-specifying-workflows.md b/data/reusables/actions/workflows/section-specifying-workflows.md new file mode 100644 index 000000000000..8c5ab4183f76 --- /dev/null +++ b/data/reusables/actions/workflows/section-specifying-workflows.md @@ -0,0 +1,31 @@ + +When using the `workflow_run` event, you can specify which workflows can trigger your workflow. + +The `workflows` filters accept glob patterns that use characters like `*`, `**`, `+`, `?`, `!` and others to match more than one workflow name. If a name contains any of these characters and you want a literal match, you need to _escape_ each of these special characters with `\`. For more information about glob patterns, see the [AUTOTITLE](/actions/writing-workflows/workflow-syntax-for-github-actions#filter-pattern-cheat-sheet). + +For example, a workflow with the following trigger will only run when the workflow named `Build` runs: + +```yaml +on: + workflow_run: + workflows: ["Build"] + types: [requested] +``` + +A workflow with the following trigger will only run when a workflow whose name starts with `Build` completed: + +```yaml +on: + workflow_run: + workflows: ["Build*"] + types: [completed] +``` + +A workflow with the following trigger will only run when the workflow named `Build C++` completed: + +```yaml +on: + workflow_run: + workflows: ["Build C\\+\\+"] + types: [completed] +``` diff --git a/data/reusables/copilot/model-compliance/us-models.md b/data/reusables/copilot/model-compliance/us-models.md index d0fa75bac1b8..58ab00ff56d6 100644 --- a/data/reusables/copilot/model-compliance/us-models.md +++ b/data/reusables/copilot/model-compliance/us-models.md @@ -6,5 +6,7 @@ * GPT-5.3-Codex * Claude Haiku 4.5 * Claude Sonnet 5 +* Claude Sonnet 5.5 * Claude Opus 4.8 * Claude Opus 5 +* Claude Opus 5.5 diff --git a/data/reusables/repositories/repo-rulesets-settings.md b/data/reusables/repositories/repo-rulesets-settings.md index 58e612798f2c..34d680cb9fd4 100644 --- a/data/reusables/repositories/repo-rulesets-settings.md +++ b/data/reusables/repositories/repo-rulesets-settings.md @@ -1 +1 @@ -1. In the left sidebar, under "Code and automation," click **Rulesets**, then click **Rulesets**. +1. In the left sidebar, under{% ifversion fpt or ghec %} "Code, planning, and automation"{% elsif ghes %} "Code and automation"{% endif %}, click **Rulesets**, then click **Rulesets**. diff --git a/package-lock.json b/package-lock.json index 26cee73d6447..3ed24cad3e95 100644 --- a/package-lock.json +++ b/package-lock.json @@ -64,7 +64,7 @@ "mdast-util-to-hast": "^13.2.1", "mdast-util-to-markdown": "2.1.2", "mdast-util-to-string": "^4.0.0", - "next": "^16.3.3", + "next": "^16.3.6", "parse5": "8.0.1", "quick-lru": "7.0.1", "react": "^19.2.5", @@ -898,9 +898,9 @@ } }, "node_modules/@eslint/config-array/node_modules/brace-expansion": { - "version": "1.1.18", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", - "integrity": "sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==", + "version": "1.1.21", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.21.tgz", + "integrity": "sha512-9zeA+KLZNNzglF2TPKRQEDyx6Yby7daAkuy8MiPzpXPsYDWi/DRM8jmwUDxokQjYqBpv5DgPiwD4h4ZZSy1Ujw==", "dev": true, "license": "MIT", "dependencies": { @@ -989,9 +989,9 @@ } }, "node_modules/@eslint/eslintrc/node_modules/brace-expansion": { - "version": "1.1.18", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", - "integrity": "sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==", + "version": "1.1.21", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.21.tgz", + "integrity": "sha512-9zeA+KLZNNzglF2TPKRQEDyx6Yby7daAkuy8MiPzpXPsYDWi/DRM8jmwUDxokQjYqBpv5DgPiwD4h4ZZSy1Ujw==", "dev": true, "license": "MIT", "dependencies": { @@ -2182,15 +2182,15 @@ } }, "node_modules/@next/env": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/env/-/env-16.3.3.tgz", - "integrity": "sha512-U2eYQRwXj+dsqxV79zFqExDdatnNY/ZWc2nsJU1p/OgT7fd3dXwlF6OjYaFQCfMoeTA19PWq+wVmYgimVA+V+g==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/env/-/env-16.3.6.tgz", + "integrity": "sha512-x9Vblze1EbtltQYnNH38xCPWU3TVfBd1eXqA3+w9+BTpedkkdNpAaltXlGQ/nsc1+E0mVTNrtcbX3GoO09zeLQ==", "license": "MIT" }, "node_modules/@next/swc-darwin-arm64": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-darwin-arm64/-/swc-darwin-arm64-16.3.3.tgz", - "integrity": "sha512-8Hiv32QJPwdV6KYJ8meR9SBA061tQqnIKTJDocvOXlEQqib0xMFpzArosuffFUUc0sslbh7QQ8a3Yey1QV8EIw==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-darwin-arm64/-/swc-darwin-arm64-16.3.6.tgz", + "integrity": "sha512-E/7GEqaUkt8mk/T8v9lAnrhzR06kdq1ZBkC12F8tAMkdIadwNp3H1KqHynDHrpcTlGCUdq/qu6vUL2aYVyYBdw==", "cpu": [ "arm64" ], @@ -2204,9 +2204,9 @@ } }, "node_modules/@next/swc-darwin-x64": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-darwin-x64/-/swc-darwin-x64-16.3.3.tgz", - "integrity": "sha512-A1lgKgwVchRYmSe467zdwhxT9040dd8lH+o65sL5Jet8fjB4kegw/rDyPIpYVRb6jAqwXFOJpjIXJLxQKLiE3A==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-darwin-x64/-/swc-darwin-x64-16.3.6.tgz", + "integrity": "sha512-yBE893/nDWTlaiBD1p+qgt7NUen4U5R6FXyH0s67Npq1S3E0cVSef1WIXC2xBRgQvwAvJq6DnS6Y6PrY0cy4Ew==", "cpu": [ "x64" ], @@ -2220,9 +2220,9 @@ } }, "node_modules/@next/swc-linux-arm64-gnu": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-linux-arm64-gnu/-/swc-linux-arm64-gnu-16.3.3.tgz", - "integrity": "sha512-bf0FIssMFueU2dm7vQEWWxk0c8UjKTdW0yzuh0sQsD8pf1+KCLDdaqhYZNMYGmXwEOiHAUzgBKudovIlcvvBjg==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-linux-arm64-gnu/-/swc-linux-arm64-gnu-16.3.6.tgz", + "integrity": "sha512-KJDpjBqBPYlvkivmyrp+Qys6k/7ksbqGQvRVc6ZEGfR+cjQxx+nUkJaWmNZJsmoOrqYNbaXByF8wa0lBwDhB3Q==", "cpu": [ "arm64" ], @@ -2239,9 +2239,9 @@ } }, "node_modules/@next/swc-linux-arm64-musl": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-linux-arm64-musl/-/swc-linux-arm64-musl-16.3.3.tgz", - "integrity": "sha512-W7viwCk9JY/cAkdz/A273rd5bb3RgT/IHwR7Upv90tunjBWNtAAhGhoecHh+teRNRSinuAFmE+l7fwZ4YKkrXg==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-linux-arm64-musl/-/swc-linux-arm64-musl-16.3.6.tgz", + "integrity": "sha512-mqNg2K+hvWskSRb/QM+Ix412DvBsuSF0XV+frTSw5vmoucNnIlynFwKYew8D01bfATErMOM7Bujrf0BA5DRKFA==", "cpu": [ "arm64" ], @@ -2258,9 +2258,9 @@ } }, "node_modules/@next/swc-linux-x64-gnu": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-linux-x64-gnu/-/swc-linux-x64-gnu-16.3.3.tgz", - "integrity": "sha512-0W46zw1N3ODpI6n0GeivHvvob1pooozgZVqy65k0mh4/7vr+FbY9+WpHzNVXjHipJf/A3FDheBG19H1s5A25rA==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-linux-x64-gnu/-/swc-linux-x64-gnu-16.3.6.tgz", + "integrity": "sha512-nFncBNGAYouRHjRVaITs9beZRfhX4ssVwpnvPIAbkZVH6LtGoAVlH4bJ8Cnf9SOo9bsXgPFer/GdHtEE3JNOkw==", "cpu": [ "x64" ], @@ -2277,9 +2277,9 @@ } }, "node_modules/@next/swc-linux-x64-musl": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-linux-x64-musl/-/swc-linux-x64-musl-16.3.3.tgz", - "integrity": "sha512-H4mBso8ZTMBPtdT0PN0pBx2ayTvQuTuvS6qT13d77yVFJXAPCxkyIhLTmdMaGTJs0krQYI/qpzdHijCeihXhbg==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-linux-x64-musl/-/swc-linux-x64-musl-16.3.6.tgz", + "integrity": "sha512-5Mf3cHDGR/Iz0ng2Bj3zUR3p5QS9YK3Hn2QiAfavFmyF48zwThAjpFoiTKNIcOHLYS4zEk+gzyJ/9deQ2ZB8yQ==", "cpu": [ "x64" ], @@ -2296,9 +2296,9 @@ } }, "node_modules/@next/swc-win32-arm64-msvc": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-win32-arm64-msvc/-/swc-win32-arm64-msvc-16.3.3.tgz", - "integrity": "sha512-cTMUJpcEGmeywofCUfhR+rSsoE33+rVPnPEYNTNdLNlsOeEg/vktOsKUSTb28vUGqD2jkm4Zaskcwn7OCI6FQg==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-win32-arm64-msvc/-/swc-win32-arm64-msvc-16.3.6.tgz", + "integrity": "sha512-0jkJy0C2kbrJWTk4YLa3xk80pVBpx8FCHJym7CnUfDAXe/FWv5qT7SQJbR0KuemyxaEDlEx5WT4VQJoTW+/9Qw==", "cpu": [ "arm64" ], @@ -2312,9 +2312,9 @@ } }, "node_modules/@next/swc-win32-x64-msvc": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/@next/swc-win32-x64-msvc/-/swc-win32-x64-msvc-16.3.3.tgz", - "integrity": "sha512-2VR4cTBzHXaBjnGsuH6GyJjENzQOmHeAh11uY1iUhjm3j5dEUrVJuUj+VL78jaGi/Dik8xS76zEj18BsFhlVZQ==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/@next/swc-win32-x64-msvc/-/swc-win32-x64-msvc-16.3.6.tgz", + "integrity": "sha512-/YXjI1e5OXcZ7YpxRwgP/1jAV/SBKTzeVKqN2mk7mLpcICsyn3Gl5+dIfDTJp70M0ccMhyMMRso4v6mPDCGepg==", "cpu": [ "x64" ], @@ -5776,14 +5776,14 @@ } }, "node_modules/axios": { - "version": "1.18.1", - "resolved": "https://registry.npmjs.org/axios/-/axios-1.18.1.tgz", - "integrity": "sha512-3nTvFlvpn9Zu/RkHUqtc7/+al4UpRW5az71ap5zccp6e8RAYEzhMTecX8Dz1wWDYrPpUoB1HAQEGEAEvUr7S9g==", + "version": "1.20.0", + "resolved": "https://registry.npmjs.org/axios/-/axios-1.20.0.tgz", + "integrity": "sha512-r8aOh8j9cGKpgQAqpzrUHnSIc6a59Y3Xf/cv8sy1DrHCkZHzQGEuoq1tARk6qSyDdtQGSDgpb9kFlruzPvrgwg==", "dev": true, "license": "MIT", "dependencies": { "follow-redirects": "^1.16.0", - "form-data": "^4.0.5", + "form-data": "^4.0.6", "https-proxy-agent": "^5.0.1", "proxy-from-env": "^2.1.0" } @@ -6096,9 +6096,9 @@ } }, "node_modules/brace-expansion": { - "version": "5.0.9", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.9.tgz", - "integrity": "sha512-ScQ4IuvIEF1TMlP7Zt+vjJ//9zlPb2SDcxWxM3bk8s6t6GGdJ7KO1dCcTidOPJKePW30LE/2cT7wCyPho9/Wxg==", + "version": "5.0.12", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.12.tgz", + "integrity": "sha512-YovQ3rzhaLMIrDjNDMkNS01tea93qhEhG5xy8f6+R0l+dw3Ki+5sCoIoI942iuLZTHWogWktgwVDhU09iNEimQ==", "license": "MIT", "dependencies": { "balanced-match": "^4.0.2" @@ -7979,9 +7979,9 @@ } }, "node_modules/eslint-plugin-import/node_modules/brace-expansion": { - "version": "1.1.18", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", - "integrity": "sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==", + "version": "1.1.21", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.21.tgz", + "integrity": "sha512-9zeA+KLZNNzglF2TPKRQEDyx6Yby7daAkuy8MiPzpXPsYDWi/DRM8jmwUDxokQjYqBpv5DgPiwD4h4ZZSy1Ujw==", "dev": true, "license": "MIT", "dependencies": { @@ -8053,9 +8053,9 @@ } }, "node_modules/eslint-plugin-jsx-a11y/node_modules/brace-expansion": { - "version": "1.1.18", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", - "integrity": "sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==", + "version": "1.1.21", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.21.tgz", + "integrity": "sha512-9zeA+KLZNNzglF2TPKRQEDyx6Yby7daAkuy8MiPzpXPsYDWi/DRM8jmwUDxokQjYqBpv5DgPiwD4h4ZZSy1Ujw==", "dev": true, "license": "MIT", "dependencies": { @@ -8199,9 +8199,9 @@ } }, "node_modules/eslint/node_modules/brace-expansion": { - "version": "1.1.18", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", - "integrity": "sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==", + "version": "1.1.21", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.21.tgz", + "integrity": "sha512-9zeA+KLZNNzglF2TPKRQEDyx6Yby7daAkuy8MiPzpXPsYDWi/DRM8jmwUDxokQjYqBpv5DgPiwD4h4ZZSy1Ujw==", "dev": true, "license": "MIT", "dependencies": { @@ -8598,9 +8598,9 @@ "dev": true }, "node_modules/fast-uri": { - "version": "3.1.7", - "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.7.tgz", - "integrity": "sha512-dOvZVzjdZdz7phd9v6jCbwxrBW3fK6n8Rc0CtdmM4bumzMnxywBYhuph6J819RRw/ku+rLbelwfMunktuzVVHg==", + "version": "3.1.8", + "resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.8.tgz", + "integrity": "sha512-GZMtZUTNRpOVIECoXwLNZS5xUGE+mVNbTB8h/7Rwh2TFWcBQiPzTgyZi05BF9UMZKkLJv8XBRJTlU7zg8+ZfMg==", "funding": [ { "type": "github", @@ -11148,9 +11148,9 @@ } }, "node_modules/matcher-collection/node_modules/brace-expansion": { - "version": "1.1.18", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", - "integrity": "sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==", + "version": "1.1.21", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.21.tgz", + "integrity": "sha512-9zeA+KLZNNzglF2TPKRQEDyx6Yby7daAkuy8MiPzpXPsYDWi/DRM8jmwUDxokQjYqBpv5DgPiwD4h4ZZSy1Ujw==", "license": "MIT", "dependencies": { "balanced-match": "^1.0.0", @@ -12196,12 +12196,12 @@ } }, "node_modules/next": { - "version": "16.3.3", - "resolved": "https://registry.npmjs.org/next/-/next-16.3.3.tgz", - "integrity": "sha512-tuRTx1nQ/yVw83cwJBo9F+njGUgMn3UHQycreWHB8XsStvvAh1AthbI8/4IpKnFaF58F+iSiHejYOlMQ/eq83g==", + "version": "16.3.6", + "resolved": "https://registry.npmjs.org/next/-/next-16.3.6.tgz", + "integrity": "sha512-L+otWM/aQbYTx98aZhgEoMb4bZAXx1YVW4UMA/vuCyCoWG5HJyZUili8QAkqzrcC+5///tsz3s0M+SlyB5bLMw==", "license": "MIT", "dependencies": { - "@next/env": "16.3.3", + "@next/env": "16.3.6", "@swc/helpers": "0.5.23", "baseline-browser-mapping": "^2.9.19", "caniuse-lite": "^1.0.30001579", @@ -12215,15 +12215,15 @@ "node": ">=20.9.0" }, "optionalDependencies": { - "@next/swc-darwin-arm64": "16.3.3", - "@next/swc-darwin-x64": "16.3.3", - "@next/swc-linux-arm64-gnu": "16.3.3", - "@next/swc-linux-arm64-musl": "16.3.3", - "@next/swc-linux-x64-gnu": "16.3.3", - "@next/swc-linux-x64-musl": "16.3.3", - "@next/swc-win32-arm64-msvc": "16.3.3", - "@next/swc-win32-x64-msvc": "16.3.3", - "sharp": "^0.35.3" + "@next/swc-darwin-arm64": "16.3.6", + "@next/swc-darwin-x64": "16.3.6", + "@next/swc-linux-arm64-gnu": "16.3.6", + "@next/swc-linux-arm64-musl": "16.3.6", + "@next/swc-linux-x64-gnu": "16.3.6", + "@next/swc-linux-x64-musl": "16.3.6", + "@next/swc-win32-arm64-msvc": "16.3.6", + "@next/swc-win32-x64-msvc": "16.3.6", + "sharp": "^0.35.4" }, "peerDependencies": { "@opentelemetry/api": "^1.1.0", @@ -12310,9 +12310,9 @@ } }, "node_modules/nodemon/node_modules/brace-expansion": { - "version": "1.1.18", - "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.18.tgz", - "integrity": "sha512-Edep/X9fGqVNmzKBVsDYIOtD+z1tuezV70LBjdCst9Tqu76lsnvRiZ6oTic1n+/BIwX6QDGAO94PN4N2SADvtw==", + "version": "1.1.21", + "resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-1.1.21.tgz", + "integrity": "sha512-9zeA+KLZNNzglF2TPKRQEDyx6Yby7daAkuy8MiPzpXPsYDWi/DRM8jmwUDxokQjYqBpv5DgPiwD4h4ZZSy1Ujw==", "dev": true, "license": "MIT", "dependencies": { diff --git a/package.json b/package.json index b3dbd41c47a5..b176ae0f75f9 100644 --- a/package.json +++ b/package.json @@ -223,7 +223,7 @@ "mdast-util-to-hast": "^13.2.1", "mdast-util-to-markdown": "2.1.2", "mdast-util-to-string": "^4.0.0", - "next": "^16.3.3", + "next": "^16.3.6", "parse5": "8.0.1", "quick-lru": "7.0.1", "react": "^19.2.5", @@ -356,7 +356,7 @@ "brace-expansion": "^5.0.8" }, "sharp": "$sharp", - "fast-uri": "^3.1.7" + "fast-uri": "^3.1.8" }, "engines": { "node": "^24 || ^26" diff --git a/src/codeql-cli/scripts/convert-markdown-for-docs.ts b/src/codeql-cli/scripts/convert-markdown-for-docs.ts index 7bdf8bcd34bb..97d13dd4b964 100644 --- a/src/codeql-cli/scripts/convert-markdown-for-docs.ts +++ b/src/codeql-cli/scripts/convert-markdown-for-docs.ts @@ -92,7 +92,9 @@ export async function convertContentToDocs( let currentNodeIsDescription = false visit(ast, (rawNode) => { const node = rawNode as unknown as MdNode - if (node.type !== 'heading' && node.type !== 'paragraph') return false + // A bare return is CONTINUE. Returning false would mean EXIT, + // which stops the whole walk on the root node. + if (node.type !== 'heading' && node.type !== 'paragraph') return // The first paragraph after Description becomes intro frontmatter. if (node.children[0]?.value === 'Description' && node.children[0]?.type === 'text') { diff --git a/src/codeql-cli/tests/convert-markdown-for-docs.ts b/src/codeql-cli/tests/convert-markdown-for-docs.ts index 86b7661fc4e4..52b77be3b372 100644 --- a/src/codeql-cli/tests/convert-markdown-for-docs.ts +++ b/src/codeql-cli/tests/convert-markdown-for-docs.ts @@ -120,6 +120,15 @@ For more information, see \`codeql database analyze\`{.interpr expect(result.content).toContain('codeql database analyze') }) + test('sets intro frontmatter from the Description section', async () => { + const result = await convertContentToDocs(testContent, {}, 'bqrs-interpret.md') + + expect(result.data).toHaveProperty('intro') + expect(result.data.intro).toBe( + 'A command that interprets a single BQRS file according to the provided\nmetadata and generates output in the specified format.', + ) + }) + test('returns proper data structure', async () => { const result = await convertContentToDocs(testContent, {}, 'bqrs-interpret.md') diff --git a/src/codeql-queries/README.md b/src/codeql-queries/README.md index c7297459ecce..72bc43777c3d 100644 --- a/src/codeql-queries/README.md +++ b/src/codeql-queries/README.md @@ -64,9 +64,43 @@ The workflow automatically creates a new pull request with changes from both scr ## Local development -To run the pipeline locally, see the comments in the scripts: -- Security queries: [generate-code-scanning-query-list.ts](scripts/generate-code-scanning-query-list.ts) -- Code quality queries: [generate-code-quality-query-list.ts](scripts/generate-code-quality-query-list.ts) +Both scripts need the CodeQL CLI and a local clone of `github/codeql`. The security script also needs a private npm package. + +### 1. Clone github/codeql + +The scripts resolve query suites by path, so the repo has to be on disk. + +```sh +git clone git@github.com:github/codeql.git /tmp/codeql +``` + +### 2. Install the CodeQL CLI + +```sh +gh extension install github/gh-codeql +gh codeql set-channel nightly +gh codeql version +``` + +`gh codeql version` prints where it installed the executable, something like `~/.local/share/gh/extensions/gh-codeql/dist/nightly/codeql-bundle-/codeql`. Pass that path as `--codeql-path`. + +### 3. Install @github/cocofix + +Only `generate-code-scanning-query-list.ts` needs this, for autofix support data. It's a private package, so get the `DOCS_BOT_PAT_BASE` PAT from the vault, export it, then run this from the root of the repo: + +```sh +npm i --no-save '--@github:registry=https://npm.pkg.github.com' '--//npm.pkg.github.com/:_authToken=${DOCS_BOT_PAT_BASE}' @github/cocofix +``` + +### 4. Run a script + +```sh +npm run generate-code-quality-query-list -- \ + --codeql-path \ + --codeql-dir /tmp/codeql python | tee /tmp/python.md +``` + +Use `generate-code-scanning-query-list` for the security tables. The last argument is the language. ## Content team diff --git a/src/codeql-queries/scripts/generate-code-quality-query-list.ts b/src/codeql-queries/scripts/generate-code-quality-query-list.ts index eb422526288c..a8dcfcf66c92 100644 --- a/src/codeql-queries/scripts/generate-code-quality-query-list.ts +++ b/src/codeql-queries/scripts/generate-code-quality-query-list.ts @@ -1,13 +1,10 @@ -// Generates reusable Markdown listing code quality queries for one language, with categories. -// Requires a local github/codeql clone and a CodeQL CLI executable. -// Set up the clone with git clone git@github.com:github/codeql.git /tmp/codeql. -// Install the CLI with gh extension install github/gh-codeql, then gh codeql set-channel nightly. -// Run gh codeql version to find the installed codeql path. -// Example: -// npm run generate-code-quality-query-list -- \ -// --codeql-path ~/.local/share/gh/extensions/gh-codeql/dist/nightly/codeql-bundle-*/codeql \ -// --codeql-dir /tmp/codeql python | tee /tmp/python.md -// Inspect the generated Markdown with less /tmp/python.md. +/** + * Generates a Markdown table of the code quality queries for one language, + * with their categories, to be saved as a reusable. + * + * Running this locally needs the CodeQL CLI and a clone of github/codeql. + * See "Local development" in src/codeql-queries/README.md. + */ import fs from 'fs' import { execFileSync } from 'child_process' @@ -94,7 +91,6 @@ async function main(options: Options, language: string) { const categories = getCategories(tags || '') const url = getDocsLink(language, id) - // Category-less queries have no code quality docs row. if (categories.length) { queries[id] = { url, name, categories, severity: severity || 'N/A' } } else { @@ -107,7 +103,6 @@ async function main(options: Options, language: string) { } function decorate(query: Query): QueryExtended { - // Maintainability outranks reliability for table sorting. const primaryCategory = query.categories.includes('maintainability') ? 'maintainability' : query.categories.includes('reliability') @@ -122,7 +117,7 @@ async function main(options: Options, language: string) { const entries = Object.values(queries).map(decorate) - // Sort by primary category, then alphabetically by name. + // Maintainability first, then alphabetical by name. entries.sort((a, b) => { if (a.primaryCategory === 'maintainability' && b.primaryCategory !== 'maintainability') return -1 @@ -172,7 +167,8 @@ function getMetadata(options: Options, queryFile: string): QueryMetadata { }) const parsed = JSON.parse(metadataJson) - // CodeQL emits severity through several metadata shapes, depending on the query source. + // `codeql resolve metadata` reports @problem.severity in several different JSON shapes, + // so try each one. const severity = parsed.problem?.severity || // Nested: { problem: { severity: "error" } } parsed['@problem']?.severity || // Nested with @: { "@problem": { severity: "error" } } @@ -182,7 +178,7 @@ function getMetadata(options: Options, queryFile: string): QueryMetadata { parsed['@severity'] // With @: { "@severity": "error" } if (options.verbose) { - // Verbose mode logs metadata keys once to avoid noisy output. + // Only dump the key list once. if (!getMetadata.shownKeys) { console.log(chalk.yellow('Available metadata keys:'), Object.keys(parsed)) if (parsed.problem) { @@ -209,13 +205,14 @@ function getMetadata(options: Options, queryFile: string): QueryMetadata { getMetadata.shownKeys = false -// Example: cpp and external-entity-expansion become +// getDocsLink('cpp', 'external-entity-expansion') returns // https://codeql.github.com/codeql-query-help/cpp/cpp-external-entity-expansion/ function getDocsLink(language: string, queryId: string) { return `https://codeql.github.com/codeql-query-help/${language}/${queryId.replaceAll('/', '-')}/` } -// Example tags with maintainability and reliability return those categories in source order. +// getCategories('maintainability readability reliability external/cwe/cwe-1078') +// returns ['maintainability', 'reliability'] function getCategories(tags: string) { const categories: string[] = [] for (const tag of tags.split(/\s+/g)) { diff --git a/src/codeql-queries/scripts/generate-code-scanning-query-list.ts b/src/codeql-queries/scripts/generate-code-scanning-query-list.ts index e9bafc2b98dd..a8d372b520da 100644 --- a/src/codeql-queries/scripts/generate-code-scanning-query-list.ts +++ b/src/codeql-queries/scripts/generate-code-scanning-query-list.ts @@ -1,16 +1,11 @@ -// Generates reusable Markdown listing CodeQL code scanning queries for one language, with CWEs. -// Requires a local github/codeql clone and a CodeQL CLI executable. -// Set up the clone with git clone git@github.com:github/codeql.git /tmp/codeql. -// Install the CLI with gh extension install github/gh-codeql, then gh codeql set-channel nightly. -// Run gh codeql version to find the installed codeql path. -// Also requires @github/cocofix, installed locally with DOCS_BOT_PAT_BASE from the vault: -// npm i --no-save '--@github:registry=https://npm.pkg.github.com' \ -// '--//npm.pkg.github.com/:_authToken=${DOCS_BOT_PAT_BASE}' @github/cocofix -// Example: -// npm run generate-code-scanning-query-list -- \ -// --codeql-path ~/.local/share/gh/extensions/gh-codeql/dist/nightly/codeql-bundle-*/codeql \ -// --codeql-dir /tmp/codeql python | tee /tmp/python.md -// Inspect the generated Markdown with less /tmp/python.md. +/** + * Generates a Markdown table of the security queries for one language, with + * their CWEs, to be saved as a reusable. + * + * Running this locally needs the CodeQL CLI, a clone of github/codeql, + * and the private @github/cocofix package. + * See "Local development" in src/codeql-queries/README.md. + */ import fs from 'fs' import { execFileSync } from 'child_process' @@ -114,7 +109,9 @@ async function main(options: Options, language: string) { const url = getDocsLink(language, id) const autofixSupport = autofixSupportedQueryIds.includes(id) ? 'default' : 'none' - // CWE-less queries cover metadata or metrics and have no docs link. + // Queries without CWEs are metadata and metrics queries, + // like counting lines of code. + // They have no docs link, so skip them. if (cwes.length) { if (!(id in queries)) { queries[id] = { url, name, packs: [], cwes, autofixSupport } @@ -199,13 +196,14 @@ function getMetadata(options: Options, queryFile: string): QueryMetadata { return parsed } -// Example: cpp and external-entity-expansion become +// getDocsLink('cpp', 'external-entity-expansion') returns // https://codeql.github.com/codeql-query-help/cpp/cpp-external-entity-expansion/ function getDocsLink(language: string, queryId: string) { return `https://codeql.github.com/codeql-query-help/${language}/${queryId.replaceAll('/', '-')}/` } -// Example tags with external/cwe/cwe-1078 and external/cwe/cwe-670 return 1078 and 670. +// getCWEs('maintainability external/cwe/cwe-1078 external/cwe/cwe-670') +// returns ['1078', '670'] function getCWEs(tags: string) { const cwes: string[] = [] for (const tag of tags.split(/\s+/g)) { diff --git a/src/content-linter/lib/linting-rules/code-annotation-comment-spacing.ts b/src/content-linter/lib/linting-rules/code-annotation-comment-spacing.ts index 8a14cd9003fd..c985ab8a5dd4 100644 --- a/src/content-linter/lib/linting-rules/code-annotation-comment-spacing.ts +++ b/src/content-linter/lib/linting-rules/code-annotation-comment-spacing.ts @@ -42,13 +42,12 @@ export const codeAnnotationCommentSpacing = { } if (commentMatch && restOfLine !== null && commentChar !== null) { - // Skip shebang lines (#!/...) + // Treat shebang lines as executable directives, not code comments. if (trimmedLine.startsWith('#!')) { continue } if (restOfLine === '' || restOfLine.startsWith(' ')) { - // If it starts with a space, make sure it's exactly one space if (restOfLine.startsWith(' ') && restOfLine.length > 1 && restOfLine[1] === ' ') { const lineNumber: number = token.lineNumber + index + 1 const fixedLine: string = line.replace( diff --git a/src/content-linter/lib/linting-rules/ctas-schema.ts b/src/content-linter/lib/linting-rules/ctas-schema.ts index 95a7419996c1..f63a808b3e13 100644 --- a/src/content-linter/lib/linting-rules/ctas-schema.ts +++ b/src/content-linter/lib/linting-rules/ctas-schema.ts @@ -13,7 +13,6 @@ export const ctasSchema: Rule = { description: 'CTA URLs must conform to the schema', tags: ['ctas', 'schema', 'urls'], function: (params: RuleParams, onError: RuleErrorCallback) => { - // Find all URLs in the content that might be CTAs const urlRegex = /https?:\/\/[^\s)\]{}'">]+/g const content = params.lines.join('\n') @@ -21,21 +20,21 @@ export const ctasSchema: Rule = { while ((match = urlRegex.exec(content)) !== null) { const url = match[0] - // A ref_ parameter is what marks a URL as a CTA. + // CTA URLs carry ref_ parameters. if (!url.includes('ref_')) continue - // Only validate CTA URLs on GitHub domains + // CTA schema checks apply only to github.com and desktop.github.com. let hostname: string try { hostname = new URL(url).hostname } catch { - // Invalid URL, skip validation + // Malformed URLs cannot be CTA schema checked. continue } const allowedHosts = ['github.com', 'desktop.github.com'] if (!allowedHosts.includes(hostname)) continue - // Skip placeholder/documentation example URLs + // Docs placeholder URLs with tokens like DESTINATION or CTA+NAME skip CTA schema checks. const isPlaceholderUrl = /[A-Z_]+/.test(url) && (url.includes('DESTINATION') || @@ -49,7 +48,6 @@ export const ctasSchema: Rule = { const urlObj = new URL(url) const searchParams = urlObj.searchParams - // Extract ref_ parameters const refParams: Record = {} const hasRefParams = Array.from(searchParams.keys()).some((key) => key.startsWith('ref_')) @@ -61,7 +59,7 @@ export const ctasSchema: Rule = { } } - // Check if this has old CTA parameters that can be auto-fixed + // ref_cta, ref_loc, and ref_page map to schema fields. const hasOldParams = 'ref_cta' in refParams || 'ref_loc' in refParams || 'ref_page' in refParams @@ -125,7 +123,7 @@ export const ctasSchema: Rule = { } } } catch { - // Invalid URL, skip validation + // Conversion and schema validation failures cannot produce a reliable lint error. continue } } diff --git a/src/content-linter/lib/linting-rules/early-access-references.ts b/src/content-linter/lib/linting-rules/early-access-references.ts index 2ee8089ac92b..327487f64a79 100644 --- a/src/content-linter/lib/linting-rules/early-access-references.ts +++ b/src/content-linter/lib/linting-rules/early-access-references.ts @@ -13,15 +13,12 @@ interface Frontmatter { const ERROR_MESSAGE = 'An early access reference appears to be used in a non-early access doc. Remove early access references or disable this rule.' -// Early access content is allowed to use early access references -// There are several existing allowed references to `early access` -// as a GitHub feature. This rule focuses on references to early -// access pages. +// The rule allows early access files to reference early access pages. +// Other files keep feature-name mentions. const isEarlyAccessFilepath = (filepath: string): boolean => filepath.includes('early-access') const EARLY_ACCESS_REGEX = /early-access/gi -// This is a pattern seen in link paths for articles about -// early access. This pattern is ok. +// This path fragment identifies articles about early access, not page references. const EARLY_ACCESS_ARTICLE_REGEX = /-early-access-/ export const earlyAccessReferences: Rule = { @@ -33,7 +30,6 @@ export const earlyAccessReferences: Rule = { function: (params: RuleParams, onError: RuleErrorCallback) => { if (isEarlyAccessFilepath(params.name)) return - // Find errors in content for (let i = 0; i < params.lines.length; i++) { const line = params.lines[i] const matches = line.match(EARLY_ACCESS_REGEX) @@ -60,21 +56,17 @@ export const frontmatterEarlyAccessReferences: Rule = { const filepath = params.name if (isEarlyAccessFilepath(filepath)) return - // Find errors in frontmatter const fm = getFrontmatter(params.lines) as Frontmatter | null if (!fm) return - // The redirect_from property is allowed to contain early-access paths + // redirect_from preserves early-access paths for redirects. delete fm.redirect_from - // The landing page must link to early-access content so the - // children property doesn't need to be checked in that case. - // Also exclude fixture index files. + // The home page links to early access content, and fixture indexes mirror that pattern. if (filepath === 'content/index.md' || filepath.includes('fixtures/content/index.md')) delete fm.children - // Convert the updated frontmatter back to a string so we can search it - // for 'early-access'. + // YAML serialization lets the rule search every remaining frontmatter value. const fmStrings = dump(fm).split('\n') for (const line of fmStrings) { diff --git a/src/content-linter/lib/linting-rules/expired-content.ts b/src/content-linter/lib/linting-rules/expired-content.ts index 1cbde7351f6f..c255c8c4b648 100644 --- a/src/content-linter/lib/linting-rules/expired-content.ts +++ b/src/content-linter/lib/linting-rules/expired-content.ts @@ -2,14 +2,9 @@ import { addError, newLineRe } from 'markdownlint-rule-helpers' import type { RuleParams, RuleErrorCallback, MarkdownToken, Rule } from '@/content-linter/types' -// This rule looks for opening and closing HTML comment tags that -// contain an expiration date in the format: -// -// This is content that is -// expired that does not expire. -// -// The `end expires` closing tag closes the content that is expired -// and must be removed. +// Expired ranges wrap content in matching HTML comments: +// This content expires here. +// The closing comment identifies the content to remove with the expired range. export const expiredContent: Rule = { names: ['GHD038', 'expired-content'], description: 'Expired content must be remediated.', @@ -20,8 +15,6 @@ export const expiredContent: Rule = { ) for (const token of tokensToCheck) { - // Looking for just opening tag with format: - // const match = token.content?.match(//) if (!match || !token.content) continue @@ -29,9 +22,7 @@ export const expiredContent: Rule = { const today = new Date() if (today < expireDate) continue - // We want the content split by line since not all token.content is in one line - // to get the correct range of the expired content. Below is how markdownlint - // grabs the token.content by line. + // Markdownlint reports inline token content by line, so range calculations follow that split. const contentByLine = token.content.replace(/^\uFEFF/, '').split(newLineRe) const lineOfMatch = contentByLine.findIndex((element) => element.includes(match[0])) const startRange = lineOfMatch !== -1 ? contentByLine[lineOfMatch].indexOf(match[0]) + 1 : 1 @@ -50,12 +41,7 @@ export const expiredContent: Rule = { export const DAYS_TO_WARN_BEFORE_EXPIRED = 14 -// This rule looks for content that will expire in `DAYS_TO_WARN_BEFORE_EXPIRED` -// days. The rule looks for opening and closing HTML comment tags that -// contain an expiration date in the format: -// -// This is content that is scheduled -// to expire that does not expire. +// Expiring-soon checks the same expires range once the date falls in the warning window. export const expiringSoon: Rule = { names: ['GHD039', 'expiring-soon'], description: 'Content that expires soon should be proactively addressed.', @@ -66,8 +52,6 @@ export const expiringSoon: Rule = { ) for (const token of tokensToCheck) { - // Looking for just opening tag with format: - // const match = token.content?.match(//) if (!match || !token.content) continue @@ -75,8 +59,7 @@ export const expiringSoon: Rule = { const today = new Date() const futureDate = new Date() futureDate.setDate(today.getDate() + DAYS_TO_WARN_BEFORE_EXPIRED) - // Don't set warning if the content is already expired or - // if the content expires later than the DAYS_TO_WARN_BEFORE_EXPIRED + // Skip content that already expired or expires after the warning window. if (today > expireDate || expireDate > futureDate) continue addError( diff --git a/src/content-linter/lib/linting-rules/frontmatter-children.ts b/src/content-linter/lib/linting-rules/frontmatter-children.ts index 49be26d85003..f09c80f1a93e 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-children.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-children.ts @@ -10,12 +10,8 @@ interface Frontmatter { [key: string]: unknown } -/** - * Check if a child path is valid. - * Supports both: - * - Relative paths (e.g., /local-child) resolved from current directory - * - Absolute /content/ paths (e.g., /content/actions/workflows) resolved from content root - */ +// Child paths such as /local-child resolve relative to the current file. +// Paths such as /content/actions/workflows resolve from the content root. function isValidChildPath(childPath: string, currentFilePath: string): boolean { const ROOT = process.env.ROOT || '.' const contentDir = path.resolve(ROOT, 'content') @@ -23,18 +19,15 @@ function isValidChildPath(childPath: string, currentFilePath: string): boolean { let resolvedPath: string if (childPath.startsWith('/content/')) { - // Absolute path from the content root: strip the /content/ prefix. const absoluteChildPath = childPath.slice('/content/'.length) resolvedPath = path.resolve(contentDir, absoluteChildPath) } else { - // Relative path from current file's directory const currentDir: string = path.dirname(currentFilePath) const normalizedPath = childPath.startsWith('/') ? childPath.substring(1) : childPath resolvedPath = path.resolve(currentDir, normalizedPath) } - // Security check: ensure resolved path stays within content directory - // This prevents path traversal attacks using sequences like '../' + // Reject paths that resolve outside content to prevent traversal with ../. if (!resolvedPath.startsWith(contentDir + path.sep) && resolvedPath !== contentDir) { return false } @@ -49,7 +42,7 @@ function isValidChildPath(childPath: string, currentFilePath: string): boolean { return true } - // Check if the path exists as a directory (may have children) + // Accept directories because they may contain nested children. if (fs.existsSync(resolvedPath) && fs.statSync(resolvedPath).isDirectory()) { return true } diff --git a/src/content-linter/lib/linting-rules/frontmatter-content-type.ts b/src/content-linter/lib/linting-rules/frontmatter-content-type.ts index 109490a08c7a..9d673269e419 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-content-type.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-content-type.ts @@ -9,26 +9,17 @@ import type { RuleParams, RuleErrorCallback } from '@/content-linter/types' const RESPONSIBLE_USE_STRING = 'responsible-use' const GETTING_STARTED_STRING = 'getting-started' -// Directory names that correspond to a known content type. -// This includes the canonical contentType values (minus the special-purpose -// ones: homepage, landing, rai, other) plus directory-name aliases: -// - "responsible-use" → contentType "rai" -// - "getting-started" → contentType "get-started" (some products use this variant) +// Recognize canonical contentType values except homepage, landing, rai, and other. +// Aliases map responsible-use to rai and getting-started to get-started. const KNOWN_CONTENT_TYPE_DIRS = new Set([ ...contentTypesEnum.filter((t) => !['homepage', 'landing', 'rai', 'other'].includes(t)), RESPONSIBLE_USE_STRING, GETTING_STARTED_STRING, ]) -// Lazily computed set of product directories whose subdirectories all follow -// the content-type directory pattern. Once computed the set is reused for -// every file processed in the same lint run. +// Cache qualifying products so every file in the lint run reuses the directory scan. let qualifyingProducts: Set | null = null -/** - * Scan `content/` and return the set of product directory names whose - * **immediate** subdirectories are all in `KNOWN_CONTENT_TYPE_DIRS`. - */ function getQualifyingProducts(): Set { if (qualifyingProducts) return qualifyingProducts @@ -44,13 +35,10 @@ function getQualifyingProducts(): Set { .readdirSync(productPath, { withFileTypes: true }) .filter((e) => e.isDirectory()) - // Skip products with no subdirectories (e.g. flat products) + // Flat products do not need contentType directory validation. if (subdirs.length === 0) continue - // A product qualifies when ALL of its subdirectories are known - // content-type directories. Use .includes() for responsible-use so - // that variations like "responsible-use-of-…" are recognised, matching - // the logic in dirToContentType(). + // responsible-use-of... variants resolve to rai, matching dirToContentType. const isKnownDir = (name: string) => KNOWN_CONTENT_TYPE_DIRS.has(name) || name.includes(RESPONSIBLE_USE_STRING) if (subdirs.every((sub) => isKnownDir(sub.name))) { @@ -62,7 +50,6 @@ function getQualifyingProducts(): Set { return products } -/** Map a directory name to its expected `contentType` value. */ function dirToContentType(dirName: string): string { if (dirName.includes(RESPONSIBLE_USE_STRING)) return 'rai' if (dirName === GETTING_STARTED_STRING) return 'get-started' @@ -70,10 +57,7 @@ function dirToContentType(dirName: string): string { return 'other' } -/** - * Reset the cached qualifying-products set. Exported so that tests can - * call it between test cases if the filesystem fixture changes. - */ +// Tests reset the cache when filesystem fixtures change. export function resetCache(): void { qualifyingProducts = null } @@ -87,21 +71,14 @@ export const frontmatterContentType = { const filePath = params.name const contentDir = path.resolve(process.env.ROOT || '.', 'content') - // Resolve the params.name relative to the content directory. - // When markdownlint runs with `strings`, params.name is the key - // (often a relative path like "content/copilot/how-tos/file.md"). - // When it runs with `files`, params.name is the real file path. - // Resolve non-absolute paths against ROOT (not CWD) so the rule - // works correctly when process.env.ROOT differs from the working directory. + // Resolve relative params.name against ROOT because strings mode passes content-relative keys. const rootDir = process.env.ROOT || '.' const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(rootDir, filePath) const relativePath = path.relative(contentDir, resolved) - // Skip files that aren't under content/ if (relativePath.startsWith('..')) return const segments = relativePath.split(path.sep) - // Need at least product/something (e.g. copilot/index.md) if (segments.length < 2) return const product = segments[0] @@ -112,19 +89,16 @@ export const frontmatterContentType = { let expectedType: string if (segments.length === 2 && segments[1] === 'index.md') { - // Product-level index.md is always a landing page + // Product index pages require contentType landing. expectedType = 'landing' } else if (segments.length === 2) { - // Non-index files directly under content// are unusual - // in qualifying products — skip them rather than requiring - // contentType: other. + // Skip non-index files under a qualifying product instead of requiring contentType other. return } else { expectedType = dirToContentType(segments[1]) } - // Find the best line number for the error. - // Prefer the contentType line; fall back to the opening `---`. + // Missing contentType has no source line, so report at the frontmatter start. const contentTypeLine = params.lines.findIndex((line) => line.trimStart().startsWith('contentType'), ) diff --git a/src/content-linter/lib/linting-rules/frontmatter-docs-team-metrics.ts b/src/content-linter/lib/linting-rules/frontmatter-docs-team-metrics.ts index 1cd9ddd29d4f..ec22fcd37c07 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-docs-team-metrics.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-docs-team-metrics.ts @@ -28,7 +28,7 @@ export const frontmatterDocsTeamMetrics = { const missingValues = expectedValues.filter((v) => !currentValues.includes(v)) if (missingValues.length === 0) return - // Report on the first line of frontmatter (the opening ---) + // Missing docsTeamMetrics has no source line, so report at the frontmatter start. addError( onError, 1, diff --git a/src/content-linter/lib/linting-rules/frontmatter-hero-image.ts b/src/content-linter/lib/linting-rules/frontmatter-hero-image.ts index 914940f0214e..8517dc006fdf 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-hero-image.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-hero-image.ts @@ -10,7 +10,6 @@ interface Frontmatter { [key: string]: unknown } -// Get the list of valid hero images (without extensions) function getValidHeroImages(): string[] { const ROOT = process.env.ROOT || '.' const heroImageDir = path.join(ROOT, 'assets/images/banner-images') @@ -21,7 +20,6 @@ function getValidHeroImages(): string[] { } const files = fs.readdirSync(heroImageDir) - // Return absolute paths without extensions as they should appear in frontmatter return files.map((file) => { const baseName = path.basename(file, path.extname(file)) return `/assets/images/banner-images/${baseName}` @@ -70,7 +68,6 @@ export const frontmatterHeroImage: Rule = { return } - // Check if the path includes a file extension (which is not allowed) if (path.extname(heroImage)) { const line = params.lines.find((ln: string) => ln.trim().startsWith('heroImage:')) const lineNumber = line ? params.lines.indexOf(line) + 1 : 1 diff --git a/src/content-linter/lib/linting-rules/frontmatter-hidden-docs.ts b/src/content-linter/lib/linting-rules/frontmatter-hidden-docs.ts index e498fa0bd320..c776652fc1f8 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-hidden-docs.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-hidden-docs.ts @@ -12,10 +12,10 @@ export const frontmatterHiddenDocs = { const fm = getFrontmatter(params.lines) if (!fm || !fm.hidden) return - // If the article has an experimental alternative, it's allowed to be hidden + // Allow hidden on articles with experimental alternatives. if (fm.hasExperimentalAlternative) return - // Hidden docs can be located in these content directories: + // Allow hidden in these product paths. const allowedProductPaths = ['content/early-access', 'content/site-policy', 'content/search'] if (allowedProductPaths.some((allowedPath) => params.name.includes(allowedPath))) return diff --git a/src/content-linter/lib/linting-rules/frontmatter-intro-links.ts b/src/content-linter/lib/linting-rules/frontmatter-intro-links.ts index c20c3c78df09..853e06c16a0f 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-intro-links.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-intro-links.ts @@ -9,7 +9,6 @@ interface Frontmatter { [key: string]: unknown } -// Get the valid introLinks keys from ui.yml function getValidIntroLinksKeys(): string[] { try { const ui = getUIDataMerged('en') @@ -38,7 +37,6 @@ export const frontmatterIntroLinks: Rule = { const validKeys = getValidIntroLinksKeys() if (validKeys.length === 0) { - // If we can't load the valid keys, skip validation return } diff --git a/src/content-linter/lib/linting-rules/frontmatter-landing-carousels.ts b/src/content-linter/lib/linting-rules/frontmatter-landing-carousels.ts index 184cc9a715d1..2c1f6ef928e0 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-landing-carousels.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-landing-carousels.ts @@ -11,10 +11,10 @@ interface Frontmatter { [key: string]: unknown } +// Try content-root paths before paths relative to the current file. function isValidArticlePath(articlePath: string, currentFilePath: string): boolean { const ROOT = process.env.ROOT || '.' - // Strategy 1: Always try as an absolute path from content root first const contentDir = path.join(ROOT, 'content') const normalizedPath = articlePath.startsWith('/') ? articlePath.substring(1) : articlePath @@ -29,7 +29,6 @@ function isValidArticlePath(articlePath: string, currentFilePath: string): boole return true } - // Strategy 2: Fall back to relative path from current file's directory const currentDir: string = path.dirname(currentFilePath) const relativePath: string = path.join(currentDir, `${normalizedPath}.md`) @@ -38,7 +37,7 @@ function isValidArticlePath(articlePath: string, currentFilePath: string): boole return true } } catch { - // Continue to next strategy + // Fall through to the relative index lookup when the file check throws. } const relativeIndexPath: string = path.join(currentDir, normalizedPath, 'index.md') @@ -81,7 +80,6 @@ export const frontmatterLandingCarousels = { ) } - // Check each carousel for duplicates and invalid paths for (const [carouselKey, articles] of Object.entries(fm.carousels!)) { if (!Array.isArray(articles)) continue diff --git a/src/content-linter/lib/linting-rules/frontmatter-rest-api-category.ts b/src/content-linter/lib/linting-rules/frontmatter-rest-api-category.ts index 7217281fd827..36db021cedf9 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-rest-api-category.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-rest-api-category.ts @@ -9,9 +9,6 @@ import type { RuleParams, RuleErrorCallback } from '@/content-linter/types' // Lazily computed list of valid category values from content/rest/index.md. let validCategories: string[] | null = null -/** - * Read the `includedCategories` frontmatter from content/rest/index.md. - */ function getValidCategories(): string[] { if (validCategories) return validCategories @@ -22,10 +19,7 @@ function getValidCategories(): string[] { return validCategories } -/** - * Reset the cached valid categories. Exported so tests can call it - * between test cases if the fixture changes. - */ +// Tests reset the cache when filesystem fixtures change. export function resetCache(): void { validCategories = null } @@ -40,15 +34,13 @@ export const frontmatterRestApiCategory = { const rootDir = process.env.ROOT || '.' const restDir = path.resolve(rootDir, 'content/rest') - // Resolve the file path against CWD (markdownlint provides paths - // relative to CWD when using `files`, or string keys when using `strings`). + // Resolve CWD-relative file paths and strings-mode keys before comparing them with content/rest. const resolved = path.isAbsolute(filePath) ? filePath : path.resolve(filePath) const relativePath = path.relative(restDir, resolved) - // Skip files that aren't under content/rest/ if (relativePath.startsWith('..')) return - // Skip index files: they are category-level pages, not endpoint pages. + // Index files are category-level pages, not endpoint pages. if (path.basename(resolved) === 'index.md') return const fm = getFrontmatter(params.lines) @@ -56,7 +48,7 @@ export const frontmatterRestApiCategory = { if (fm.autogenerated !== 'rest') return - // Find error line: prefer the category line, fall back to opening --- + // Missing category has no source line, so report at the frontmatter start. const categoryLineIndex = params.lines.findIndex((line) => line.trimStart().startsWith('category'), ) diff --git a/src/content-linter/lib/linting-rules/frontmatter-schema.ts b/src/content-linter/lib/linting-rules/frontmatter-schema.ts index 2fae6b10c5c5..eb5b9cee884c 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-schema.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-schema.ts @@ -14,13 +14,10 @@ export const frontmatterSchema: Rule = { const fm = getFrontmatter(params.lines) if (!fm) return - // Check that frontmatter does not contain any deprecated keys - // Currently we only deprecate top-level properties and historically - // tests have only checked top-level properties. But, if more properties - // are deprecated in the future, we'll need to do a deep check. + // Deprecated checks cover only top-level frontmatter properties. const deprecatedKeys = intersection(Object.keys(fm), deprecatedProperties) for (const key of deprecatedKeys) { - // Early access articles are allowed to have deprecated properties + // Allow deprecated properties in early access articles. if (params.name.includes('early-access')) continue const line = params.lines.find((ln: string) => ln.trim().startsWith(key)) const lineNumber = params.lines.indexOf(line!) + 1 @@ -34,8 +31,7 @@ export const frontmatterSchema: Rule = { ) } - // Check that the frontmatter matches the schema. - // readFrontmatter returns errors as { property, message, reason } objects. + // readFrontmatter returns property, message, and reason for each schema error. const { errors } = readFrontmatter(params.lines.join('\n'), { schema: frontmatter.schema }) for (const error of errors) { const property = error.property || '' @@ -50,12 +46,11 @@ export const frontmatterSchema: Rule = { if (reason === 'additionalProperties') { detail = 'The frontmatter includes an unsupported property.' context = `Remove the property \`${property}\`.` - // Search for the offending additional property directly searchProperty = parts[parts.length - 1] || '' } else if (reason === 'required') { detail = 'The frontmatter has a missing required property' context = `Add the missing property \`${property}\`` - // The property is missing, so point to its parent container + // Missing properties report on their parent container when one exists. searchProperty = parts.length > 1 ? parts[parts.length - 2] : '' } else { detail = `Frontmatter ${message}.` @@ -63,8 +58,7 @@ export const frontmatterSchema: Rule = { searchProperty = parts[0] || '' } - // If the property is at the top level or missing, we don't have a line - // to point to. In that case, the error will be added to line 1. + // Missing top-level properties have no key to report, so they fall back to the file start. const query = (line: string) => line.trim().startsWith(`${searchProperty}:`) const line = searchProperty === '' ? null : params.lines.find(query) const lineNumber = line ? params.lines.indexOf(line) + 1 : 1 diff --git a/src/content-linter/lib/linting-rules/frontmatter-versions-whitespace.ts b/src/content-linter/lib/linting-rules/frontmatter-versions-whitespace.ts index 005a51d2a612..3518603ea75b 100644 --- a/src/content-linter/lib/linting-rules/frontmatter-versions-whitespace.ts +++ b/src/content-linter/lib/linting-rules/frontmatter-versions-whitespace.ts @@ -55,35 +55,29 @@ export const frontmatterVersionsWhitespace: Rule = { }, } -// Allows whitespace in complex expressions like '<3.6 >3.8' but disallows -// leading and trailing whitespace. +// Complex ranges like '<3.6 >3.8' keep internal spaces. +// Empty or whitespace-only values pass unchanged, and no other value keeps edge spaces. function checkForUnwantedWhitespace(value: string): boolean { - // Don't flag if the value is just whitespace or empty if (!value || value.trim() === '') return false if (value !== value.trim()) return true - // Values containing <, > or = are treated as ranges like '<3.6 >3.8', where - // internal whitespace is meaningful. + // Operators <, >, and = make internal spacing meaningful. const hasOperators = /[<>=]/.test(value) if (hasOperators) { - // Leading and trailing whitespace was already checked above. return false } - // For simple version strings (like 'fpt', 'ghec'), no internal whitespace should be allowed - // This catches cases like 'f pt' where there's whitespace in the middle + // Simple version aliases cannot contain internal spaces such as f pt. return /\s/.test(value) } function getCleanedValue(value: string): string { - // Values containing <, > or = keep their internal whitespace and are only - // trimmed at the ends. + // Range expressions keep internal operator spacing and trim only the ends. const hasOperators = /[<>=]/.test(value) if (hasOperators) { return value.trim() } - // For simple version strings, remove all whitespace return value.replace(/\s/g, '') } diff --git a/src/content-linter/lib/linting-rules/github-owned-action-references.ts b/src/content-linter/lib/linting-rules/github-owned-action-references.ts index f02d9c58e646..f91ac2fb337b 100644 --- a/src/content-linter/lib/linting-rules/github-owned-action-references.ts +++ b/src/content-linter/lib/linting-rules/github-owned-action-references.ts @@ -2,11 +2,6 @@ import { addError, ellipsify } from 'markdownlint-rule-helpers' import type { RuleParams, RuleErrorCallback } from '../../types' import { getRange } from '../helpers/utils' -/* - This rule currently only checks for one hardcoded string but - can be generalized in the future to check for strings that - have data reusables. -*/ export const githubOwnedActionReferences = { names: ['GHD013', 'github-owned-action-references'], description: 'GitHub-owned action references should not be hardcoded', diff --git a/src/content-linter/lib/linting-rules/hardcoded-data-variable.ts b/src/content-linter/lib/linting-rules/hardcoded-data-variable.ts index a6d74dc22299..9ab40987d406 100644 --- a/src/content-linter/lib/linting-rules/hardcoded-data-variable.ts +++ b/src/content-linter/lib/linting-rules/hardcoded-data-variable.ts @@ -4,11 +4,6 @@ import type { RuleParams, RuleErrorCallback } from '../../types' import { getRange } from '../helpers/utils' import frontmatter from '@/frame/lib/read-frontmatter' -/* - This rule currently only checks for one hardcoded string but - can be generalized in the future to check for strings that - have data variables. -*/ export const hardcodedDataVariable = { names: ['GHD005', 'hardcoded-data-variable'], description: diff --git a/src/content-linter/lib/linting-rules/image-alt-text-end-punctuation.ts b/src/content-linter/lib/linting-rules/image-alt-text-end-punctuation.ts index d00b8819969c..4e565110605c 100644 --- a/src/content-linter/lib/linting-rules/image-alt-text-end-punctuation.ts +++ b/src/content-linter/lib/linting-rules/image-alt-text-end-punctuation.ts @@ -16,10 +16,7 @@ export const imageAltTextEndPunctuation: Rule = { forEachInlineChild(params, 'image', function forToken(token: MarkdownToken) { const imageAltText = token.content?.trim() - // If the alt text is empty, there is nothing to check and you can't - // produce a valid range. - // We can safely return early because the image-alt-text-length rule - // will fail this one. + // Empty alt text belongs to image-alt-text-length and cannot produce a range. if (!imageAltText) return if (isStringPunctuated(imageAltText)) return diff --git a/src/content-linter/lib/linting-rules/image-alt-text-exclude-start-words.ts b/src/content-linter/lib/linting-rules/image-alt-text-exclude-start-words.ts index e8796520b4f1..b4ecea14d1a4 100644 --- a/src/content-linter/lib/linting-rules/image-alt-text-exclude-start-words.ts +++ b/src/content-linter/lib/linting-rules/image-alt-text-exclude-start-words.ts @@ -5,10 +5,6 @@ import type { RuleParams, RuleErrorCallback, MarkdownToken, Rule } from '../../t const excludeStartWords = ['image', 'graphic'] -/* - Images should have meaningful alternative text (alt text) - and should not begin with words like "image" or "graphic". - */ export const imageAltTextExcludeStartWords: Rule = { names: ['GHD031', 'image-alt-text-exclude-words'], description: 'Alternate text for images should not begin with words like "image" or "graphic"', @@ -16,10 +12,7 @@ export const imageAltTextExcludeStartWords: Rule = { parser: 'markdownit', function: (params: RuleParams, onError: RuleErrorCallback) => { forEachInlineChild(params, 'image', function forToken(token: MarkdownToken) { - // If the alt text is empty, there is nothing to check and you can't - // produce a valid range. - // We can safely return early because the image-alt-text-length rule - // will fail this one. + // Empty alt text belongs to image-alt-text-length and cannot produce a range. if (!token.content) return const imageAltText = token.content.trim() diff --git a/src/content-linter/lib/linting-rules/image-alt-text-length.ts b/src/content-linter/lib/linting-rules/image-alt-text-length.ts index a1ae7b87d134..8c4477251600 100644 --- a/src/content-linter/lib/linting-rules/image-alt-text-length.ts +++ b/src/content-linter/lib/linting-rules/image-alt-text-length.ts @@ -30,9 +30,7 @@ export const incorrectAltTextLength = { renderedString = await liquid.parseAndRender(token.content, context) } - // You can't compute a range if the content is empty - // because getRange() would throw an error. It's because it assumes to - // be able to find one string in another string. + // Empty alt text cannot produce a range because getRange requires a string it can find. const range = token.content ? getRange(token.line, token.content) : null if (renderedString.length < 40 || renderedString.length > 150) { diff --git a/src/content-linter/lib/linting-rules/index.ts b/src/content-linter/lib/linting-rules/index.ts index 0a8c51517bef..123118687ca8 100644 --- a/src/content-linter/lib/linting-rules/index.ts +++ b/src/content-linter/lib/linting-rules/index.ts @@ -68,11 +68,11 @@ const noGenericLinkText = markdownlintGitHub.find((elem: { names: string[] }) => export const gitHubDocsMarkdownlint = { rules: [ - // GH rules (markdownlint-github) + // markdownlint-github supplies the GH rule IDs. noDefaultAltText, // GH001 noGenericLinkText, // GH002 - // GHD rules (GitHub Docs custom rules, in numerical order) + // Keep custom GitHub Docs rules in numerical order by GHD ID. linkPunctuation, // GHD001 internalLinksNoLang, // GHD002 internalLinksSlash, // GHD003 @@ -106,7 +106,7 @@ export const gitHubDocsMarkdownlint = { thirdPartyActionPinning, // GHD041 liquidTagWhitespace, // GHD042 linkQuotation, // GHD043 - // GHD044 removed: octicon aria-labels are now auto-generated. + // GHD044 is absent because octicon aria-labels are auto-generated. codeAnnotationCommentSpacing, // GHD045 outdatedReleasePhaseTerminology, // GHD046 tableColumnIntegrity, // GHD047 @@ -125,7 +125,7 @@ export const gitHubDocsMarkdownlint = { frontmatterDocsTeamMetrics, // GHD066 frontmatterRestApiCategory, // GHD067 - // Search-replace rules + // markdownlint-rule-search-replace supplies this rule. searchReplace, // Open-source plugin ], } diff --git a/src/content-linter/lib/linting-rules/internal-links-no-lang.ts b/src/content-linter/lib/linting-rules/internal-links-no-lang.ts index 6334f6c6b5e1..c9d3b0baaa1a 100644 --- a/src/content-linter/lib/linting-rules/internal-links-no-lang.ts +++ b/src/content-linter/lib/linting-rules/internal-links-no-lang.ts @@ -14,13 +14,8 @@ export const internalLinksNoLang: Rule = { for (const child of token.children!) { if (child.type !== 'link_open') continue - // Example child.attrs: - // [ - // ['href', 'get-started'], ['target', '_blank'], - // ['rel', 'canonical'], - // ] const hrefsWithLanguageCode = child - // The attribute could also be `target` or `rel`. + // markdown-it attrs also include target and rel, so filter for href. .attrs!.filter((attr: [string, string]) => attr[0] === 'href') .filter((attr: [string, string]) => attr[1].startsWith('/') || !attr[1].startsWith('//')) .filter((attr: [string, string]) => diff --git a/src/content-linter/lib/linting-rules/internal-links-old-version.ts b/src/content-linter/lib/linting-rules/internal-links-old-version.ts index e47b24bf3845..7c77e685892d 100644 --- a/src/content-linter/lib/linting-rules/internal-links-old-version.ts +++ b/src/content-linter/lib/linting-rules/internal-links-old-version.ts @@ -3,6 +3,11 @@ import { addError, filterTokens } from 'markdownlint-rule-helpers' import { getRange } from '../helpers/utils' import type { RuleParams, RuleErrorCallback, MarkdownToken, Rule } from '../../types' +// Match hardcoded Enterprise version paths on legacy Docs hosts and root-relative links. +// Match /enterprise/2.19/admin/blah, https://docs.github.com/enterprise/11.10.340/admin/blah, +// and http://help.github.com/enterprise/2.8/admin/blah. +// Ignore https://someservice.com/enterprise/1.0/blah +// and /github/site-policy/enterprise/2.2/admin/blah. export const internalLinksOldVersion: Rule = { names: ['GHD006', 'internal-links-old-version'], description: 'Internal links must not have a hardcoded version using old versioning syntax', @@ -18,23 +23,10 @@ export const internalLinksOldVersion: Rule = { for (const child of token.children || []) { if (child.type !== 'link_open') continue if (!child.attrs) continue - // Things matched by this RegExp: - // - /enterprise/2.19/admin/blah - // - https://docs.github.com/enterprise/11.10.340/admin/blah - // - http://help.github.com/enterprise/2.8/admin/blah - // - // Things intentionally NOT matched by this RegExp: - // - https://someservice.com/enterprise/1.0/blah - // - /github/site-policy/enterprise/2.2/admin/blah const versionLinkRegEx = /(?:(?:https?:\/\/(?:help|docs|developer)\.github\.com)(?:\/enterprise\/\d+(\.\d+)+\/[^)\s]*)?|^\/enterprise\/\d+(\.\d+)+\/[^)\s]*)(?=\s|$)/gm - // Example child.attrs: - // [ - // ['href', 'get-started'], ['target', '_blank'], - // ['rel', 'canonical'], - // ] const hrefsWithHardcodedVersion = child.attrs - // The attribute could also be `target` or `rel` + // markdown-it attrs also include target and rel, so filter for href. .filter((attr) => attr[0] === 'href') .filter((attr) => attr[1].startsWith('/') || !attr[1].startsWith('//')) .filter((attr) => attr[1].match(versionLinkRegEx)) diff --git a/src/content-linter/lib/linting-rules/internal-links-slash.ts b/src/content-linter/lib/linting-rules/internal-links-slash.ts index cba78378ebcc..04f6c8723876 100644 --- a/src/content-linter/lib/linting-rules/internal-links-slash.ts +++ b/src/content-linter/lib/linting-rules/internal-links-slash.ts @@ -14,23 +14,17 @@ export const internalLinksSlash: Rule = { for (const child of token.children) { if (child.type !== 'link_open') continue - // Example child.attrs: - // [ - // ['href', '/get-started'], ['target', '_blank'], - // ['rel', 'canonical'], - // ] if (!child.attrs) continue const hrefsMissingSlashes = child.attrs - // The attribute could also be `target` or `rel` + // markdown-it attrs also include target and rel, so filter for href. .filter((attr: [string, string]) => attr[0] === 'href') - // Filter out prefixes we don't want to check .filter( (attr: [string, string]) => !['http', 'mailto', '#', '/'].some((ignorePrefix) => attr[1].startsWith(ignorePrefix), ), ) - // We can ignore empty links because MD042 from markdownlint catches empty links + // MD042 from markdownlint catches empty links. .filter((attr: [string, string]) => attr[1] !== '') .map((attr: [string, string]) => attr[1]) diff --git a/src/content-linter/lib/linting-rules/journey-tracks-guide-path-exists.ts b/src/content-linter/lib/linting-rules/journey-tracks-guide-path-exists.ts index 05e354e416e6..44c770e3f382 100644 --- a/src/content-linter/lib/linting-rules/journey-tracks-guide-path-exists.ts +++ b/src/content-linter/lib/linting-rules/journey-tracks-guide-path-exists.ts @@ -5,12 +5,10 @@ import { addError } from 'markdownlint-rule-helpers' import { getFrontmatter } from '../helpers/utils' import type { RuleParams, RuleErrorCallback } from '@/content-linter/types' -// Same two-strategy path resolution as isValidArticlePath in -// frontmatter-landing-carousels.ts. +// Match frontmatter-landing-carousels.ts: content-root lookup, then relative lookup. function isValidGuidePath(guidePath: string, currentFilePath: string): boolean { const ROOT = process.env.ROOT || '.' - // Strategy 1: Always try as an absolute path from content root first const contentDir = path.join(ROOT, 'content') const normalizedPath = guidePath.startsWith('/') ? guidePath.substring(1) : guidePath @@ -25,7 +23,6 @@ function isValidGuidePath(guidePath: string, currentFilePath: string): boolean { return true } - // Strategy 2: Fall back to relative path from current file's directory const currentDir = path.dirname(currentFilePath) const relativePath = path.join(currentDir, `${normalizedPath}.md`) @@ -34,7 +31,7 @@ function isValidGuidePath(guidePath: string, currentFilePath: string): boolean { return true } } catch { - // Continue to next strategy + // Fall through to the relative index lookup when the file check throws. } const relativeIndexPath = path.join(currentDir, normalizedPath, 'index.md') @@ -50,7 +47,7 @@ export const journeyTracksGuidePathExists = { description: 'Journey track guide paths must reference existing content files', tags: ['frontmatter', 'journey-tracks'], function: (params: RuleParams, onError: RuleErrorCallback) => { - // Using unknown for frontmatter as it's a dynamic YAML object with varying properties + // Frontmatter is a dynamic YAML object, so narrow it before reading journeyTracks. const fm: unknown = getFrontmatter(params.lines) if (!fm || typeof fm !== 'object' || !('journeyTracks' in fm)) return const fmObj = fm as Record diff --git a/src/content-linter/lib/linting-rules/journey-tracks-liquid.ts b/src/content-linter/lib/linting-rules/journey-tracks-liquid.ts index 2a815f3b9856..23cc1552025b 100644 --- a/src/content-linter/lib/linting-rules/journey-tracks-liquid.ts +++ b/src/content-linter/lib/linting-rules/journey-tracks-liquid.ts @@ -22,17 +22,14 @@ export const journeyTracksLiquid = { for (let trackIndex = 0; trackIndex < fm.journeyTracks.length; trackIndex++) { const track = (fm.journeyTracks as Array>)[trackIndex] - // Try to find the line number for this specific journey track so we can use that for the error - // line number. Getting the exact line number is probably more work than it's worth for this - // particular rule. + // Approximate the track location instead of parsing every nested frontmatter node. - // Look for the track by finding the nth occurrence of track-like patterns after journeyTracks let trackLineNumber: number = baseLineNumber if (journeyTracksLine) { let trackCount: number = 0 for (let i = params.lines.indexOf(journeyTracksLine) + 1; i < params.lines.length; i++) { const line: string = params.lines[i].trim() - // Look for track indicators (array item with id, title, or description) + // Track entries can start with id, title, or a bare array marker followed by id or title. if ( line.startsWith('- id:') || line.startsWith('- title:') || @@ -50,7 +47,7 @@ export const journeyTracksLiquid = { } } - // The only check is that Liquid can parse each string property. + // Liquid parsing is the rule's only validation. const properties = [ { name: 'title', value: track.title }, { name: 'description', value: track.description }, diff --git a/src/content-linter/lib/linting-rules/journey-tracks-unique-ids.ts b/src/content-linter/lib/linting-rules/journey-tracks-unique-ids.ts index 44b496f37b03..fd7889de3dbd 100644 --- a/src/content-linter/lib/linting-rules/journey-tracks-unique-ids.ts +++ b/src/content-linter/lib/linting-rules/journey-tracks-unique-ids.ts @@ -33,7 +33,6 @@ export const journeyTracksUniqueIds = { } trackCount++ - // Stop once we've found all the tracks we know exist if (Array.isArray(fmObj.journeyTracks) && trackCount >= fmObj.journeyTracks.length) { break } @@ -42,7 +41,6 @@ export const journeyTracksUniqueIds = { return baseLineNumber } - // Track seen journey track IDs and line number for error reporting const seenIds = new Map() for (let index = 0; index < fmObj.journeyTracks.length; index++) { diff --git a/src/content-linter/lib/linting-rules/link-punctuation.ts b/src/content-linter/lib/linting-rules/link-punctuation.ts index afeecb5656c8..e3b883da92b5 100644 --- a/src/content-linter/lib/linting-rules/link-punctuation.ts +++ b/src/content-linter/lib/linting-rules/link-punctuation.ts @@ -3,7 +3,6 @@ import type { RuleParams, RuleErrorCallback, Rule } from '../../types' import { doesStringEndWithPeriod, getRange, isStringQuoted } from '../helpers/utils' -// Minimal type for markdownit tokens used in this rule interface MarkdownToken { children?: MarkdownToken[] line?: string diff --git a/src/content-linter/lib/linting-rules/liquid-data-tags.ts b/src/content-linter/lib/linting-rules/liquid-data-tags.ts index d0809801c502..de5ec7a6e220 100644 --- a/src/content-linter/lib/linting-rules/liquid-data-tags.ts +++ b/src/content-linter/lib/linting-rules/liquid-data-tags.ts @@ -12,10 +12,6 @@ import { import type { RuleParams, RuleErrorCallback } from '@/content-linter/types' -/* - Checks for instances where a Liquid data or indented_data_reference - tag is used but is not defined. -*/ export const liquidDataReferencesDefined = { names: ['GHD014', 'liquid-data-references-defined'], description: @@ -31,10 +27,7 @@ export const liquidDataReferencesDefined = { if (!tokens.length) return for (const token of tokens) { - // When the liquid tag is indented_data_reference, there are - // two arguments: the path in the data directory and the number of - // spaces to indent. We only want the first argument to - // validate if the data reference is defined. + // indented_data_reference has path and indentation args; validate only the data path. const dataDirectoryReference = token.args.split(/\s+/)[0] if (hasData(dataDirectoryReference)) continue @@ -67,17 +60,12 @@ export const liquidDataTagFormat = { const indentedDataTags = tokenTags.filter((token) => token.name === 'indented_data_reference') for (const token of dataTags) { - // A data tag has only one argument, the data directory path. + // A data tag accepts only the data directory path. const args = token.args.split(/\s+/) - // When the string is empty and a non-empty separator is specified, - // split() returns [''], so we need to check for that case. + // split() returns [''] for an empty string with a non-empty separator. if (args.length === 1 && token.args !== '') continue - // When we filter out the data tokens from getLiquidTokens, we are left with the data content itself - // without the liquid opening/closing tags. If we see that it is in the args of the token, we can - // assume that the data tag is not formatted correctly. - // This is not necessary as the liquid tests will later catch badly formatted liquid, but badly - // formatted data tags prevents getting the correct position data for the test below. + // Bad data tags can leave tag delimiters inside args, which breaks position data. const containsBadLiquidDataTags = CHECK_LIQUID_TAGS.some((tag) => token.args.includes(tag)) if (containsBadLiquidDataTags) { @@ -107,9 +95,7 @@ export const liquidDataTagFormat = { } for (const token of indentedDataTags) { - // When the liquid tag is indented_data_reference, there are - // two arguments: the path in the data directory and the number - // of spaces to indent. + // indented_data_reference requires a path and a spaces argument. const args = token.args.split(/\s+/) const isSpacesArgOk = /^spaces=\d{1,2}$/.test(args[1]) if (args.length === 2 && isSpacesArgOk) continue @@ -129,14 +115,12 @@ export const liquidDataTagFormat = { }, } -// Convenient wrapper because linting is always about English content +// Linting always checks English content. const getData = (liquidRef: string) => getDataByLanguage(liquidRef, 'en') const hasData = (liquidRef: string): boolean => { try { - // If a reusable contains a nonexistent data reference, it will - // return undefined. If the data reference is inherently broken - // (e.g., {% data reus.foo %}), it will throw an error. + // Missing refs return undefined; malformed refs such as {% data reus.foo %} throw. const data = getData(liquidRef) return data !== undefined } catch { diff --git a/src/content-linter/lib/linting-rules/liquid-ifversion-versions.ts b/src/content-linter/lib/linting-rules/liquid-ifversion-versions.ts index c0ee0a0bb467..b1471c314e3c 100644 --- a/src/content-linter/lib/linting-rules/liquid-ifversion-versions.ts +++ b/src/content-linter/lib/linting-rules/liquid-ifversion-versions.ts @@ -21,9 +21,7 @@ import { import { oldestSupported } from '@/versions/lib/enterprise-server-releases' import type { RuleParams, RuleErrorCallback } from '@/content-linter/types' -// A liquidjs token, as exposed by getLiquidIfVersionTokens. liquidjs's TopLevelToken -// type does not declare all of the runtime properties we rely on (begin/end, content, -// contentRange, name), so we narrow it here. +// getLiquidIfVersionTokens exposes runtime properties that liquidjs TopLevelToken omits. type LiquidConditionalToken = TopLevelToken & { name: string content: string @@ -32,8 +30,7 @@ type LiquidConditionalToken = TopLevelToken & { contentRange: [number, number] } -// Frontmatter `versions` declaration. May be a wildcard string ("*") or a record -// keyed by short version names (fpt, ghec, ghes, feature, ...) with semver-range values. +// Frontmatter versions can be a wildcard string or a short-name map with semver ranges. type VersionsObject = Record type FileVersionsFm = string | VersionsObject | undefined @@ -48,11 +45,8 @@ type CondTagAction = { content?: unknown } -// Internal representation of an ifversion/elsif/else/endif tag that flows through -// the rule. fileVersionsFmAll, versionsObj, featureVersionsObj, versionsObjAll, and -// versions are populated for ifversion/elsif tags and may be absent on else/endif. -// `action` is always populated by decorateCondTagItems before setLiquidErrors and -// updateConditionals run. +// CondTagItem carries derived version data between decoration, updates, and error reporting. +// Condition version objects stay empty on else and endif entries; else gets leftover versions. type CondTagItem = { name: string cond: string @@ -68,7 +62,7 @@ type CondTagItem = { versionsObjAll: VersionsObject versions: string[] action: CondTagAction - // Cached error range (used by addError); never set by this rule but accepted by addError. + // addError accepts this cached range, but this rule never sets it. contentRange?: [number, number] | number[] | string | null } @@ -86,8 +80,7 @@ export const liquidIfversionVersions = { tags: ['liquid', 'versioning'], asynchronous: true, function: async (params: RuleParams, onError: RuleErrorCallback) => { - // The versions frontmatter object or all versions if the file - // being processed is a data file. + // Data files read all product versions instead of page frontmatter versions. const fm = getFrontmatter(params.lines) const content = fm ? getFrontmatterLines(params.lines).join('\n') : params.lines.join('\n') @@ -97,20 +90,17 @@ export const liquidIfversionVersions = { ? (fm.versions as FileVersionsFm) : (getFrontmatter(params.frontMatterLines)?.versions as FileVersionsFm) if (!fileVersionsFm) return - // This will only contain valid (non-deprecated) and future versions + // getApplicableVersions includes supported and upcoming versions, but not deprecated ones. const fileVersions = getApplicableVersions(fileVersionsFm, '', { doNotThrow: true, includeNextVersion: true, }) const tokens = getLiquidIfVersionTokens(content) as LiquidConditionalToken[] - // Array of arrays - each array entry is an array of items that - // make up a full if/elsif/else/endif statement. - // [ [ifversion, elsif, else, endif], [nested ifversion, elsif, else, endif] ] + // Each stack entry holds one ifversion, elsif, else, and endif chain. const condStmtStack: CondTagItem[][] = [] - // Tokens are in the order they are read in file, so we need to iterate - // through and group full if/elsif/else/endif statements together. + // Build each conditional chain from source order before decorating its actions. const defaultProps: DefaultProps = { fileVersionsFm, fileVersions, @@ -135,8 +125,7 @@ export const liquidIfversionVersions = { } else if (token.name === 'else') { const condTagItems = condStmtStack.pop()! const condTagItem = await initTagObject(token, defaultProps) - // The versions of an else tag are the set of file versions that are - // not supported by the previous ifversion or elsif tags. + // else covers file versions excluded by previous ifversion and elsif tags. const siblingVersions = condTagItems .filter((item) => item.name === 'ifversion' || item.name === 'elsif') .map((item) => item.versions) @@ -163,7 +152,7 @@ function setLiquidErrors(condTagItems: CondTagItem[], onError: RuleErrorCallback const itemErrorName = tagNameNoCond ? item.name : `${item.name} ${item.cond}` if (item.action?.type === 'delete') { - // There is no next stack item; the endif tag is always last in a conditional. + // endif deletes through its own end because no following stack item exists. const nextStackItem = item.name === 'endif' ? condTagItems[i].end : condTagItems[i + 1].begin const deleteItems = getContentDeleteData( condTagItems[i] as unknown as TopLevelToken, @@ -188,7 +177,7 @@ function setLiquidErrors(condTagItems: CondTagItem[], onError: RuleErrorCallback } if (item.action?.type === 'all') { - // Position is just the tag. + // The all action removes only the tag. const { lineNumber, column, length } = getPositionData( { begin: item.begin, @@ -213,7 +202,7 @@ function setLiquidErrors(condTagItems: CondTagItem[], onError: RuleErrorCallback } if (item.action?.type === 'change') { - // Position is just the inside of the tag. + // The change action replaces only the tag contents. const { lineNumber, column, length } = getPositionData( { begin: item.contentrange[0], @@ -245,24 +234,19 @@ async function getApplicableVersionFromLiquidTag(conditionStr: string): Promise< const condition = conditionStr.replace('not ', '') const liquidTagVersions = condition.split(' or ').map((item) => item.trim()) for (const ver of liquidTagVersions) { - // When the version is not a release, e.g. fpt or ghec, or is a - // feature version. + // Bare product and feature names, such as fpt or ghec, map directly to frontmatter versions. if (ver.split(' ').length === 1) { - // handle feature versions (only supports a single feature version) + // Frontmatter represents one feature version at a time. if (ver !== 'fpt' && ver !== 'ghec' && ver !== 'ghes') { newConditionObject['feature'] = ver } else { newConditionObject[ver] = '*' } } else if (ver.includes(' and ')) { - // When the version is a release e.g. ghes and the version is a range - // e.g. ghes >= 3.1 and ghes < 3.4 + // Compound GHES ranges such as ghes >= 3.1 and ghes < 3.4 collapse into one range string. const ands = ver.split(' and ') const firstAnd = ands[0].split(' ')[0] - // if all ands don't start with the same version it's invalid - // Note: This edge case (e.g., "fpt and ghes >= 3.1") doesn't occur in our content. - // All actual uses have matching versions (e.g., "ghes and ghes > 3.19"). - // If this edge case appears in the future, additional logic would be needed here. + // This rule only handles and conditions where every clause starts with the same product. if (!ands.every((and) => and.startsWith(firstAnd))) { return {} } @@ -276,7 +260,7 @@ async function getApplicableVersionFromLiquidTag(conditionStr: string): Promise< const andVersionFmString = andValues.join(' ') newConditionObject[andVersion] = andVersionFmString } else { - // When the version is a release e.g. ghes >= 3.1 + // Single GHES ranges such as ghes >= 3.1 map to the frontmatter range string. const [version, ...release] = ver.split(' ') const versionFmString = release.join(' ').replaceAll("'", '') newConditionObject[version] = versionFmString @@ -298,10 +282,7 @@ async function initTagObject( props: DefaultProps, ): Promise { const fileVersionsFm = props.fileVersionsFm - // Normalize a wildcard string ('*') frontmatter `versions` value into the - // canonical all-versions object so downstream consumers (Object.keys, ghes / - // feature lookups) behave consistently. In practice no content file uses the - // string form today, but handling it keeps the rule type-safe and future-proof. + // Normalize wildcard frontmatter so Object.keys, GHES, and feature lookups read one shape. const fmObject: VersionsObject = typeof fileVersionsFm === 'string' ? { ghec: '*', ghes: '*', fpt: '*' } @@ -366,19 +347,11 @@ function decorateCondTagItems(condTagItems: CondTagItem[]) { } function updateConditionals(condTagItems: CondTagItem[]) { - // iterate through the ifversion, elsif, and else - // tags but NOT the endif tag. endif tags have - // no versions associated with them and are handled - // after the loop. + // Skip endif during action updates because endif has no versions. for (let i = 0; i < condTagItems.length - 1; i++) { const item = condTagItems[i] - // check if the condition is all versions, if so - // the liquid should always be removed regardless - // of whether it's a feature version or a nested - // condition. - // `item.versionObj` (no `s`) is not a property of CondTagItem, so the - // fallback is always undefined and could be dropped. + // Collapse feature conditions that cover all versions. if ( isAllVersions( item.featureVersionsObj || @@ -389,15 +362,9 @@ function updateConditionals(condTagItems: CondTagItem[]) { break } - // START feature versions - - // Feature versions that have all versions were removed above - // Deprecatable features are those that are either available - // in NO supported GHES releases or are available in ALL - // supported GHES releases. + // Deprecatable features must be absent from every supported GHES release or present in all. if (item.versionsObj?.feature && item.versionsObjAll?.ghes) { - // Checks for features that are only available in all - // supported GHES releases + // A feature available in every supported GHES release can collapse into the parent condition. if ( Object.keys(item.fileVersionsFmAll).length === 1 && item.fileVersionsFmAll.ghes === '*' && @@ -409,8 +376,7 @@ function updateConditionals(condTagItems: CondTagItem[]) { processConditionals(item, condTagItems, i) break } - // Checks for features that are only available in no - // supported GHES releases + // Delete a feature absent from every supported GHES release. if (isGhesReleaseDeprecated(oldestSupported, item.versionsObjAll.ghes)) { item.action.type = 'delete' continue @@ -420,10 +386,7 @@ function updateConditionals(condTagItems: CondTagItem[]) { item.fileVersionsFm && typeof item.fileVersionsFm === 'object' ? item.fileVersionsFm : {} if (item.versionsObj?.feature || fileVersionsFmObject.feature) break - // When the parent of a nested condition is a feature - // we don't want to assume that the feature versions - // won't change in the future. So we ignore checking - // nested conditions against their feature parent. + // Skip nested conditions under feature parents because feature versions can change. if ( item.parent && item.parent.versions && @@ -433,10 +396,7 @@ function updateConditionals(condTagItems: CondTagItem[]) { ) continue - // END feature versions - - // Check if a nested condition has all versions - // compared to it's parent. + // A nested condition that covers every parent version can collapse into the parent. if ( item.parent && item.parent.versions && @@ -447,24 +407,23 @@ function updateConditionals(condTagItems: CondTagItem[]) { break } - // Check if the condition matches the page frontmatter + // A condition matching page frontmatter applies to every rendered version for this file. const noDiffInFileVersions = difference(item.fileVersions, item.versions).length === 0 if (noDiffInFileVersions) { processConditionals(item, condTagItems, i) break } - // Only an item with ghes versioning only can be deleted + // Delete tags whose version set leaves no rendered versions. if (item.versions.length === 0) { item.action.type = 'delete' continue } - // If the else condition hasn't already been marked as available in all - // versions or as a delete, then there are no other changes possible. + // For unchanged else conditions, no other changes can apply. if (item.name === 'else') continue - // Does the condition contain any versions not defined in the frontmatter + // Remove condition products that the page frontmatter does not define. const versionsNotInFrontmatter = difference( Object.keys(item.versionsObjAll), Object.keys(item.fileVersionsFmAll), @@ -478,13 +437,11 @@ function updateConditionals(condTagItems: CondTagItem[]) { continue } - // All remaining changes only apply if the ghes release number - // must be updated + // Remaining changes only simplify GHES release ranges. if (!item.versionsObjAll.ghes || item.versionsObjAll.ghes === '*') continue const simplifiedSemver = getSimplifiedSemverRange(item.versionsObjAll.ghes) - // Change - Remove the GHES version but keep the other versions in - // the conditional + // Drop GHES from the conditional when other products still apply. if (simplifiedSemver === '' && Object.keys(item.versionsObj).length > 1) { item.action.type = 'change' delete item.versionsObj.ghes @@ -492,13 +449,12 @@ function updateConditionals(condTagItems: CondTagItem[]) { continue } - // Change - Update the GHES semver range + // Replace changed GHES ranges with simplified semver. if (item.versionsObjAll.ghes !== simplifiedSemver && !item.versionsObjAll.feature) { item.action.type = 'change' item.versionsObj.ghes = simplifiedSemver - // Create the new cond by translating the semver to the format - // used in frontmatter + // Translate the simplified range back to Liquid condition syntax. if (simplifiedSemver !== '*') { const newVersions = Object.entries(item.versionsObj).map(([key, value]) => { if (key === 'ghes') { @@ -513,8 +469,7 @@ function updateConditionals(condTagItems: CondTagItem[]) { } } - // Delete - When the ifversion tag is deleted and an elsif tag exists, - // the elsif tag name must be changed to ifversion. + // If the deleted ifversion has a surviving elsif, promote the elsif to ifversion. if (condTagItems[0].action.type === 'delete') { const elsifVersionIndex = condTagItems.findIndex( (item) => item.name === 'elsif' && item.action.type !== 'delete', @@ -527,8 +482,7 @@ function updateConditionals(condTagItems: CondTagItem[]) { } } - // Delete - If an ifversion/else and the ifversion is deleted, change - // the else tag to all. + // If the deleted ifversion leaves a full-coverage else, remove the else wrapper too. if ( condTagItems.length - 1 === 2 && condTagItems[0].action.type === 'delete' && @@ -538,7 +492,7 @@ function updateConditionals(condTagItems: CondTagItem[]) { condTagItems[2].action.type = 'delete' } - // Delete - If all items except endif are marked for delete, delete the endif + // Delete endif when every condition body in the chain is deleted. const isAllDelete = condTagItems .slice(0, condTagItems.length - 1) .every((item) => item.action.type === 'delete') @@ -553,8 +507,7 @@ function processConditionals( indexOfAllItem: number, ) { item.action.type = 'all' - // if any tag in a statement is 'all', the - // remaining tags are obsolete. + // When any tag covers all versions, every other tag in the statement is obsolete. for (let i = 0; i < condTagItems.length; i++) { const stackItem = condTagItems[i] if (indexOfAllItem !== i) { diff --git a/src/content-linter/lib/linting-rules/liquid-quoted-conditional-arg.ts b/src/content-linter/lib/linting-rules/liquid-quoted-conditional-arg.ts index 8c06d732976c..6ca6ac8c787c 100644 --- a/src/content-linter/lib/linting-rules/liquid-quoted-conditional-arg.ts +++ b/src/content-linter/lib/linting-rules/liquid-quoted-conditional-arg.ts @@ -6,17 +6,7 @@ import { getLiquidTokens, conditionalTags, getPositionData } from '../helpers/li import { isStringQuoted } from '../helpers/utils' import type { RuleParams, RuleErrorCallback, Rule } from '../../types' -/* - Checks for instances where a Liquid conditional tag's argument is - quoted because it will always evaluate to true. - - For example, the following would be flagged: - {% if "foo" %} - {% ifversion "bar" %} - - Quoted strings used as operands in comparisons are valid and not flagged: - {% if entry.provider == "openai" %} -*/ +// Quoted Liquid conditional arguments always evaluate to true; comparison operands can stay quoted. const comparisonOperators = new Set(['==', '!=', '<>', '<', '>', '<=', '>=', 'contains']) @@ -34,7 +24,7 @@ export const liquidQuotedConditionalArg: Rule = { if ( tokensArray.some((arg, index) => { if (!isStringQuoted(arg)) return false - // A quoted string is valid as an operand of a comparison operator + // A quoted string is valid as an operand of a comparison operator. const prev = index > 0 ? tokensArray[index - 1] : '' const next = index < tokensArray.length - 1 ? tokensArray[index + 1] : '' if (comparisonOperators.has(prev) || comparisonOperators.has(next)) return false @@ -50,9 +40,8 @@ export const liquidQuotedConditionalArg: Rule = { for (const token of tokens) { const lines = params.lines const { lineNumber, column, length } = getPositionData(token, lines) - // LineNumber starts at 1, but lines is 0-based + // lineNumber starts at 1, but lines indexes from 0. const line = lines[lineNumber - 1].slice(column - 1, column + length) - // Trim the first and last character off of the token args const replaceWith = token.args.slice(1, token.args.length - 1) const replaceString = line.replace(token.args, replaceWith) diff --git a/src/content-linter/lib/linting-rules/liquid-syntax.ts b/src/content-linter/lib/linting-rules/liquid-syntax.ts index 741d0ad874e0..40d80326b6a4 100644 --- a/src/content-linter/lib/linting-rules/liquid-syntax.ts +++ b/src/content-linter/lib/linting-rules/liquid-syntax.ts @@ -12,10 +12,6 @@ interface ErrorMessageInfo { columnNumber: number } -/* - Attempts to parse all liquid in the frontmatter of a file - to verify the syntax is correct. -*/ export const frontmatterLiquidSyntax = { names: ['GHD017', 'frontmatter-liquid-syntax'], description: 'Frontmatter properties must use valid Liquid', @@ -24,9 +20,6 @@ export const frontmatterLiquidSyntax = { const fm = getFrontmatter(params.lines) if (!fm) return - // Currently this list is hardcoded, but in the future we plan to - // use a custom key in the frontmatter to determine which keys - // contain Liquid. const keysWithLiquid = ['title', 'shortTitle', 'intro', 'product', 'permissions'].filter( (key) => Boolean(fm[key]), ) @@ -37,16 +30,13 @@ export const frontmatterLiquidSyntax = { try { liquid.parse(value) } catch (error) { - // If the error source is not a Liquid error but rather a - // ReferenceError or bad type we should allow that error to be thrown + // Let non-Liquid parser errors propagate. if (!isLiquidError(error)) throw error const { errorDescription, columnNumber } = getErrorMessageInfo((error as Error).message) const lineNumber = params.lines.findIndex((line) => line.trim().startsWith(`${key}:`)) + 1 - // Add the key length plus 3 to the column number to account for the colon, - // the space after the key, and column numbers starting at 1. - // If there is no space after the colon, a YAMLException will be thrown. + // Offset for the key, colon, space, and 1-based column; missing spaces throw YAMLException. const startRange = columnNumber + key.length + 3 - // If the range is greater than the length of the line, we need to adjust the range to the end of the line + // Clamp the range to the line length when Liquid reports past the end. const endRange = startRange + value.length - 1 > params.lines[lineNumber - 1].length ? params.lines[lineNumber - 1].length - startRange + 1 @@ -65,10 +55,6 @@ export const frontmatterLiquidSyntax = { }, } -/* - Attempts to parse all liquid in the Markdown content of a file - to verify the syntax is correct. -*/ export const liquidSyntax = { names: ['GHD018', 'liquid-syntax'], description: 'Markdown content must use valid Liquid', @@ -77,17 +63,13 @@ export const liquidSyntax = { try { liquid.parse(params.lines.join('\n')) } catch (error) { - // If the error source is not a Liquid error but rather a - // ReferenceError or bad type we should allow that error to be thrown + // Let non-Liquid parser errors propagate. if (!isLiquidError(error)) throw error const { errorDescription, lineNumber, columnNumber } = getErrorMessageInfo( (error as Error).message, ) const line = params.lines[lineNumber - 1] - // We don't have enough information to know the length of the full - // liquid tag without doing some regex testing and making assumptions - // about if the end tag is correctly formed, so we just give a - // range from the start of the tag to the end of the line. + // Without a trustworthy closing tag, report from the tag start through the line end. const range: [number, number] = [columnNumber, line.slice(columnNumber - 1).length] addError( onError, @@ -103,8 +85,6 @@ export const liquidSyntax = { function getErrorMessageInfo(message: string): ErrorMessageInfo { const [errorDescription, lineString, columnString] = message.split(',') - // There has to be a line number so we'll default to line 1 if the message - // doesn't contain a line number. if (!columnString || !lineString) throw new Error('Liquid error message does not contain line or column number') const lineNumber = parseInt(lineString.trim().replace('line:', ''), 10) diff --git a/src/content-linter/lib/linting-rules/liquid-tag-whitespace.ts b/src/content-linter/lib/linting-rules/liquid-tag-whitespace.ts index 1bdae8501fdd..27236e128a6f 100644 --- a/src/content-linter/lib/linting-rules/liquid-tag-whitespace.ts +++ b/src/content-linter/lib/linting-rules/liquid-tag-whitespace.ts @@ -13,18 +13,7 @@ interface LiquidToken { end: number } -/* -Liquid tags should start and end with one whitespace. For example: - - DO use a single whitespace character - {% data %} - - DON'T use 0 or more than 1 whitespace - {%data %} - - DON'T use more than 1 whitespace between args - {%data arg1 arg2 %} -*/ +// Liquid tag delimiters and arguments each need one separating space. export const liquidTagWhitespace: Rule = { names: ['GHD042', 'liquid-tag-whitespace'], @@ -45,8 +34,7 @@ export const liquidTagWhitespace: Rule = { const range = [column, length] const tag = params.lines[lineNumber - 1].slice(column - 1, column - 1 + length) - // Get just the opening and closing tags, which includes any whitespace - // added before the tag name or any arguments + // openTag and closeTag preserve whitespace around the tag name and arguments. const openTag = tag.slice(0, token.contentRange[0] - token.begin) const closeTag = tag.slice(-(token.end - token.contentRange[1])) diff --git a/src/content-linter/lib/linting-rules/liquid-versioning.ts b/src/content-linter/lib/linting-rules/liquid-versioning.ts index 2607f9ff0a10..18dc5fe79c16 100644 --- a/src/content-linter/lib/linting-rules/liquid-versioning.ts +++ b/src/content-linter/lib/linting-rules/liquid-versioning.ts @@ -21,12 +21,7 @@ type AllFeatures = Record const allShortnames: string[] = Object.keys(allVersionShortnames) const getAllPossibleVersionNames = memoize((): Set => { - // This function might appear "slow" but it's wrapped in a memoizer - // so it's only ever executed once for all files that the - // Liquid linting rule functions on. - // The third argument passed to getDeepDataByLanguage() is only - // there for the sake of being able to write a unit test on these - // lint functions. + // Memoization loads features once, and process.env.ROOT lets tests read fixtures. return new Set([...Object.keys(getAllFeatures()), ...allShortnames]) }) @@ -117,23 +112,19 @@ export const liquidIfVersionTags = { }, } +// Conditions join clauses with "or" or "and". +// Each clause is a version name, "not" plus a version name, or a product range. +// Feature-based versioning supports only the first two formats. +// Examples: fpt, not ghec, and ghes > 3.0. function validateIfversionConditionals(cond: string, possibleVersionNames: Set): string[] { const validateVersion = (version: string): boolean => possibleVersionNames.has(version) const errors: string[] = [] - // `cond` is a string of conditions joined by ` or ` / ` and `. Each condition - // has one of the following space-separated formats: - // * Length 1: `` (example: `fpt`) - // * Length 2: `not ` (example: `not ghae`) - // * Length 3: ` ` (example: `ghes > 3.0`) - // - // Note that Lengths 1 and 2 may be used with feature-based versioning, but NOT Length 3. const condParts = cond.split(/ (or|and) /).filter((part) => !(part === 'or' || part === 'and')) for (const str of condParts) { const strParts = str.split(' ') - // if length = 1, this should be a valid short version or feature version name. if (strParts.length === 1) { const version = strParts[0] const isValidVersion = validateVersion(version) @@ -142,7 +133,6 @@ function validateIfversionConditionals(cond: string, possibleVersionNames: Set 3.0 - // where the first item is `ghes` (currently the only version with numbered releases), - // the second item is a supported operator, and the third is a supported GHES release. + // Only products with numbered releases support semantic comparisons. if (strParts.length === 3) { const [version, operator, release] = strParts const hasSemanticVersioning = Object.values(allVersions).some( @@ -170,16 +158,13 @@ function validateIfversionConditionals(cond: string, possibleVersionNames: Set3.1 or some-cool-feature` we need to open - // that `some-cool-feature` and if that has `{ghes:'>3.0', ghec:'*', fpt:'*'}` - // then *combined* versions will be `{ghes:'>3.0', ghec:'*', fpt:'*'}`. + // Expand feature conditions to their version maps before checking combined product ranges. - // If the conditions use `and` then we bail because it's too complex to handle. - // Note, don't use \b (word boundary) regex because it would match `foo-and-bar`. + // Skip combined-version checks for "and"; spaces avoid matching hyphenated feature slugs. if (/\sand\s/.test(cond)) { return [] } @@ -218,15 +199,12 @@ export function validateIfversionConditionalsVersions( const versions: Record = {} let hasFutureLessThan: boolean = false for (const part of cond.split(/\sor\s/)) { - // For example `fpt or not ghec` or `not ghes or ghec or not fpt` if (/(^|\s)not(\s|$)/.test(part)) { - // Bail because it's too complex to handle. + // Skip combined-version checks for "not", because this rule does not implement inversion. return [] } for (const [ver, value] of Object.entries(getVersionsObject(part.trim(), allFeatures))) { - // If the version value is something like `<=3.0` and the versioning is set - // to `<3.19` then it means the version can *potentially* match a version - // that doesn't exist yet, but will, in the future. + // Less-than ranges can match upcoming GHES releases, so avoid flagging them as always true. if (/<=?[\d.]+/.test(value)) { hasFutureLessThan = true } @@ -243,7 +221,7 @@ export function validateIfversionConditionalsVersions( try { applicableVersions.push(...getApplicableVersions(versions)) } catch { - // Do nothing + // A rejected range leaves applicableVersions empty, so this check reports no error. } if (isAllVersions(applicableVersions) && !hasFutureLessThan) { diff --git a/src/content-linter/lib/linting-rules/outdated-release-phase-terminology.ts b/src/content-linter/lib/linting-rules/outdated-release-phase-terminology.ts index 2da991ed7c63..c7f76f43b48a 100644 --- a/src/content-linter/lib/linting-rules/outdated-release-phase-terminology.ts +++ b/src/content-linter/lib/linting-rules/outdated-release-phase-terminology.ts @@ -5,29 +5,23 @@ import frontmatter from '@/frame/lib/read-frontmatter' import type { RuleParams, RuleErrorCallback } from '@/content-linter/types' -// Mapping of outdated terms to their new replacements // Order matters: longer phrases must come first to avoid partial matches. const TERMINOLOGY_REPLACEMENTS: [string, string][] = [ - // Beta variations → public preview ['limited public beta', 'public preview'], ['public beta', 'public preview'], ['private beta', 'private preview'], ['beta', 'public preview'], - // Alpha → private preview ['alpha', 'private preview'], - // Deprecated variations → closing down ['deprecation', 'closing down'], ['deprecated', 'closing down'], - // Sunset → retired ['sunset', 'retired'], ] -// Don't lint filepaths that have legitimate uses of these terms +// Exclude paths that use release-phase terms as legitimate content. const EXCLUDED_PATHS: string[] = [ - // Individual files 'content/actions/reference/runners/github-hosted-runners.md', 'content/actions/reference/workflows-and-actions/metadata-syntax.md', 'content/admin/administering-your-instance/administering-your-instance-from-the-command-line/command-line-utilities.md', @@ -39,7 +33,6 @@ const EXCLUDED_PATHS: string[] = [ 'data/reusables/dependabot/dependabot-updates-supported-versioning-tags.md', 'data/variables/release-phases.yml', 'data/release-notes/enterprise-server/3-17/0-rc1.yml', - // Directories 'content/site-policy/', 'data/features/', 'data/release-notes/enterprise-server/3-14/', @@ -60,7 +53,7 @@ interface MatchInfo { outdatedTerm: string } -// Precompile RegExp objects for better performance +// Compile the regexes once instead of for every line. const COMPILED_REGEXES: CompiledRegex[] = TERMINOLOGY_REPLACEMENTS.map( ([outdatedTerm, replacement]) => ({ regex: new RegExp(`(?' @@ -18,13 +14,9 @@ const PLACEHOLDER = 'APPLICATION-OR-PLATFORM-SERVICE' interface TemplateHeading { level: number text: string - /** Regex pattern for matching this heading in actual articles. */ pattern: RegExp - /** Human-readable label for error messages. */ label: string - /** If true, the section may be removed from a real article. */ optional: boolean - /** For H3 headings, the pattern of the parent H2. */ parentPattern?: RegExp } @@ -34,8 +26,8 @@ export interface ParsedTemplate { reusables: string[] } -// Finds the sentinel HTML comment, then captures the first fenced yaml block -// after it. Strips {% raw %} / {% endraw %} and {% comment %} blocks. +// The template data lives in the first fenced yaml block after the sentinel marker. +// Strip Liquid wrappers before parsing the embedded template. function extractTemplateBlock(): string { const content = fs.readFileSync(TEMPLATES_PATH, 'utf-8') const sentinelIndex = content.indexOf(SENTINEL) @@ -60,8 +52,7 @@ function extractTemplateBlock(): string { .replace(/\{%\s*comment\s*%\}[\s\S]*?\{%\s*endcomment\s*%\}/g, '') } -// Headings containing the placeholder get a pattern that matches any text in -// place of the placeholder. Fixed headings get an exact match. +// The placeholder maps to any article-specific service name; fixed headings must match exactly. function headingToPattern(text: string): RegExp { const pattern = text .split(PLACEHOLDER) @@ -70,15 +61,14 @@ function headingToPattern(text: string): RegExp { return new RegExp(`^${pattern}$`, 'i') } -// Replaces the placeholder with "..." to keep error messages concise. +// Replace the placeholder with "..." to keep error messages concise. function headingLabel(level: number, text: string): string { const prefix = '#'.repeat(level) const label = text.includes(PLACEHOLDER) ? text.replace(PLACEHOLDER, '...') : text return `${prefix} ${label}` } -// Heading text and required reusable paths all come from the template rather -// than from constants in this file. +// Read headings and required reusable paths from the template instead of duplicating them here. function parseTemplate(): ParsedTemplate { const block = extractTemplateBlock() const lines = block.split('\n') @@ -134,7 +124,7 @@ function parseTemplate(): ParsedTemplate { } } - // Reset optional flag if line has non-whitespace content (not a heading or marker) + // Non-heading content consumes the optional marker so it applies only to the next heading. if (line.trim() !== '') { nextIsOptional = false } @@ -143,7 +133,7 @@ function parseTemplate(): ParsedTemplate { return { h2s, h3s, reusables } } -// Lazy singleton: parsed once on first use. +// Parse templates.md once on first use. let _parsed: ParsedTemplate | null = null export function getTemplate(): ParsedTemplate { @@ -151,10 +141,6 @@ export function getTemplate(): ParsedTemplate { return _parsed } -// --------------------------------------------------------------------------- -// File-level heading extraction -// --------------------------------------------------------------------------- - interface Heading { level: number text: string @@ -176,11 +162,6 @@ function extractHeadings(lines: string[]): Heading[] { return headings } -// --------------------------------------------------------------------------- -// Validators -// --------------------------------------------------------------------------- - -// Validate that the required H2 sections exist and appear in the correct order. function validateH2Sections( headings: Heading[], template: ParsedTemplate, @@ -216,15 +197,13 @@ function validateH2Sections( } } -// Required H3s must exist, and every H3 under a structured parent must match a -// known template heading. +// Required H3s must exist, and structured parents reject unknown H3 headings. function validateH3Subsections( headings: Heading[], template: ParsedTemplate, onError: RuleErrorCallback, ): void { - // Group template H3s by parent pattern (keyed by pattern source string to - // avoid relying on RegExp reference equality). + // Key by pattern source instead of RegExp object identity. const h3sByParent = new Map() for (const h3 of template.h3s) { if (!h3.parentPattern) continue @@ -236,7 +215,7 @@ function validateH3Subsections( for (const [, { parentPattern, h3s: templateH3s }] of h3sByParent) { const parentIndex = headings.findIndex((h) => h.level === 2 && parentPattern.test(h.text)) - if (parentIndex === -1) continue // Missing parent caught by validateH2Sections + if (parentIndex === -1) continue // validateH2Sections reports missing parents. const childH3s: Heading[] = [] for (let i = parentIndex + 1; i < headings.length; i++) { @@ -275,7 +254,6 @@ function validateH3Subsections( } } -// Validate that all required boilerplate reusable references are present. function validateReusables( lines: string[], template: ParsedTemplate, @@ -297,10 +275,6 @@ function validateReusables( } } -// --------------------------------------------------------------------------- -// Rule export -// --------------------------------------------------------------------------- - interface Frontmatter { contentType?: string [key: string]: unknown @@ -308,7 +282,7 @@ interface Frontmatter { function isFileRaiCard(params: RuleParams): boolean { const fm: Frontmatter = (getFrontmatter(params.frontMatterLines) as Frontmatter) || {} - // Files with children: are landing pages that aggregate cards, not cards themselves. + // Files with children are landing pages that aggregate cards, not cards themselves. return fm.contentType === 'rai' && !('children' in fm) } diff --git a/src/content-linter/lib/linting-rules/rai-reusable-usage.ts b/src/content-linter/lib/linting-rules/rai-reusable-usage.ts index ec3fcc8a587b..f6fa5cf3faaa 100644 --- a/src/content-linter/lib/linting-rules/rai-reusable-usage.ts +++ b/src/content-linter/lib/linting-rules/rai-reusable-usage.ts @@ -35,12 +35,11 @@ export const raiReusableUsage: Rule = { .filter( (token: LiquidToken) => token.name === 'data' || token.name === 'indented_data_reference', ) - // It's ok to reference variables from rai content + // Allow RAI content to reference variables. .filter((token: LiquidToken) => !token.args.startsWith('variables')) for (const token of tokens) { - // If the token is `data foo.bar` or `indented_data_reference foo.bar spaces=3`, - // we only want the `foo.bar` part. + // data and indented_data_reference tokens put the reusable path first. const dataDirectoryReference = token.args.split(/\s+/)[0] if (dataDirectoryReference.startsWith('reusables.rai')) continue @@ -61,11 +60,9 @@ export const raiReusableUsage: Rule = { }, } -// Rai file content can be in either the data/reusables/rai directory -// or anywhere in the content directory +// RAI content can live in data/reusables/rai or anywhere under content. function isFileRai(params: RuleParams): boolean { - // ROOT is set in the test environment to src/fixtures/fixtures otherwise - // it is set to the root of the project. + // Tests set ROOT to src/fixtures/fixtures; production uses the repository root. const ROOT = process.env.ROOT || '.' const dataPath = path.join(ROOT, 'data/reusables') const dataRai = path.join(dataPath, 'rai') diff --git a/src/content-linter/lib/linting-rules/table-column-integrity.ts b/src/content-linter/lib/linting-rules/table-column-integrity.ts index d60daf3ed717..51682abde73d 100644 --- a/src/content-linter/lib/linting-rules/table-column-integrity.ts +++ b/src/content-linter/lib/linting-rules/table-column-integrity.ts @@ -21,10 +21,9 @@ function countColumns(row: string): number { return 0 } - // Split by '|' (but ignore escaped '\|' as these are not true separators) + // Split on unescaped pipes so escaped pipes stay in cell content. const cells = trimmed.split(NON_ESCAPED_PIPE_REGEX) - // Remove first and last elements if they're empty (from leading/trailing |) if (cells.length > 0 && cells[0].trim() === '') { cells.shift() } @@ -40,7 +39,6 @@ function isLiquidOnlyRow(row: string): boolean { if (!trimmed.includes('|')) return false const cells = trimmed.split(NON_ESCAPED_PIPE_REGEX) - // Remove empty cells from leading/trailing | const filteredCells = cells.filter((cell, index) => { if (index === 0 && cell.trim() === '') return false if (index === cells.length - 1 && cell.trim() === '') return false @@ -58,7 +56,6 @@ export const tableColumnIntegrity = { tags: ['tables', 'accessibility', 'formatting'], severity: 'error', function: (params: RuleParams, onError: RuleErrorCallback) => { - // Skip autogenerated files const frontmatterString = params.frontMatterLines.join('\n') const fm = frontmatter(frontmatterString).data if (fm && fm.autogenerated) return @@ -84,7 +81,7 @@ export const tableColumnIntegrity = { const isSeparatorRow = TABLE_SEPARATOR_REGEX.test(line) if (!inTable && isTableRow) { - // Look ahead to see if next line is a separator (confirming this is a table) + // A following separator row confirms the current row starts a table. const nextLine = lines[i + 1] if (nextLine && TABLE_SEPARATOR_REGEX.test(nextLine)) { inTable = true @@ -100,7 +97,7 @@ export const tableColumnIntegrity = { } if (inTable && isTableRow && !isSeparatorRow) { - // Skip Liquid-only rows as they're allowed to have different column counts + // Liquid-only rows can span a different number of table columns. if (isLiquidOnlyRow(line)) { continue } diff --git a/src/content-linter/lib/linting-rules/table-liquid-versioning.ts b/src/content-linter/lib/linting-rules/table-liquid-versioning.ts index 69c6ed3200e0..3243767ef7e8 100644 --- a/src/content-linter/lib/linting-rules/table-liquid-versioning.ts +++ b/src/content-linter/lib/linting-rules/table-liquid-versioning.ts @@ -2,13 +2,9 @@ import { addError } from 'markdownlint-rule-helpers' import type { RuleParams, RuleErrorCallback, Rule } from '@/content-linter/types' -// Detects a Markdown table delimiter row const delimiterRegexPure = /(\s)*(:)?(-+)(:)?(\s)*(\|)/ -// Detects a Markdown table delimiter row with a Liquid tag const delimiterRegex = /(\s)*(:)?(-+)(:)?(\s)*(\|).*({%.*(ifversion|else|endif).*%})/ -// Detects a Liquid versioning tag const liquidRegex = /^{%-?\s*(ifversion|else|endif).*-?%}/ -// Detects a Markdown table row with a Liquid versioning tag const liquidAfterRowRegex = /(\|{1}).*(\|{1}).*{%\s*(ifversion|else|endif).*%}$/ export const tableLiquidVersioning: Rule = { @@ -29,7 +25,7 @@ export const tableLiquidVersioning: Rule = { } if (delimiterRegexPure.test(line)) { - // A table with rows is at least 3 lines + // A table needs a header, delimiter, and body row. if (lines[i - 1] && lines[i + 1]) { inTable = true if (liquidAfterRowRegex.test(lines[i - 1])) { diff --git a/src/content-linter/lib/linting-rules/third-party-action-pinning.ts b/src/content-linter/lib/linting-rules/third-party-action-pinning.ts index 59abba7009c2..8e883466e0a3 100644 --- a/src/content-linter/lib/linting-rules/third-party-action-pinning.ts +++ b/src/content-linter/lib/linting-rules/third-party-action-pinning.ts @@ -5,9 +5,7 @@ import { liquid } from '@/content-render/index' import { allVersions } from '@/versions/lib/all-versions' import type { RuleParams, RuleErrorCallback, MarkdownToken, Rule } from '@/content-linter/types' -// Detects third-party actions in the format `owner/repo@ref` const actionRegex = /[\w-]+\/[\w-]+@[\w-]+/ -// Detects a full-length commit SHA (40 hexadecimal characters) const shaRegex = /[\w-]+\/[\w-]+@[0-9a-fA-F]{40}/ const firstPartyPrefixes = ['actions/', './.github/actions/', 'github/', 'octo-org/', 'OWNER/'] @@ -46,7 +44,7 @@ export const thirdPartyActionPinning: Rule = { currentLanguage: 'en', currentVersionObj: allVersions['free-pro-team@latest'], } - // If we don't parse the Liquid first, yaml loading chokes on {% raw %} tags + // Parse Liquid first because yaml loading chokes on {% raw %} tags. const renderedYaml = await liquid.parseAndRender(token.content, context) try { const yamlObj = load(renderedYaml) as WorkflowYaml diff --git a/src/content-linter/lib/linting-rules/third-party-actions-reusable.ts b/src/content-linter/lib/linting-rules/third-party-actions-reusable.ts index 3bd0ccaf83b4..633607dc678e 100644 --- a/src/content-linter/lib/linting-rules/third-party-actions-reusable.ts +++ b/src/content-linter/lib/linting-rules/third-party-actions-reusable.ts @@ -8,7 +8,7 @@ export const thirdPartyActionsReusable = { tags: ['actions', 'reusable', 'third-party'], function: (params: RuleParams, onError: RuleErrorCallback) => { filterTokens(params, 'fence', (token: MarkdownToken) => { - // Only check YAML code blocks (GitHub Actions workflows) + // Check yaml and yaml copy fences because they hold GitHub Actions examples. if (token.info !== 'yaml' && token.info !== 'yaml copy') return const codeContent = token.content @@ -29,7 +29,7 @@ export const thirdPartyActionsReusable = { lineNumber, `Code examples with third-party actions must include the disclaimer reusable. Found third-party actions: ${actionList}. Add '{% data reusables.actions.actions-not-certified-by-github-comment %}' before or inside this code block.`, token.line, - null, // No specific range within the line + null, // No exact range exists within the fence info line. null, // No fix possible: the reusable has to be added by hand ) } @@ -57,12 +57,9 @@ function findThirdPartyActions(yamlContent: string): string[] { function isExampleOrGitHubAction(actionRef: string): boolean { const excludePatterns = [ - // GitHub-owned /^actions\//, /^github\//, - // Example organizations /^(octo-org|octocat|different-org|fakeaction|some|OWNER|my-org)\//, - // Example repos (any owner) /\/example-repo[/@]/, /\/octo-repo[/@]/, /\/hello-world-composite-action[/@]/, @@ -84,11 +81,10 @@ function checkForDisclaimer( return true } - // Convert from 1-based line number to 0-based array index + // Convert from 1-based line number to 0-based array index. const codeBlockIndex = codeBlockLineNumber - 1 - // Search backwards from the code block (up to 10 lines before) - // This is reasonable since disclaimers are typically right before code blocks + // Disclaimers usually sit right before code blocks, so search up to 10 earlier lines. const searchStart = Math.max(0, codeBlockIndex - 10) for (let i = codeBlockIndex - 1; i >= searchStart; i--) { diff --git a/src/content-linter/lib/linting-rules/yaml-scheduled-jobs.ts b/src/content-linter/lib/linting-rules/yaml-scheduled-jobs.ts index 30c0bc0792e4..514ff512432e 100644 --- a/src/content-linter/lib/linting-rules/yaml-scheduled-jobs.ts +++ b/src/content-linter/lib/linting-rules/yaml-scheduled-jobs.ts @@ -37,7 +37,7 @@ export const yamlScheduledJobs: Rule = { currentLanguage: 'en', currentVersionObj: allVersions['free-pro-team@latest'], } - // If we don't parse the Liquid first, yaml loading chokes on {% raw %} tags + // Parse Liquid first because yaml loading chokes on {% raw %} tags. const renderedYaml = await liquid.parseAndRender(token.content, context) const yamlObj = load(renderedYaml) as YamlWorkflow if (!yamlObj.on) return diff --git a/src/content-render/unified/README.md b/src/content-render/unified/README.md new file mode 100644 index 000000000000..cb00579c0798 --- /dev/null +++ b/src/content-render/unified/README.md @@ -0,0 +1,23 @@ +# Unified content rendering notes + +## Annotated code blocks + +The `annotate` plugin parses fenced code blocks whose info string includes `annotate`. It splits the rendered output into `.annotate-row` elements, with code in `.annotate-code` and rendered notes in `.annotate-note`. + +Authoring rules: + +- Include `annotate` in the info string. +- Include a language on the opening code fence. +- Start notes with the single-line comment marker for the fenced language: `#`, `//`, `` after the annotations to keep syntax highlighting. + +`parse-info-string.ts` must run before `remark-rehype`, and `annotate` must run before `highlight`. diff --git a/src/content-render/unified/alerts.ts b/src/content-render/unified/alerts.ts index 8e6660eee2cb..deb0b78d5bf9 100644 --- a/src/content-render/unified/alerts.ts +++ b/src/content-render/unified/alerts.ts @@ -1,6 +1,4 @@ -/* -Custom "Alerts", based on similar filter/styling in the monolith code. -*/ +// Matches the monolith alert syntax and styling. import { visit } from 'unist-util-visit' import { h } from 'hastscript' @@ -23,7 +21,7 @@ const alertTypes: Record = { CAUTION: { icon: 'stop', color: 'danger' }, } -// Must contain one of [!NOTE], [!IMPORTANT], ... +// Matches alert markers such as [!NOTE] and [!IMPORTANT]. const ALERT_REGEXP = new RegExp(`\\[!(${Object.keys(alertTypes).join('|')})\\]`, 'gi') // Non-global version for .test() and .match() to avoid stateful lastIndex issues const ALERT_REGEXP_DETECT = new RegExp(`\\[!(${Object.keys(alertTypes).join('|')})\\]`, 'i') diff --git a/src/content-render/unified/annotate.ts b/src/content-render/unified/annotate.ts index 120acb615e87..8de84bd6e703 100644 --- a/src/content-render/unified/annotate.ts +++ b/src/content-render/unified/annotate.ts @@ -1,32 +1,4 @@ -/* -Parses fenced code blocks with `annotate` in info string. -Results in single line comments split out, output format is: - -.annotate - .annotate-row (n) - .annotate-code - .annotate-note - -Contributing rules: -- You must include `annotate` in the info string -- You must include a language on the starting ` ``` ` tag. -- Notes must start with one of: `#`, `//`, `` to maintain syntax highlighting; this will not impact what renders. - -`parse-info-string.ts` plugin is required for this to work, and must come before `remark-rehype`. -`annotate` must come before the `highlight` plugin. -*/ +// Annotate fences split single-line comments into rendered notes beside code examples. import { load } from 'js-yaml' import fs from 'fs' @@ -89,22 +61,7 @@ const languages = load(fs.readFileSync('./data/code-languages.yml', 'utf8')) as > const commentRegexes = { - // Also known has hash or sharp; but the unicode name is "number sign". - // The reason this has 2 variants is because the hash is used, in bash - // for both hash-hang and for comments. - // For example: - // - // #!/bin/bash - // - // ...is not a comment. - // But if you only look for `#` followed by anything-but `!` it will not - // match if the line is just `#`. - // - // > /^\s*#[^!]\s*/.test('#') - // false - // - // Which makes sense, because the `#` is not followed by anything. - // That's why we use the | operator to make an "exception" for that case. + // Keep shebang lines in code, but treat a line that only contains a number sign as a comment. number: /^\s*#[^!]\s*|^\s*#$/, slash: /^\s*\/\/\s*/, xml: /^\s*` }, - // Cross-page anchor section. `blocking` reflects FAIL_ON_ANCHOR_FLAW so the wording - // can't claim the check is advisory once the rollout flips it to failing. + // FAIL_ON_ANCHOR_FLAW controls blocking wording so advisory text cannot survive rollout. anchorSection: (anchors: CrossPageAnchorFlaw[], blocking = false) => { if (anchors.length === 0) return '' const shown = anchors.slice(0, 10) @@ -219,23 +202,15 @@ function groupByTarget(links: BrokenLink[]): Map { const VERSION_PREFIX_RE = /^\/[a-z-]+@[^/]+/ -/** - * True when a redirect target is the same path with a version prefix bolted on. - * - * These aren't renames, they're the versionless link resolving into a version. Telling - * an author to "update to the new path" here is actively wrong: hardcoding - * `/enterprise-server@3.21/...` into content breaks as soon as 3.22 ships. - */ +// Version-only redirects are not renames. They let shared versionless links follow the +// reader's product version instead of hardcoding one product's path. function isVersionOnlyRedirect(target: string, redirectTarget: string): boolean { const withoutVersion = redirectTarget.replace(VERSION_PREFIX_RE, '') return withoutVersion === target } -/** - * Two redirect targets that differ only by version prefix are the same rename seen from - * two versions, not a disagreement. `/enterprise-server@3.21/new` and - * `/enterprise-server@3.17/new` both mean "the page moved to /new". - */ +// Redirect targets that differ only by version prefix describe the same rename from +// different checked versions. function sameDestination(a: string, b: string): boolean { return a.replace(VERSION_PREFIX_RE, '') === b.replace(VERSION_PREFIX_RE, '') } @@ -256,9 +231,7 @@ function createRedirectSuggestion( ) } - // A versionless link that lands on a versioned path is a rename plus the version the - // check happened to run in. Only the rename is real. Suggesting the target verbatim - // would bake `enterprise-server@3.21` into content that never asked for a version. + // Strip the checked-version prefix so versionless sources do not hardcode that release. const sourceIsVersionless = !VERSION_PREFIX_RE.test(target) const versionPrefix = redirectTarget.match(VERSION_PREFIX_RE)?.[0] if (sourceIsVersionless && versionPrefix) { @@ -298,7 +271,6 @@ export function groupBrokenLinks( } }) - // Sort: errors first, then alphabetically return groups.sort((a, b) => { if (a.isWarning !== b.isWarning) return a.isWarning ? 1 : -1 return a.target.localeCompare(b.target) @@ -349,13 +321,8 @@ function createSummary(errorCount: number, warningCount: number, totalOccurrence return `Found ${parts.join(' and ')} across ${totalOccurrences} occurrence${plural}.` } -/** - * Describe which versions a link breaks in, but only when that is news. - * - * Nearly every broken link breaks in every version, so printing the full list on every - * group is noise that also blows past the issue body size limit. Say something only when a - * link is version-specific. - */ +// Mention versions only for version-specific breakage, because the common case breaks in +// every checked version and can push the issue body past its size limit. export function describeVersions( versions: string[] | undefined, versionsChecked: string[] | undefined, @@ -366,13 +333,8 @@ export function describeVersions( return versions.join(', ') } -/** - * Merge one report per version into a single report. - * - * The workflow used to concatenate each version's rendered Markdown, so a link broken in - * every version produced an identical section per version. Merging on the link itself means - * one section per real problem, with the versions recorded on the occurrence. - */ +// Merge by link so one real problem produces one section, with affected versions recorded +// on the occurrence. export function mergeInternalLinkReports( reports: { version: string; report: LinkReport }[], options: { actionUrl?: string; versionsChecked?: string[] } = {}, @@ -394,10 +356,7 @@ export function mergeInternalLinkReports( existing.isRedirect = existing.isRedirect || occurrence.isRedirect existing.requiresVersionContext = existing.requiresVersionContext || occurrence.requiresVersionContext - // Keeping the first target and dropping the rest is only safe while every - // version agrees on where the page went. Today they always do, but if that ever - // stops being true the report would confidently name a destination that is - // right for one version and wrong for the others. Flag it instead. + // Keep the first redirect target and flag conflicts when later versions point elsewhere. if ( existing.redirectTarget && occurrence.redirectTarget && @@ -413,9 +372,7 @@ export function mergeInternalLinkReports( } } - // A version with no broken links writes no report, so the files on disk undercount what - // was actually checked. Callers that know the full matrix pass it in, otherwise fall back - // to what was found. + // Supplied matrix versions preserve versions that produced no report file. const versionsChecked = options.versionsChecked?.length ? options.versionsChecked : reports.map((r) => r.version) @@ -440,8 +397,7 @@ export function generateInternalLinkReport( const errors = groups.filter((g) => !g.isWarning) const warnings = groups.filter((g) => g.isWarning) - // The workflow concatenates every version's report into one issue, so without this - // label there's no way to tell which version a section covers. + // Per-version JSON reports also render standalone artifacts, so each title names its scope. const scope = [options.version, options.language].filter(Boolean).join(' ') const scopeLabel = scope ? ` (${scope})` : '' @@ -482,55 +438,31 @@ export function generateExternalLinkReport( } } -/** - * How a writer actually fixes a group. - * - * Grouping by target URL produces one section per broken URL, which is why the report runs - * to hundreds of sections that all look equally urgent. Grouping by fix strategy instead - * means each section is one decision: run a command, repoint a heading anchor, or choose a - * new destination by hand. - */ +// Fix buckets replace hundreds of URL sections with decisions: +// run a command, repoint a heading anchor, or choose a new destination by hand. export type FixStrategy = 'codemod' | 'versionless' | 'anchor' | 'decide' -/** - * Past this many docsets, listing one command per docset is noisier than a single pass over - * all of `content`. - */ +// Past this cap, one content-wide command is clearer than one command per docset. +// Content-wide runs take minutes; three docsets take seconds. const MAX_LISTED_CODEMOD_PATHS = 8 -/** - * How many rows of the codemod table to print. The codemod does this work, so the full list - * is reference material, not a task list. Printing all of it costs more than half the issue - * body budget, and the complete list is in the workflow artifact either way. - */ +// The codemod handles this bucket, so the table is reference material, not a task list. +// Printing every row can consume more than half the issue body budget. const MAX_CODEMOD_ROWS = 40 -/** - * How many stale anchors to print. This bucket is real work, but 70-plus entries is more - * than anyone picks up in a week, and each entry costs several times a table row because it - * lists every file the link appears in. The rest are in the workflow artifact. - */ +// Stale anchors need human work, but runs exceed 70 entries and each lists every file. +// The workflow artifact keeps entries over the cap. const MAX_ANCHOR_GROUPS = 25 -/** - * How many version-only redirects to print. This bucket needs no action at all, so the list - * exists to show what was ruled out, not to be worked through. - */ +// Version-only redirect rows document ruled-out links, not writer tasks. const MAX_VERSIONLESS_ROWS = 25 -/** - * How many files to list under a single broken link. Nothing bounds how many pages reuse - * one link, so without this a single popular link could fill the issue body on its own. The - * busiest link in the current eight-version data appears in 31 files, so this does not - * trigger today. - */ +// Cap files per target so one reused link cannot fill the issue body. The busiest +// eight-version run found 31 files for one target, so this cap truncates known input. const MAX_FILES_PER_GROUP = 20 -/** - * Split a list at a cap and describe what is missing, so no section can grow without bound. - * GitHub rejects issue bodies over 65,536 characters and the workflow truncates at 60,000 - * with a blind slice, which can cut a table in half. - */ +// Split capped lists with an explicit hidden count. GitHub rejects issue bodies over +// 65,536 characters, and the workflow truncates at 60,000 with a blind slice. function capGroups( groups: GroupedBrokenLinks[], max: number, @@ -546,44 +478,29 @@ export function classifyFixStrategy(group: GroupedBrokenLinks): FixStrategy { .filter((target): target is string => Boolean(target)) if (group.isWarning && redirectTargets.length > 0) { - // The path is unchanged and the redirect only adds a version. Rewriting these would - // hardcode a version into content, which breaks when the next release ships. The - // codemod leaves them alone, so promising that it fixes them is a lie. - // - // Every target has to be version-only, not just the first. A group can span versions, - // and a link that merely gains a version prefix in one version but points at a renamed - // page in another is real work. Ties go to the actionable bucket. + // Only all-version-only redirects land in the no-action bucket; mixed groups stay actionable. if (redirectTargets.every((target) => isVersionOnlyRedirect(group.target, target))) { return 'versionless' } - // A redirect to a genuinely different path. `update-internal-links` rewrites these - // with no human judgment involved, but only when it can find the redirect from the - // href as written. If any occurrence needed version context to resolve, the codemod - // would be a no-op, so send the whole group to a human instead. + // Version-context redirects need human review because the codemod looks up hrefs as written. if (group.occurrences.some((occ) => occ.requiresVersionContext)) { return 'decide' } - // Versions disagree about where the page went, so there is no single correct rewrite. + // Conflicting redirect targets need human review because no single rewrite is correct. if (group.occurrences.some((occ) => occ.hasConflictingRedirectTargets)) { return 'decide' } return 'codemod' } - // The link carries a fragment, so the stale part is likely a renamed heading. + // A fragment on a broken target usually means the heading moved, not the page. if (group.target.includes('#')) { return 'anchor' } return 'decide' } -/** - * The directories the codemod needs to be pointed at, derived from the files that actually - * contain the links. Running it against all of `content` takes minutes; running it against - * three docsets takes seconds. - * - * The checker records file paths relative to `content`, so `actions/foo.md` means - * `content/actions/foo.md`. Paths that already name a top-level directory are left alone. - */ +// Derive codemod directories from files that contain links so runs can stay scoped. +// Checker paths are relative to content, and already-rooted content or data paths stay as is. function codemodPaths(groups: GroupedBrokenLinks[]): string[] { const paths = new Set() for (const group of groups) { @@ -596,7 +513,6 @@ function codemodPaths(groups: GroupedBrokenLinks[]): string[] { return [...paths].sort() } -/** The union of versions across a group's occurrences. */ function groupVersions(group: GroupedBrokenLinks): string[] { const versions = new Set() for (const occ of group.occurrences) { @@ -665,14 +581,8 @@ ${rows}${truncationNote}
    ` } -/** - * Version-only redirects: the path is unchanged and the redirect just adds a version. - * - * These are not renames. A versionless link is supposed to resolve into whichever version - * the reader is on, and that is exactly what the redirect does. Rewriting them would pin - * content to a version that goes stale on the next release, so the codemod leaves them - * alone and so should writers. - */ +// Version-only redirects are not renames. Versionless shared links follow the reader's +// product version, and rewriting them would pin content to one product path. function renderVersionlessSection(groups: GroupedBrokenLinks[]): string { const { listed, hidden } = capGroups(groups, MAX_VERSIONLESS_ROWS) const rows = listed @@ -733,10 +643,7 @@ ${blurb} ${sections}${truncationNote}` } -/** - * Render an internal report as four buckets ordered by how much work each one costs, from - * one command down to nothing at all. - */ +// Order internal report buckets from one command down to no action. function renderByFixStrategy( groups: GroupedBrokenLinks[], isExternal: boolean, @@ -842,8 +749,7 @@ export function reportToMarkdown(report: LinkReport, isExternal = false): string return parts.join('\n') } - // Table of contents for large reports. The internal report is grouped by fix strategy - // instead, where the three bucket headings are the navigation. + // Large external reports need a table of contents; internal bucket headings navigate. if (isExternal && report.groups.length > 5) { parts.push(TEMPLATES.tableOfContents(report.groups)) parts.push('') @@ -857,7 +763,7 @@ export function reportToMarkdown(report: LinkReport, isExternal = false): string ) } - // Self-referential links section (external report only) + // Self-referential links only appear in external reports. if (hasSelfReferentialGroups) { parts.push( TEMPLATES.selfReferentialLinks('Potential Internal Links', report.selfReferentialGroups!), diff --git a/src/links/lib/page-anchors.ts b/src/links/lib/page-anchors.ts index bd7738104228..71d6478b4404 100644 --- a/src/links/lib/page-anchors.ts +++ b/src/links/lib/page-anchors.ts @@ -9,37 +9,25 @@ import { } from '@/links/lib/extract-links' import { computeHeadingIds } from '@/links/lib/heading-anchors' -/** - * Shared, version-aware helpers for validating that a `path#fragment` link lands on a - * real heading of its target page. Kept neutral (no PR/CI specifics) so both the - * scheduled internal checker and the PR-time gate can render a target page in a given - * version and compute its heading anchor IDs the same way. - */ +// Version-aware anchor helpers stay neutral so the scheduled checker and PR gate render +// target pages with the same version context and compute the same heading IDs. -// Explicit version prefix on a resolved pageMap key, e.g. the `enterprise-cloud@latest` -// in `/en/enterprise-cloud@latest/actions/foo`. `resolveInternalLinkKey` already maps -// `enterprise-server@latest` to the stable release, so `@latest` never appears for GHES. +// Matches explicit version prefixes on resolved pageMap keys such as +// /en/enterprise-cloud@latest/actions/foo. resolveInternalLinkKey maps +// enterprise-server@latest to the stable release, so @latest never appears for GHES. const EXPLICIT_VERSION_RE = /^\/(?:[a-z]{2}(?:-[a-z]{2})?\/)?(free-pro-team@latest|enterprise-cloud@latest|enterprise-server@[0-9.]+)(?:\/|$)/ -/** - * Extract the version a resolved link key pins to, or null when the key carries no - * explicit version segment (an unversioned link that inherits the source page's version). - */ +// Null means the key carries no explicit version and inherits the source page's version. export function versionFromResolvedKey(key: string): string | null { const match = key.match(EXPLICIT_VERSION_RE) if (!match) return null return allVersions[match[1]] ? match[1] : null } -/** - * Line numbers (1-based) in `content` where a Markdown link points at exactly - * `hrefWithFragment`. - * - * The destination must END at the match: a bare substring search is a prefix match, so - * looking for `](/a#foo` would also hit `](/a#foobar)` and misreport that line. A Markdown - * destination ends at `)` or at whitespace before an optional title. - */ +// Find 1-based line numbers where a Markdown link points exactly at hrefWithFragment. +// The destination must end at the match; substring searches confuse /a#foo with /a#foobar. +// Markdown destinations end at ) or at whitespace before an optional title. export function findLinkLines(content: string, hrefWithFragment: string): number[] { const needle = `](${hrefWithFragment}` const lines = content.split('\n') @@ -57,18 +45,9 @@ export function findLinkLines(content: string, hrefWithFragment: string): number return found } -/** - * Resolve a link href to a pageMap key, considering the version the source page is being - * rendered in. - * - * `resolveInternalLinkKey` handles the common shapes once it knows the version, so hand - * the version to it directly. That also fixes the precedence: a target that applies to - * both FPT and the source version has a key for each, and without the version the `/en` - * key wins even during an enterprise run. - * - * The explicit retry below still earns its place for hrefs that carry a language prefix, - * which `resolveInternalLinkKey` refuses to reinterpret as relative to a version. - */ +// Resolve hrefs in the source page's version so enterprise keys beat the /en key. +// The explicit retry handles hrefs with a language prefix, which resolveInternalLinkKey +// refuses to reinterpret as relative to a version. export function resolveLinkKeyForVersion( href: string, version: string, @@ -85,18 +64,10 @@ export function resolveLinkKeyForVersion( return resolveInternalLinkKey(`/${version}${withoutLang}`, pageMap) } -/** - * Whether a target page's heading anchors can be derived from its Markdown headings. - * - * Two page kinds compute their headings from data at runtime, not from static Markdown - * headings, so `computeHeadingIds` can't see them and would report false positives: - * - `autogenerated` pages (REST/GraphQL/webhooks/audit-log-events): headings come from - * OpenAPI operation IDs / action prefixes. - * - the glossary page: headings come from a `{% for glossary in glossaries %}` loop over - * `data/glossaries/external.yml`, which isn't populated in this lightweight render. - * - * Links into these pages are skipped rather than flagged. - */ +// Some target pages compute headings from runtime data, not static Markdown headings. +// Autogenerated REST, GraphQL, webhook, and audit-log pages derive headings from data files. +// The glossary page loops over data/glossaries/external.yml, which lightweight renders skip. +// Links into these pages are skipped rather than flagged. export function isAnchorCheckableTarget(page: Page): boolean { if ((page as unknown as { autogenerated?: unknown }).autogenerated) return false const markdown = page.markdown @@ -109,22 +80,16 @@ export function isAnchorCheckableTarget(page: Page): boolean { return true } -// Loaded once and shared: getDeepDataByLanguage walks the whole data/tables tree, which is -// far too expensive to repeat per page per version. +// getDeepDataByLanguage walks the whole data/tables tree, which is too costly per page. let tablesCache: Record | null = null function getTables(): Record { if (!tablesCache) tablesCache = getDeepDataByLanguage('tables', 'en') return tablesCache } -/** - * Build a minimal per-version Liquid context for rendering a page's Markdown. Mirrors the - * context the scheduled internal checker uses (version flags + feature flags), plus the - * `variables` site data so `{% data variables.* %}` in headings resolves, and the `tables` - * data so pages that generate headings from a `{% for %}` loop over a reference table - * (e.g. `content/copilot/reference/copilot-feature-matrix.md`) produce their real heading - * text instead of a literal `{{ groupName }}`, which would otherwise false-positive. - */ +// Match the scheduled internal checker's version flags and feature flags. +// Include variables data so data tags in headings resolve, and tables data so reference-table +// loops such as content/copilot/reference/copilot-feature-matrix.md render real heading text. export function buildRenderContext( page: Page, version: string, @@ -146,11 +111,7 @@ export function buildRenderContext( } as unknown as Context } -/** - * Compute the set of heading anchor IDs for a page rendered in a specific version. - * Results are memoized in the optional `cache` (keyed by version + relative path) so a - * target referenced by many links is only rendered once per version. - */ +// Cache by version and relative path so repeated target links render only once per version. export async function getPageHeadingIds( page: Page, version: string, diff --git a/src/links/lib/update-internal-links.ts b/src/links/lib/update-internal-links.ts index 3120fd1c6ec3..550e24e2568c 100644 --- a/src/links/lib/update-internal-links.ts +++ b/src/links/lib/update-internal-links.ts @@ -51,22 +51,19 @@ type Warning = { column?: number } -// A fragment (`#anchor`) that was carried over when a redirect rewrote the link's path. -// It needs validating against the destination page before we decide to keep or drop it. +// Redirected paths can carry anchors to pages with different headings, so validate first. type CarriedFragment = { hash: string destPage?: Page } type NewHrefResult = { - // The rewritten href WITHOUT the fragment when `fragment` is set (the caller decides - // whether to re-append it); otherwise the full href including any fragment/search. + // href omits a carried fragment so the caller can decide whether to re-append it. href: string fragment?: CarriedFragment } -// A link/definition replacement collected during the synchronous AST walk. Applying it is -// deferred until after async fragment validation, so a carried-over anchor can be dropped. +// Defer replacements until async fragment validation can drop stale carried anchors. type PendingReplacement = { asMarkdown: string line: number @@ -74,11 +71,7 @@ type PendingReplacement = { baseHref: string makeMarkdown: (href: string) => string fragment?: CarriedFragment - /** - * Byte range of this link in the source, when the node's position could be mapped back - * and the slice matches `asMarkdown` exactly. Replacing by range instead of by string - * search keeps identical text elsewhere in the file untouched. - */ + // Ranged replacements keep identical link text elsewhere in the file untouched. span?: [number, number] } @@ -87,9 +80,7 @@ const Options = { fixHref: false, verbose: false, strict: false, - // When a redirect rewrites a link's path, the carried-over `#anchor` is validated - // against the destination page's real headings. If it exists in no applicable version - // it is dropped by default. Set this to keep such fragments (warn only). + // Keep stale carried anchors as warnings instead of dropping them after validation. keepStaleFragments: false, } @@ -126,21 +117,14 @@ export async function updateInternalLinks(files: string[], options = {}) { return results } -/** - * Exported so tests can drive a single file with a hand-built context. Loading the real - * page tree takes tens of seconds, which is too slow to cover the rewrite branches. - */ +// A hand-built test context avoids loading the real page tree, which takes tens of seconds. export async function updateFile(file: string, context: LinkContext, opts: typeof Options) { const rawContent = fs.readFileSync(file, 'utf8') let { data, content } = frontmatter(rawContent) data = data || {} content = content || '' - // Since this function can process both `.md` and `.yml` files, - // `frontmatter(rawContent).data` gives what we need for a `.md` file, but always - // returns `{}` for a `.yml` file. - // And since the Yaml file might contain arrays of internal linked - // pathnames, we have to re-read it fully. + // frontmatter data is empty for .yml files, so read YAML files again for link arrays. const isYaml = file.endsWith('.yml') if (isYaml) { Object.assign(data, loadYaml(content)) @@ -151,20 +135,12 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo // Captured so the closure below sees a non-reassignable string. const source = content - // A YAML file is parsed as Markdown to find its links, and that parse is - // indentation-sensitive: a value indented four or more spaces reads as a code block, - // so the AST holds fewer link nodes than the text has occurrences. Stripping the - // leading whitespace from every line exposes all of them. Line numbers are unaffected, - // and `sourceSpan` maps each node's columns back onto the original text so the - // rewrite still lands on the real bytes. + // Dedent YAML because values indented four or more spaces parse as code blocks; lines stay put. const parseSource = isYaml ? dedentLines(source) : source const lineStarts = buildLineStarts(source) const indents = isYaml ? source.split('\n').map((line) => /^[ \t]*/.exec(line)![0].length) : null - /** - * Column of a node in the original text. The YAML parse runs on dedented lines, so the - * indent has to go back on before the column is reported to a human. - */ + // Report original columns by adding YAML indentation back onto the dedented parse. function sourceColumn(node: Nodes): number | undefined { const pos = node.position if (!pos?.start.column) return undefined @@ -172,10 +148,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo return pos.start.column + indent } - /** - * Byte range of a node in the source, or undefined when the range can't be trusted: - * a node spanning several lines, or a serialization that doesn't match the source. - */ + // Trust a source range only when one line serializes back to the original text. function sourceSpan(node: Nodes, asMarkdown: string): [number, number] | undefined { const pos = node.position if (!pos?.start.line || !pos.end.line || pos.start.line !== pos.end.line) return undefined @@ -200,7 +173,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo const ANY = Symbol('any') const IS_ARRAY = Symbol('is array') - // Which frontmatter keys hold links, and which of their sub-keys to descend into. + // Only these frontmatter keys hold links this script rewrites. const HAS_LINKS: Record = { featuredLinks: ['gettingStarted', 'startHere', 'guideCards', 'popular'], introLinks: ANY, @@ -240,10 +213,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo } } } catch (error) { - // When in strict mode, if it throws an error that stacktrace will - // bubble up to the CLI. And the CLI will mention which file it - // was processing when it failed. But we have a valuable piece of - // information here about which frontmatter key it was that failed. + // Include the failing frontmatter key because the CLI warning only names the file. logger.warn('Frontmatter key processing failed', { key }) throw error } @@ -251,19 +221,15 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo const lineOffset = rawContent.replace(content, '').split(/\n/g).length - 1 - // Replacements are collected here during the synchronous AST walk and applied after - // async fragment validation below, so a carried-over `#anchor` can be dropped when the - // redirected destination page doesn't actually have that heading. + // Apply replacements after async fragment validation so stale carried anchors can drop. const pending: PendingReplacement[] = [] visit(ast, definitionMatcher as Test, (node: Nodes) => { const asMarkdown = toMarkdown(node).trim() - // E.g. `[foo]: /bar` if (opts.fixHref && content.includes(asMarkdown) && isDefinition(node)) { const { label } = node const result = getNewHref(node.url, context, opts, file) - // getNewHref() might return a deliberate `undefined` if the - // new href value could not be computed for some reason. + // getNewHref returns undefined when non-strict mode cannot resolve the link. const baseHref = result === undefined ? node.url : result.href const column = sourceColumn(node) const line = (node.position?.start.line ?? 0) + lineOffset @@ -282,14 +248,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo visit(ast, linkMatcher as Test, (node: Nodes) => { const asMarkdown = toMarkdown(node).trim() if (content.includes(asMarkdown) && isLink(node)) { - // The title part of the link might be more Markdown. - // For example... - // - // [This *is* cool](/articles/link) - // - // The title is the combined serialization of `node.children`, and `toMarkdown()` - // always appends `\n`, hence the slice. The example above yields `This *is* cool`, - // which still carries its emphasis markers and so won't match a page title. + // Serializing children preserves Markdown markers, so [This *is* cool] stays unmatched. const title = node.children.map((child: Nodes) => toMarkdown(child).slice(0, -1)).join('') let newTitle = title @@ -302,33 +261,10 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo if (opts.setAutotitle) { if (hasQuotesAroundLink) { - /** - * A lot of internal links are bullet points like: - * - * - [Creating a repository](/articles/create-a-repo) - * - [Forking a repository](/articles/fork-a-repo) - * or - * 1. [Set your username in Git](/github/getting-started-with-github/setting-your-username-in-git). - * 1. [Set your commit email address in Git](/articles/setting-your-commit-email-address). - * - * Perhaps we could recognize them as such and consider them - * matches anyway. In particular if the title consists of - * a leading capital letter and most of the rest lower case. - */ - if (title !== AUTOTITLE) { newTitle = AUTOTITLE } } else { - /** - * The Markdown link sometimes is written like this: - * - * ["This is the title](/foo/bar)." - * - * or... - * - * ["This is the title"](/foo/bar). - */ if (xValue) { if (singleStartingQuote(xValue)) { const column = sourceColumn(node) @@ -354,8 +290,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo } if (opts.fixHref) { const result = getNewHref(node.url, context, opts, file) - // getNewHref() might return a deliberate `undefined` if the - // new href value could not be computed for some reason. + // getNewHref returns undefined when non-strict mode cannot resolve the link. if (result !== undefined) { baseHref = result.href fragment = result.fragment @@ -380,8 +315,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo } }) - // Resolve any carried-over fragments, which renders the destination page, then apply - // every replacement to `newContent` in document order. + // Validate carried fragments before applying replacements in document order. for (const item of pending) { let finalHref = item.baseHref if (item.fragment) { @@ -397,7 +331,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo column: item.column, }) } else { - // Drop the fragment: leave `finalHref` as the fragment-less base href. + // Drop the stale fragment by leaving finalHref at the fragmentless base href. warnings.push({ warning: `Removed stale anchor '${hash}': not found on the redirected destination page in any applicable version`, asMarkdown: item.asMarkdown, @@ -422,7 +356,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo column: item.column, }) } else { - // 'keep': the anchor exists on the destination in every applicable version. + // keep means the anchor exists in every applicable destination version. finalHref = item.baseHref + hash } } @@ -437,16 +371,13 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo if (item.span) { spanEdits.push({ start: item.span[0], end: item.span[1], text: newAsMarkdown }) } else { - // No trustworthy range for this node, so fall back to a string search. Left for - // the second pass, after the ranged edits, since a search can't be offset-aware. + // String-search fallback runs after ranged edits because it cannot adjust offsets. stringEdits.push({ find: item.asMarkdown, text: newAsMarkdown }) } } } - // Ranged edits go in descending order so earlier offsets stay valid, and each one - // touches exactly the bytes the parser identified as a link. That's what keeps an - // identical string in a comment or a code example from being rewritten too. + // Descending ranged edits preserve offsets and keep identical text elsewhere untouched. spanEdits.sort((a, b) => b.start - a.start) for (const edit of spanEdits) { newContent = newContent.slice(0, edit.start) + edit.text + newContent.slice(edit.end) @@ -468,7 +399,7 @@ export async function updateFile(file: string, context: LinkContext, opts: typeo } } -/** Strip leading whitespace from every line, preserving the line count. */ +// Preserve line count while stripping leading whitespace from every line. function dedentLines(content: string): string { return content .split('\n') @@ -476,7 +407,6 @@ function dedentLines(content: string): string { .join('\n') } -/** Byte offset where each line begins, so a line/column pair can become an offset. */ function buildLineStarts(content: string): number[] { const starts = [0] for (let i = 0; i < content.length; i++) { @@ -506,23 +436,16 @@ function linkMatcher(node: Node) { if (isLink(node) && node.url) { const { url } = node if (url.startsWith('/') || url.startsWith('./')) { - // Sometimes there's a link to view the asset as a separate link. - // Skip these because they ultimately link to an actual Page. + // Asset and public paths serve static assets or generated schema files, not pageMap entries. if (url.startsWith('/assets') || url.startsWith('/public/')) { return false } - // If a link uses Liquid we can't process it. It would require full - // rendering which this script is not doing. + // Liquid links need full rendering, which this script does not do. if (url.includes('{{') || url.includes('{%')) { return false } - // Sometimes we link to archived enterprise-server versions. These - // can never be updated because although they appear to be internal, - // they are, in a sense external. For example: - // See "[This old thing](/enterprise-server@3.1/some/page)". - // Skip these const version = getVersionStringFromPath(url) if ( version && @@ -532,9 +455,7 @@ function linkMatcher(node: Node) { return false } - // Really old versions like `/enterprise/2.1` don't need to be - // corrected because they're deliberately pointing to archived - // versions. + // Legacy enterprise paths deliberately point to archived content. if (patterns.getEnterpriseVersionNumber.test(url)) { return false } @@ -557,19 +478,6 @@ function getNewFrontmatterLinkList( file: string, rawContent: string, ) { - /** - * The `list` is expected to all be strings. Sometimes they're like this: - * - * /search-github/searching-on-github/searching-for-repositories - * - * Sometimes they're like this: - * - * {% ifversion fpt or ghec or ghes > 3.4 %}/pages/getting-started-with-github-pages{% endif %} - * - * In the case of Liquid, we have to temporarily remove it to be able to - * test the path as a URL. - **/ - const better = [] for (const entry of list) { if (/{%\s*else\s*%}/.test(entry)) { @@ -601,7 +509,7 @@ function getNewFrontmatterLinkList( logger.warn(msg, { file, pure, lineNumber }) better.push(entry) } else { - // Perhaps it just redirected to a specific version + // Keep links whose redirect only adds a supported version prefix. const redirectedWithoutLanguage = getPathWithoutLanguage(redirected) const asURLWithoutVersion = getPathWithoutVersion(redirectedWithoutLanguage) if (asURLWithoutVersion === pure) { @@ -615,8 +523,7 @@ function getNewFrontmatterLinkList( return better } -// Try to return the line in the raw content that entry was on. -// Only approximate: `entry` comes out of the parsed YAML, so its original text is gone. +// Find the raw line for a parsed YAML entry when the original text survives unchanged. function findLineNumber(entry: string, rawContent: string) { let number = 0 for (const line of rawContent.split(/\n/g)) { @@ -632,15 +539,6 @@ function findLineNumber(entry: string, rawContent: string) { const liquidStartRex = /^{%-?\s*ifversion .+?\s*%}/ const liquidEndRex = /{%-?\s*endif\s*-?%}$/ -// Return -// -// /foo/bar -// -// if the text input was -// -// {% ifversion ghes%}/foo/bar{%endif %} -// -// And if no liquid, just return as is. function stripLiquid(text: string) { if (liquidStartRex.test(text) && liquidEndRex.test(text)) { return text.replace(liquidStartRex, '').replace(liquidEndRex, '').trim() @@ -672,11 +570,10 @@ function getNewHref( const pure = parsed.pathname let newHref = pure.replace(patterns.trailingSlash, '$1') - // Before testing if it redirects somewhere, we temporarily - // pretend it's already prefixed for English (/en) + // Redirect checks need the English prefix even though source links omit it. const [language, withoutLanguage] = splitPathByLanguage(newHref, currentLanguage) if (withoutLanguage !== newHref) { - // It means the link already had a language in it + // Skip hardcoded-language links because source links stay language-neutral. const msg = `Unable to cope with internal links with hardcoded language '${newHref}' (file: ${file})` if (opts.strict) { throw new Error(msg) @@ -688,11 +585,9 @@ function getNewHref( const newHrefWithLanguage = getPathWithLanguage(withoutLanguage, language) const redirected = getRedirect(newHrefWithLanguage, context) - // `undefined` means the link didn't need redirecting. The broken-link check below is - // belt and braces: the link checkers cover it too. + // A missing redirect plus no pageMap entry means the link is broken. if (redirected === undefined) { if (!context.pages[newHrefWithLanguage]) { - // If this happens, it's very possible that it's a broken link const msg = `A link appears to be broken. Neither redirect or a findable page '${href}' (${file})` if (opts.strict) { throw new Error(msg) @@ -704,26 +599,8 @@ function getNewHref( } if (redirected) { - // The getRedirect() function will produce a final URL that the user - // can use, but that means it also injects the language in there. - // For updating the content statically, we don't want that. - // Note: It could be an idea to somehow tell getRedirect() to not - // bother but perhaps it adds unnecessarily complexity to a function that - // has to work perfectly for runtime. + // Static rewrites drop getRedirect's language prefix because source links stay neutral. const redirectedWithoutLanguage = getPathWithoutLanguage(redirected) - // Some paths can't be viewed in free-pro-team so the getRedirect() - // function will inject the version that you're supposed to go to. - // For example `/enterprise/admin/guides/installation/configuring-a-hostname` - // redirects to `/enterprise-server@3.7/admin/configuration/configuring-...` - // (at the time of writing) which is good when you're actually clicking - // the link but not good when we're trying to update the source - // content. - // `getPathWithoutVersion` strips a supported version prefix and leaves everything - // else alone, so `/enterprise-server@3.22/get-started` becomes `/get-started` while - // `/get-started` is returned unchanged. - // Two exceptions: content sometimes links to a specific version deliberately, which - // must be left alone, and `getRedirect()` always strips a `/free-pro-team@latest/` - // prefix, which has to be put back. if (withoutLanguage.includes(`/${nonEnterpriseDefaultVersion}/`)) { newHref = `/${nonEnterpriseDefaultVersion}${redirectedWithoutLanguage}` } else if (withoutLanguage.startsWith('/enterprise-server/')) { @@ -737,9 +614,7 @@ function getNewHref( return } } else if (withoutLanguage.startsWith('/enterprise-server@latest')) { - // getRedirect() will always replace `enterprise-server@latest` with - // whatever the latest number is. E.g. `enterprise-server@3.9`. - // But we have to "undo" that. + // Preserve enterprise-server@latest because source content tracks the moving release. newHref = `/enterprise-server@latest${getPathWithoutVersion(redirectedWithoutLanguage)}` } else if (getPathWithoutVersion(withoutLanguage) !== withoutLanguage) { newHref = redirectedWithoutLanguage @@ -750,27 +625,21 @@ function getNewHref( const base = search ? `${newHref}${search}` : newHref - // No fragment → nothing to validate; return the (possibly rewritten) path as-is. if (!hash) { return { href: base } } - // The path wasn't rewritten, so the original fragment still points at the same page and - // remains valid. Keep it appended, exactly as before. + // Unredirected paths keep fragments because they still point at the same page. if (!redirected) { return { href: base + hash } } - // The path WAS rewritten by a redirect, so the carried-over fragment may be stale on the - // destination page. Surface it (with the destination Page, if resolvable) so the caller - // can validate it against the destination's real headings and decide keep vs. drop. + // Redirected paths validate carried fragments against the destination page's headings. const destPage = resolveDestinationPage(context.pages, redirected) return { href: base, fragment: { hash, destPage } } } -// Resolve the Page a redirect points at, so its headings can be validated. `redirected` is -// a full permalink-style URL from getRedirect() (e.g. `/en/get-started/foo` or -// `/en/enterprise-cloud@latest/get-started/foo`), which is how the page map is keyed. +// getRedirect returns language-prefixed permalinks, matching the primary pageMap keys. function resolveDestinationPage(pages: Record, redirected: string): Page | undefined { return pages[redirected] || pages[getPathWithLanguage(getPathWithoutLanguage(redirected), 'en')] } @@ -783,18 +652,9 @@ function isSimpleQuote(text: string) { return text.startsWith('"') && text.endsWith('"') && text.split('"').length === 3 } -/** - * Write a YAML data file back out. - * - * For `.yml` files every link fix lands in `newContent`, the file's own text, because - * `updateFile` finds Markdown links by parsing that text and rewrites them in place. - * `newData` is only mutated for the structured link keys (`featuredLinks` and - * `introLinks`), which no file under `data/` currently uses. - * - * Writing `dump(newData)` therefore threw away every fix and reserialized the untouched - * data instead: pure churn, no change. Prefer the surgically edited text, and only fall - * back to reserializing when the structured data genuinely changed. - */ +// YAML link fixes choose newContent because updateFile rewrites parsed Markdown links +// against the file's own text. newData changes only for structured link keys such as +// featuredLinks and introLinks, which no data file uses. dump would reserialize untouched data. export function serializeYaml( newContent: string, newData: Record | undefined, @@ -803,9 +663,7 @@ export function serializeYaml( ): string { if (!differentData) return newContent if (differentContent) { - // The two kinds of change live in different representations and there is no - // format-preserving way to merge them, so `dump` would silently drop the text - // fixes. No file hits this today. Fail loudly rather than lose edits quietly. + // No format-preserving merge exists for simultaneous text and structured data edits. throw new Error( 'Cannot serialize a YAML file that has both text and structured data changes ' + 'without losing one of them. This needs a format-preserving merge.', @@ -814,15 +672,9 @@ export function serializeYaml( return dump(newData || {}) } -/** - * Write a Markdown page back out, preserving the original frontmatter text verbatim - * whenever the frontmatter data itself didn't change. - * - * Round-tripping frontmatter through the YAML serializer reflows values that were - * never touched: long `intro` strings become block scalars, `redirect_from` entries get - * rewrapped, and quote styles change. That churn dwarfs the actual link fixes and makes - * a bulk run unreviewable, which is why this only reserializes when it has to. - */ +// Preserve original Markdown frontmatter when frontmatter data did not change. The YAML +// serializer reflows untouched intro, redirect_from, and quoting, which hides link fixes +// in bulk runs. export function serializeMarkdown( rawContent: string, content: string, @@ -830,8 +682,7 @@ export function serializeMarkdown( newData: Record | undefined, differentData: boolean, ): string { - // `content` is the tail of the file, so everything before it is the frontmatter - // block exactly as the author wrote it, delimiters and all. + // content is the file tail, so everything before it is the original frontmatter block. if (!differentData && rawContent.endsWith(content)) { return rawContent.slice(0, rawContent.length - content.length) + newContent } diff --git a/src/links/lib/validate-docs-urls.ts b/src/links/lib/validate-docs-urls.ts index 1fd3ba9f23cf..736468f91d4d 100644 --- a/src/links/lib/validate-docs-urls.ts +++ b/src/links/lib/validate-docs-urls.ts @@ -31,10 +31,9 @@ export type Check = { fragment: string | undefined fragmentFound?: boolean fragmentCandidates?: string[] - // If the URL led to a redirect, this is its URL (starting with /en/...) + // Redirect destination with the /en prefix, when the source URL redirects. redirectPageURL?: string - // If the URL led to a redirect, this is what the new URL should be - // (for example /the/new/pathname#my-fragment) + // Suggested replacement URL without /en, preserving any fragment. redirect?: string } @@ -49,8 +48,7 @@ export async function validateDocsUrl(docsUrls: DocsUrls, { checkFragments = fal throw new Error(`URL doesn't start with '/': ${url} (identifier: ${identifier})`) } const pathname = url.split('?')[0] - // If the url is just '/' we want to check the homepage, - // which is `/en`, not `/en/`. + // The homepage resolves at /en, not /en/. const [pageURL, fragment] = `/en${pathname === '/' ? '' : pathname}`.split('#') const page = pages[pageURL] @@ -69,7 +67,7 @@ export async function validateDocsUrl(docsUrls: DocsUrls, { checkFragments = fal pages, }) if (redirect && isEnterpriseCloudRedirectOnly(pageURL, redirect)) { - // Ignore this one. It just added enterprise-cloud@latest to the URL. + // Skip redirects that only add the Enterprise Cloud prefix; they are not reported. continue } if (redirect) { @@ -108,8 +106,7 @@ export async function validateDocsUrl(docsUrls: DocsUrls, { checkFragments = fal } function isEnterpriseCloudRedirectOnly(originalUrl: string, redirectUrl: string) { - // A lot of URLs don't work in free-pro-team so all they do is redirect - // from {OLD-URL} to "/enterprise-cloud@latest/{OLD-URL}" + // Many URLs only redirect by adding /enterprise-cloud@latest under free-pro-team. return redirectUrl.replace('/enterprise-cloud@latest', '') === originalUrl } @@ -123,8 +120,7 @@ async function renderInnerHTML(page: Page, permalink: Permalink) { language: permalink.languageCode, pagePath, cookies: {}, - // The contextualize() middleware will create a new one. - // Here it just exists for the sake of TypeScript. + // contextualize replaces this placeholder, but TypeScript needs the key upfront. context: {}, } await contextualize(req as ExtendedRequest, res as Response, next) diff --git a/src/links/lib/validate-redirected-fragment.ts b/src/links/lib/validate-redirected-fragment.ts index 5682cf83e96d..3f55991a7b52 100644 --- a/src/links/lib/validate-redirected-fragment.ts +++ b/src/links/lib/validate-redirected-fragment.ts @@ -1,20 +1,9 @@ -/** - * When `update-internal-links` rewrites a link's path via a redirect, the original - * URL fragment (`#anchor`) is carried over from the old target. If the destination - * page doesn't have that heading, the anchor is silently stale and lands the reader - * at the top of the page. - * - * This validator renders the destination page through Liquid (once per version, cached) - * and checks whether the anchor exists in the page's real heading IDs, so the caller - * can decide whether to keep or drop the fragment. - * - * Correctness note: dropping a fragment is a destructive edit, so we only ever return - * `'drop'` when the anchor is absent from EVERY applicable version's render. If we can't - * render the page faithfully (render error, autogenerated page, unknown version), we - * return `'unvalidatable'` so the caller keeps the fragment rather than risk a false drop. - * - * Towards github/docs-engineering#6738. - */ +// update-internal-links carries URL fragments across redirect path rewrites. +// If the destination lacks the old heading, the stale fragment sends readers to the page top. +// Renders each applicable destination version through Liquid, with caching, +// and checks the fragment against real heading IDs before the caller keeps or drops it. +// Dropping a fragment is destructive, so drop only when every applicable version lacks it. +// Render errors, autogenerated pages, and unknown versions stay unvalidatable to avoid false drops. import warmServer from '@/frame/lib/warm-server' import { allVersions } from '@/versions/lib/all-versions' import { getFeaturesByVersion } from '@/versions/middleware/features' @@ -25,10 +14,10 @@ import type { Context, Page } from '@/types' const logger = createLogger(import.meta.url) -// 'keep' → anchor exists in every applicable version → leave the fragment. -// 'drop' → anchor is absent from every applicable version → safe to remove. -// 'mixed' → anchor exists in some but not all versions → keep, but flag for review. -// 'unvalidatable' → couldn't faithfully render the destination → keep, don't risk a drop. +// keep means the anchor exists in every applicable version, so leave the fragment. +// drop means the anchor is absent from every applicable version, so removal is safe. +// mixed means some versions have the anchor and some do not, so keep it and flag review. +// unvalidatable means faithful rendering failed, so keep the fragment. export type FragmentDecision = 'keep' | 'drop' | 'mixed' | 'unvalidatable' export class RedirectedFragmentValidator { @@ -41,9 +30,7 @@ export class RedirectedFragmentValidator { private language = 'en', ) {} - // Rendering a page's Liquid faithfully (e.g. `{% data variables.x %}` inside a heading) - // depends on the site data being loaded. Warm it lazily so callers that never hit a - // redirected fragment don't pay the cost. + // Faithful Liquid rendering needs site data, so warm lazily only for redirected fragments. private ensureWarmed(): Promise { if (!this.warmPromise) { this.warmPromise = warmServer([this.language]) @@ -51,7 +38,8 @@ export class RedirectedFragmentValidator { return this.warmPromise } - // Overridable in tests so the decision logic can be exercised without rendering. + // Tests override headingIdsFor so decision cases avoid rendering. + // Use renderMarkdownLiquid, not renderAndExtractLinks, to avoid raw fallback false drops. protected async headingIdsFor(page: Page, version: string): Promise | null> { const cacheKey = `${version}\u0000${page.relativePath}` const cached = this.headingIdCache.get(cacheKey) @@ -75,11 +63,6 @@ export class RedirectedFragmentValidator { page, ...getFeaturesByVersion(version), } as unknown as Context - // Use renderMarkdownLiquid, NOT renderAndExtractLinks: the latter swallows Liquid - // render failures and falls back to the raw markdown, which would yield heading IDs - // computed from unrendered `{% data %}` expressions. Those never match a real anchor, - // so classify() would return 'drop' and destructively delete a valid fragment. Here a - // render failure throws, lands in the catch below, and becomes 'unvalidatable'. const renderedMarkdown = await renderMarkdownLiquid(page.markdown, context) ids = computeHeadingIds(renderedMarkdown) } catch (error) { @@ -94,27 +77,18 @@ export class RedirectedFragmentValidator { return ids } - /** - * Decide what to do with a fragment carried over onto a redirected destination page. - * `fragment` may include the leading `#`. - */ + // fragment can include the leading # from the redirected link. async classify(destPage: Page | undefined, fragment: string): Promise { - // Can't find the destination page (redirect to an archived/external target, or a - // lookup miss) → we have nothing to validate against, so keep the fragment. + // Keep fragments for missing destination pages because no heading set exists. if (!destPage) return 'unvalidatable' - // Autogenerated pages (REST, GraphQL, webhooks) build their headings from data files - // at render time in a way `computeHeadingIds` on the source markdown can't see, so we - // can't reliably tell whether the anchor exists. Keep it and let the human/checker decide. + // Autogenerated REST, GraphQL, and webhook pages build headings from data at render time. if (destPage.autogenerated) return 'unvalidatable' const anchor = fragment.replace(/^#/, '') if (!anchor) return 'keep' - // `#top` is always valid: per the HTML spec a browser scrolls to the top of the - // document when nothing carries that ID, so it never appears in computed heading IDs - // and would otherwise always classify as 'drop'. Mirrors the same-page anchor check - // in check-links-internal.ts, which skips `#` and `#top`. + // #top mirrors check-links-internal.ts: browsers scroll to the top when no element has it. if (anchor === 'top') return 'keep' const versions = destPage.applicableVersions || [] @@ -123,7 +97,7 @@ export class RedirectedFragmentValidator { let presentCount = 0 for (const version of versions) { const ids = await this.headingIdsFor(destPage, version) - // If even one applicable version fails to render, don't risk a destructive drop. + // Any render failure keeps the fragment, avoiding a destructive drop. if (ids === null) return 'unvalidatable' if (ids.has(anchor)) presentCount++ } diff --git a/src/links/tests/cross-page-anchors.ts b/src/links/tests/cross-page-anchors.ts index e9615ff39e9c..52cd026fd594 100644 --- a/src/links/tests/cross-page-anchors.ts +++ b/src/links/tests/cross-page-anchors.ts @@ -29,8 +29,7 @@ describe('validateCrossPageAnchors (pass 2)', () => { }) test('skips (does not flag) a target with no cache entry', () => { - // A missing cache entry means the target resolved to a version not covered by this - // run, or is an autogenerated page. Both are validated elsewhere. + // Missing cache entries cover uncovered versions and autogenerated pages. const emptyCache = new Map>() expect(validateCrossPageAnchors([anchor()], emptyCache)).toEqual([]) }) @@ -58,9 +57,7 @@ describe('validateCrossPageAnchors (pass 2)', () => { }) test('does not flag a cross-page #top link', () => { - // Mirrors the same-page anchor exemption for `#` / `#top`. Browsers scroll to the - // top of the document for `#top` when nothing carries that ID, so it is always - // valid even though it never appears in the target's computed heading IDs. + // Browsers scroll to the top for #top, so computed heading IDs never need to include it. const cache = new Map([['/en/some/target', new Set(['a-heading'])]]) const pending = [ anchor({ fragment: 'top', href: '/some/target#top' }), diff --git a/src/links/tests/extract-links.ts b/src/links/tests/extract-links.ts index 91d9994648c3..6dc309e40285 100644 --- a/src/links/tests/extract-links.ts +++ b/src/links/tests/extract-links.ts @@ -108,7 +108,7 @@ Read [the docs](/docs/config) for more. ` const result = extractLinksFromMarkdown(content) - // 3 internal links: AUTOTITLE link, the image link (starts with /), and docs/config + // Internal image hrefs that start with / count as internal links too. expect(result.internalLinks.length).toBeGreaterThanOrEqual(2) expect(result.externalLinks).toHaveLength(1) expect(result.imageLinks).toHaveLength(1) @@ -136,8 +136,7 @@ Also [versioned](/enterprise-server@{{ currentVersion }}/admin). ` const result = extractLinksFromMarkdown(content) - // Extraction is regex-based, so the second link matches even with Liquid syntax - // inside it. Liquid rendering happens separately. + // Regex extraction can match Liquid syntax because rendering happens separately. expect(result.internalLinks.length).toBeGreaterThanOrEqual(0) }) @@ -187,7 +186,7 @@ Line 6 expect(result.internalLinks).toHaveLength(2) expect(result.internalLinks[0].line).toBe(2) - // Line numbers are preserved because code block content is replaced with spaces + // Code block content becomes spaces, preserving line numbers. expect(result.internalLinks[1].line).toBe(8) }) @@ -234,9 +233,7 @@ And [another real link](/another/real/path) here. }) test('does not mask links when backtick runs are mismatched', () => { - // Per CommonMark, a code span needs equal-length, maximal backtick runs on - // both ends. These lines have mismatched runs, so they are NOT code spans - // and the links between the backticks are real and must be extracted. + // Mismatched backtick runs are not CommonMark code spans, so the links remain real. const content = [ `A single-open, double-close: \`[one](/real/one)\`\``, `A double-open, triple-close: \`\`[two](/real/two)\`\`\``, @@ -247,8 +244,7 @@ And [another real link](/another/real/path) here. }) test('still masks links inside valid multi-backtick code spans', () => { - // A matched double-backtick run is a real code span, even when it wraps an - // inner single backtick, so the link inside must be ignored. + // A matched double-backtick run stays a code span even when it wraps a single backtick. const content = `Example: \`\` \`[skip](/placeholder)\` \`\` and see [the guide](/real/guide).` const result = extractLinksFromMarkdown(content) @@ -273,8 +269,7 @@ Broken: [AUTOTITLE](/code-security/create-custom-configuration. ` const result = extractLinksFromMarkdown(content) - // The unclosed link is not extracted, and it does not swallow the real link - // on the next line into a giant multi-line href. + // The unclosed link cannot swallow a real link on the next line. expect(result.internalLinks.map((l) => l.href)).toEqual(['/real/target']) }) @@ -484,7 +479,7 @@ describe('checkInternalLink', () => { }) test('finds redirect after stripping language prefix', () => { - // Links from rendered HTML have /en/ prefix but redirects are stored without it + // Rendered HTML links have the /en prefix, but redirects are stored without it. const result = checkInternalLink( '/en/enterprise-server@3.19/actions/old-path', pageMap, @@ -503,10 +498,7 @@ describe('checkInternalLink', () => { }) describe('version-aware resolution', () => { - // A non-FPT page has no versionless permalink, so a versionless link to it only - // resolves once you know which version is being checked. The versionless form is - // also in the redirect table as a fallback, which is what made these look like - // redirects that needed updating. + // Non-FPT links need source-version context; the redirect fallback otherwise misreports them. const versionedPageMap = { '/en/enterprise-server@3.21/billing/set-up-payment': {} as unknown as Page, '/en/actions/fpt-only': {} as unknown as Page, @@ -641,8 +633,7 @@ describe('checkInternalLink', () => { }) test('treats archived Enterprise Server versions as valid', () => { - // Deprecated GHES versions are served by the archived enterprise versions - // system, which isn't loaded into pageMap. They must not be reported broken. + // Archived Enterprise Server versions are valid even when pageMap does not load them. const result = checkInternalLink( '/enterprise-server@3.7/admin/release-notes', pageMap, @@ -663,8 +654,7 @@ describe('checkInternalLink', () => { }) test('resolves free-pro-team@latest prefixed links via the redirect resolver', () => { - // The flat redirects map has no literal key for this; getRedirect computes - // the correction (strip the version prefix) the same way production does. + // getRedirect strips the version prefix because the flat redirects map has no literal key. const result = checkInternalLink('/free-pro-team@latest/actions/guides', pageMap, redirects) expect(result.exists).toBe(true) expect(result.isRedirect).toBe(true) @@ -675,13 +665,11 @@ describe('checkInternalLink', () => { const result = checkInternalLink(`/enterprise-server/admin/overview`, pageMap, redirects) expect(result.exists).toBe(true) expect(result.isRedirect).toBe(true) - // Normalized to the latest stable Enterprise Server version. expect(result.redirectTarget).toBe(`/enterprise-server@${latestStable}/admin/overview`) }) test('strips hyphenated locale prefixes without double-prefixing', () => { - // /pt-br/ is a hyphenated locale; it must be stripped (not turned into - // /en/pt-br/...) so the underlying path resolves against the redirects map. + // Hyphenated locales such as /pt-br/ must strip cleanly before redirect lookup. const result = checkInternalLink('/pt-br/actions/legacy-path', pageMap, redirects) expect(result.exists).toBe(true) expect(result.isRedirect).toBe(true) @@ -689,8 +677,7 @@ describe('checkInternalLink', () => { }) test('normalizes a bare language-root redirect target to /', () => { - // getRedirect collapses '/free-pro-team@latest' to the language root ('/en'); - // after stripping the locale that would be empty, so it must normalize to '/'. + // getRedirect collapses /free-pro-team@latest to /en, which strips to empty. const result = checkInternalLink('/free-pro-team@latest', pageMap, redirects) expect(result.exists).toBe(true) expect(result.isRedirect).toBe(true) diff --git a/src/links/tests/heading-anchors.ts b/src/links/tests/heading-anchors.ts index f67a7c94565f..85c857ec49ba 100644 --- a/src/links/tests/heading-anchors.ts +++ b/src/links/tests/heading-anchors.ts @@ -57,7 +57,7 @@ describe('computeHeadingIds', () => { }) test('reproduces Liquid-rendered heading with variable already expanded', () => { - // After Liquid render, the variable is expanded to plain text; the full slug includes it. + // Liquid rendering expands the variable to plain text before slugging. const ids = computeHeadingIds( '## Disabling or enabling Copilot coding agent in your repositories', ) diff --git a/src/links/tests/link-report.ts b/src/links/tests/link-report.ts index 4bed90628b8d..5388463b35e9 100644 --- a/src/links/tests/link-report.ts +++ b/src/links/tests/link-report.ts @@ -166,8 +166,7 @@ describe('generateInternalLinkReport', () => { }) test('labels the title with version and language when supplied', () => { - // The workflow concatenates every version's report into one issue, so an - // unlabelled title leaves no way to tell the sections apart. + // Concatenated version reports need labelled titles so sections stay identifiable. const report = generateInternalLinkReport([{ href: '/broken', file: 'a.md', lines: [1] }], { version: 'enterprise-server@3.21', language: 'en', @@ -191,8 +190,7 @@ describe('createRedirectSuggestion', () => { ] test('does not tell authors to hardcode a version', () => { - // Following "update to the new path" here bakes 3.21 into content, which breaks - // as soon as 3.22 ships. + // Updating to the new path would bake 3.21 into content and break on the next release. const report = generateInternalLinkReport( linkTo('/admin/all-releases', '/enterprise-server@3.21/admin/all-releases'), ) @@ -463,8 +461,7 @@ describe('generatePRComment', () => { }) test('anchor wording tracks the blocking mode', () => { - // The comment must not claim the check is advisory once FAIL_ON_ANCHOR_FLAW flips it - // to failing, and vice versa. + // FAIL_ON_ANCHOR_FLAW controls whether anchor text says advisory or failing. const brokenAnchors = [ { href: '/a#x', file: 'content/a.md', lines: [1], versions: ['free-pro-team@latest'] }, ] @@ -577,7 +574,7 @@ describe('internal report grouped by fix strategy', () => { expect(markdown).toContain( 'npm run update-internal-links -- content/admin --keep-stale-fragments --dont-set-autotitle', ) - // content/issues only appears in the manual bucket, so it is not a codemod target. + // content/issues appears only in the manual bucket, so it is not a codemod target. expect(markdown).not.toContain('npm run update-internal-links -- content/issues ') }) @@ -693,7 +690,7 @@ describe('mergeInternalLinkReports', () => { const markdown = reportToMarkdown(merged) expect(markdown).toContain('**Only in:** ghes') - // The shared link breaks everywhere, so saying so on every group would be noise. + // Shared links that break everywhere omit per-group version labels to reduce noise. expect(markdown).not.toContain('**Only in:** fpt, ghes') }) }) @@ -719,7 +716,7 @@ describe('describeVersions', () => { describe('codemod table truncation', () => { const links: BrokenLink[] = [ - // `/old-0` appears in five files, so it should survive truncation. + // /old-0 appears in five files, so it must survive truncation. ...Array.from({ length: 5 }, (_, f) => ({ href: '/old-0', file: `actions/busy-${f}.md`, @@ -810,7 +807,7 @@ describe('version-only redirects', () => { describe('section caps', () => { test('caps stale anchors, keeping the busiest ones and counting the rest', () => { const anchors: BrokenLink[] = [ - // `/page-0#gone` appears in four files, so it must survive the cut. + // /page-0#gone appears in four files, so it must survive the cut. ...Array.from({ length: 4 }, (_, f) => ({ href: '/page-0#gone', file: `actions/busy-${f}.md`, @@ -857,8 +854,7 @@ describe('section caps', () => { describe('version-only classification across versions', () => { test('a link that is version-only in one version and renamed in another is codemod work', () => { - // Merged reports put every version's occurrences in one group. Classifying on the first - // redirect target alone would file this under "no action" and hide the rename. + // Classifying only the first merged redirect target would hide the later rename. const mixed: BrokenLink[] = [ { href: '/admin/overview', @@ -906,8 +902,7 @@ describe('version-only classification across versions', () => { describe('per-link file list cap', () => { test('caps the file table under one link and counts the rest', () => { - // Nothing bounds how many pages reuse a link, so one popular link could otherwise - // fill the whole issue body. + // One popular link could otherwise fill the whole issue body. const many: BrokenLink[] = Array.from({ length: 30 }, (_, i) => ({ href: '/page#gone', file: `actions/page-${i}.md`, @@ -935,7 +930,7 @@ describe('versions checked when some come back clean', () => { generateInternalLinkReport([{ href, file: 'actions/a.md', lines: [1] }]) test('a caller-supplied version list wins over what was found on disk', () => { - // A clean version uploads no report, so counting files undercounts the matrix. + // Clean versions upload no report, so counting report files undercounts the matrix. const merged = mergeInternalLinkReports( [ { version: 'free-pro-team@latest en', report: report('/a#gone') }, @@ -946,7 +941,7 @@ describe('versions checked when some come back clean', () => { expect(merged.versionsChecked).toHaveLength(3) expect(merged.summary).toContain('Checked 3 versions') - // Two of three versions is now worth saying out loud, where two of two was not. + // Two of three checked versions merit a version-specific label. expect(reportToMarkdown(merged)).toContain('**Only in:**') }) diff --git a/src/links/tests/page-anchors.ts b/src/links/tests/page-anchors.ts index ee8135ea84b5..d38a18713d78 100644 --- a/src/links/tests/page-anchors.ts +++ b/src/links/tests/page-anchors.ts @@ -73,7 +73,7 @@ describe('findLinkLines', () => { }) test('does not match a fragment that merely starts with the needle', () => { - // A substring search would report line 1 for `/a#foo` because `/a#foobar` starts with it. + // A substring search would report /a#foobar as a match for /a#foo. const content = ['[x](/a#foobar)', '[y](/a#foo)'].join('\n') expect(findLinkLines(content, '/a#foo')).toEqual([2]) }) @@ -106,8 +106,7 @@ describe('findLinkLines', () => { }) describe('resolveLinkKeyForVersion', () => { - // A GHEC-only target has no unversioned permalink, so a bare href only resolves when - // retried against the source page's version. + // GHEC-only targets need source-version retry because they have no unversioned permalink. const pageMap = { '/en/get-started/foo': {} as Page, '/en/enterprise-cloud@latest/admin/bar': {} as Page, @@ -150,9 +149,7 @@ describe('resolveLinkKeyForVersion', () => { expect(resolveLinkKeyForVersion('/nope/nope', 'free-pro-team@latest', pageMap)).toBe(null) }) - // A target that applies to both FPT and an enterprise version has a key for each. - // The enterprise key has to win during that version's run, otherwise the anchor is - // looked up under the FPT key while the heading cache holds the enterprise permalink. + // Shared targets have FPT and enterprise keys; enterprise runs must prefer the latter. const sharedPageMap = { '/en/get-started/shared': {} as Page, '/en/enterprise-server@3.17/get-started/shared': {} as Page, diff --git a/src/links/tests/update-internal-links-yaml.ts b/src/links/tests/update-internal-links-yaml.ts index 05cd61bc63ca..08d0ec3a8806 100644 --- a/src/links/tests/update-internal-links-yaml.ts +++ b/src/links/tests/update-internal-links-yaml.ts @@ -76,8 +76,7 @@ describe('serializeYaml', () => { expect(result).toContain('/new/path') }) - // The text lives in `newContent` and the structured links live in `newData`, and there - // is no format-preserving way to merge them. Silently picking one loses the other. + // Text and structured link changes cannot merge format-preservingly, so picking one loses data. test('throws rather than silently dropping fixes when both changed', () => { expect(() => serializeYaml(RELEASE_NOTE, { featuredLinks: { guide: '/new/path' } }, true, true), @@ -114,7 +113,7 @@ describe('rewriting links in YAML', () => { } test('rewrites a link hidden by indentation, which reads as a code block', async () => { - // Ten spaces of indent makes mdast see a code block, not a paragraph with a link. + // Ten spaces of indent makes mdast read a code block, not a paragraph with a link. const yaml = `sections: bugs: - | @@ -129,10 +128,9 @@ describe('rewriting links in YAML', () => { expect(result.replacements).toHaveLength(2) }) - // A `#` comment reads as a Markdown heading, so a link inside one is a real link node - // and does get rewritten. That is a documented limit, not corruption: telling the two - // apart needs a YAML parse, and a stale link in a comment is worth fixing anyway. The - // cases that would be corruption, code examples, are covered below. + // YAML comments read as Markdown headings, so links inside them are real link nodes. + // Distinguishing them needs a YAML parse, and stale links in comments are worth fixing. + // Code-example corruption is tested below. test('rewrites a link in a comment, but only there', async () => { const yaml = `# TODO: drop [x](/admin/old-path) from the copy below sections: diff --git a/src/links/tests/update-internal-links.ts b/src/links/tests/update-internal-links.ts index d2e369186d53..588102187c23 100644 --- a/src/links/tests/update-internal-links.ts +++ b/src/links/tests/update-internal-links.ts @@ -17,16 +17,15 @@ versions: Body text with a [link](/en/old-path). ` -// `frontmatter()` only omits `content` when the YAML fails to parse, which none of -// these fixtures do. Narrow it once here so each test can stay readable. +// These fixtures parse as YAML, so content exists and tests can use this narrower type. function parse(raw: string): { content: string; data: Record } { const { content, data } = frontmatter(raw) if (content === undefined) throw new Error('fixture failed to parse') return { content, data: data || {} } } -// The whole fix rests on gray-matter's `content` being an exact suffix of the raw file. -// Test that against the real parser, not a hand-rolled stand-in. +// serializeMarkdown depends on gray-matter content being an exact suffix of the raw file. +// Test the real parser, not a hand-rolled stand-in. describe('frontmatter parse invariant', () => { const cases: [string, string][] = [ ['standard page', PAGE], @@ -71,7 +70,7 @@ describe('serializeMarkdown', () => { expect(result).toContain('/en/new-path') expect(result).not.toContain('/en/old-path') - // The single-quoted intro and the list-style redirect_from must survive untouched. + // Single-quoted intro and list-style redirect_from must survive untouched. expect(result).toContain( "intro: 'You can allow contributors with push access to merge their pull requests", ) diff --git a/src/links/tests/validate-redirected-fragment.ts b/src/links/tests/validate-redirected-fragment.ts index be079ac0b340..50157f074972 100644 --- a/src/links/tests/validate-redirected-fragment.ts +++ b/src/links/tests/validate-redirected-fragment.ts @@ -4,8 +4,7 @@ import { RedirectedFragmentValidator } from '../lib/validate-redirected-fragment import { renderMarkdownLiquid } from '@/links/lib/extract-links' import type { Page } from '@/types' -// headingIdsFor renders the destination page's Liquid and warms the server to do it. Stub -// both so the render-failure path can be exercised without loading the site. +// Stub rendering and warming so render-failure tests avoid loading the site. vi.mock('@/links/lib/extract-links', () => ({ renderMarkdownLiquid: vi.fn(), })) @@ -13,8 +12,7 @@ vi.mock('@/frame/lib/warm-server', () => ({ default: vi.fn(async () => ({})), })) -// A test double that overrides the (expensive, render-based) heading lookup with a fixed -// map, so the keep/drop/mixed/unvalidatable decision logic can be exercised in isolation. +// Fixed heading IDs isolate keep, drop, mixed, and unvalidatable decisions from rendering. class StubValidator extends RedirectedFragmentValidator { constructor(private idsByKey: Record | null>) { super({}, {}, 'en') @@ -108,8 +106,7 @@ describe('RedirectedFragmentValidator.classify', () => { }) test('keep: #top is always valid', async () => { - // Browsers scroll to the top of the document for `#top` when nothing carries that ID, - // so it never appears in computed heading IDs and must not be treated as stale. + // Browsers scroll to the top for #top, so computed heading IDs never include it. const page = fakePage({}) const validator = new StubValidator({ 'free-pro-team@latest|content/foo.md': new Set(['intro']), @@ -128,10 +125,7 @@ describe('RedirectedFragmentValidator.classify', () => { describe('RedirectedFragmentValidator.headingIdsFor render failures', () => { test('unvalidatable, not drop, when the destination fails to render', async () => { - // Regression guard: renderAndExtractLinks swallows Liquid failures and returns the raw - // markdown, so heading IDs would be computed from unrendered `{% data %}` expressions. - // Those never match a real anchor, so classify() would return 'drop' and destructively - // delete a valid fragment. headingIdsFor must use the throwing renderer instead. + // headingIdsFor must use renderMarkdownLiquid, not renderAndExtractLinks, to avoid false drops. vi.mocked(renderMarkdownLiquid).mockRejectedValueOnce(new Error('Liquid syntax error')) const page = fakePage({ diff --git a/src/rest/scripts/utils/create-rest-examples.ts b/src/rest/scripts/utils/create-rest-examples.ts index 00d2d3bb173d..6a94996a55a6 100644 --- a/src/rest/scripts/utils/create-rest-examples.ts +++ b/src/rest/scripts/utils/create-rest-examples.ts @@ -1,16 +1,12 @@ import type { OpenApiMediaType } from './openapi-types' -// In the case that there are more than one example requests, and -// no content responses, a request with an example key that matches the -// status code of a response will be matched. const DEFAULT_EXAMPLE_DESCRIPTION = 'Example' const DEFAULT_EXAMPLE_KEY = 'default' const DEFAULT_ACCEPT_HEADER = 'application/vnd.github.v3+json' -// These functions only read the request body, parameters, and responses of an -// operation, so they accept a narrower shape than the full OpenApiOperation. -// Content maps are typed as `unknown` values (cast to OpenApiMediaType at the -// point of use) so the partial operation fixtures in tests remain assignable. +// These helpers accept partial operation shapes because they read only request bodies, +// parameters, and responses. Unknown content maps keep test fixtures assignable until +// each use casts the value to OpenApiMediaType. interface CodeSampleParameter { in?: string name: string @@ -70,10 +66,7 @@ export interface MergedExample { } } -// Retrieves request and response examples and attempts to -// merge them to create matching request/response examples -// The key used in the media type `examples` property is -// used to match requests to responses. +// getCodeSamples builds request and response examples, then applies merge rules before rendering. export default async function getCodeSamples( operation: CodeSampleOperation, ): Promise { @@ -82,8 +75,7 @@ export default async function getCodeSamples( const mergedExamples = mergeExamples(requestExamples, responseExamples) - // If there are multiple examples and if the request body - // has the same description, add a number to the example + // Duplicate descriptions get status-code suffixes so each docs example has a distinct label. if (mergedExamples.length > 1) { const count: Record = {} for (const item of mergedExamples) { @@ -110,28 +102,23 @@ export default async function getCodeSamples( return mergedExamples } +// mergeExamples applies direct, status-code, and example-key rules to pair requests with responses. +// If earlier rules do not apply, the fallback path matches request and response example keys. export function mergeExamples( requestExamples: RequestExample[], responseExamples: ResponseExample[], ): MergedExample[] { - // There is always at least one request example, but it won't create - // a meaningful example unless it has a response example. + // A lone request without a response cannot create a meaningful docs example. if (requestExamples.length === 1 && responseExamples.length === 0) { return [] } - // If there is one request and one response example, we don't - // need to merge the requests and responses, and we don't need - // to match keys directly. This allows falling back in the - // case that the existing OpenAPI schema has mismatched example keys. + // A single request and response pair directly, so mismatched OpenAPI example keys still render. if (requestExamples.length === 1 && responseExamples.length === 1) { return [{ ...requestExamples[0], response: responseExamples[0].response }] } - // If there is a request with no request body parameters and all of - // the responses have no content, then we can create a docs - // example for just status codes below 300. All other status codes will - // be listed in the status code table in the docs. + // A single request with example-less responses documents success status codes below 300. if ( requestExamples.length === 1 && responseExamples.length > 1 && @@ -142,10 +129,7 @@ export function mergeExamples( .map((ex) => ({ ...requestExamples[0], ...ex })) } - // If there is exactly one request example and one or more response - // examples, we can make a docs example for the response examples that - // have content. All remaining status codes with no content - // will be listed in the status code table in the docs. + // When one request has multiple responses, only responses with examples become docs examples. if ( requestExamples.length === 1 && responseExamples.length > 1 && @@ -156,17 +140,11 @@ export function mergeExamples( .map((ex) => ({ ...requestExamples[0], ...ex })) } - // Finally, we'll attempt to match examples with matching keys. - // This iterates through the longer array and compares key values to keys in - // the shorter array. const requestsExamplesLarger = requestExamples.length >= responseExamples.length const target = requestsExamplesLarger ? requestExamples : responseExamples const source = requestsExamplesLarger ? responseExamples : requestExamples - // Walk the longer array ("target", or the requests when the two are equal - // length) looking for a matching key in the other one ("source"). A request - // and a response with the same key are merged into one example. If several - // keys match, the first one wins. + // The longer list drives key matching. Requests win ties on length, and the first key match wins. return target .filter((targetEx) => { const match = source.find((srcEx) => srcEx.key === targetEx.key) @@ -176,17 +154,12 @@ export function mergeExamples( .map((ex) => ex as MergedExample) } -// Builds request examples from the media types in the operation's requestBody, -// falling back to a path-parameter or generic example when there is no body -// example. Every result has a key plus a request with description and -// acceptHeader; contentType, bodyParameters and parameters are optional. +// Request examples fall back to path parameters or generic examples when bodies lack examples. export function getRequestExamples(operation: CodeSampleOperation): RequestExample[] { const requestExamples: RequestExample[] = [] const parameterExamples = getParameterExamples(operation) - // When no request body or parameters are defined, we create a generic - // request example. Not all operations have request bodies or parameters, - // but we always want to show at least an example with the path. + // Operations without request bodies or path parameters still need a path-only example. if (!operation.requestBody && Object.keys(parameterExamples).length === 0) { return [ { @@ -199,7 +172,7 @@ export function getRequestExamples(operation: CodeSampleOperation): RequestExamp ] } - // When no request body exists, we create an example from the parameters + // Path parameter examples create requests when an operation has no request body. if (!operation.requestBody) { return Object.keys(parameterExamples).map((key) => { return { @@ -213,16 +186,10 @@ export function getRequestExamples(operation: CodeSampleOperation): RequestExamp }) } - // Requests can have multiple content types each with their own set of - // examples. for (const contentType of Object.keys(operation.requestBody.content)) { const mediaType = operation.requestBody.content[contentType] as OpenApiMediaType let examples: Record = {} - // This is a fallback to allow using the `example` property in - // the schema. If we start to enforce using examples vs. example using - // a linter, we can remove the check for `example`. - // For now, we'll use the key default, which is a common default - // example name in the OpenAPI schema. + // Treat a media type with a singular example field as examples under the default key. if (mediaType.example) { examples = { default: { @@ -232,7 +199,7 @@ export function getRequestExamples(operation: CodeSampleOperation): RequestExamp } else if (mediaType.examples) { examples = mediaType.examples } else { - // Example for this content type doesn't exist so we'll try and create one + // Missing media type examples still need a generic request for this content type. requestExamples.push({ key: DEFAULT_EXAMPLE_KEY, request: { @@ -245,13 +212,8 @@ export function getRequestExamples(operation: CodeSampleOperation): RequestExamp continue } - // There can be more than one example for a given content type. We need to - // iterate over the keys of the examples to create individual - // example objects for (const key of Object.keys(examples)) { - // A content type that includes `+json` is a custom media type - // The default accept header is application/vnd.github.v3+json - // Which would have a content type of `application/json` + // Custom +json media types must also become the Accept header. const acceptHeader = contentType.includes('+json') ? contentType : 'application/vnd.github.v3+json' @@ -272,9 +234,7 @@ export function getRequestExamples(operation: CodeSampleOperation): RequestExamp return requestExamples } -// Recursively removes the `example` and `examples` annotation fields from a -// JSON Schema object. Nothing at runtime reads them, and they account for -// ~131 MB of the total schema.json size across all versions. +// Strip unused example annotations because they add about 131 MB across versioned schemas. function stripSchemaExamples(schema: unknown): unknown { if (!schema || typeof schema !== 'object') return schema if (Array.isArray(schema)) return schema.map(stripSchemaExamples) @@ -287,23 +247,17 @@ function stripSchemaExamples(schema: unknown): unknown { return result } -// Builds examples for the operation's responses below status 400. Every result -// has a key plus a response with statusCode and description; contentType, -// example and schema are only present when the media type had an example. export function getResponseExamples(operation: CodeSampleOperation): ResponseExample[] { const responseExamples: ResponseExample[] = [] const responses = operation.responses as Record for (const statusCode of Object.keys(responses)) { - // We don't want to create examples for error codes - // Error codes are displayed in the status table in the docs + // Error responses already render in the docs status code table. if (parseInt(statusCode, 10) >= 400) continue const response = responses[statusCode] const content = response.content as Record | undefined - // A response doesn't always have content (ex:, status 304) - // In this case we create a generic example for the status code - // with a key that matches the status code. + // Responses without content still need a status-code example. if (!content) { const example = { key: statusCode, @@ -316,16 +270,10 @@ export function getResponseExamples(operation: CodeSampleOperation): ResponseExa continue } - // Responses can have multiple content types each with their own set of - // examples. for (const contentType of Object.keys(content)) { const mediaType = content[contentType] as OpenApiMediaType let examples: Record = {} - // This is a fallback to allow using the `example` property in - // the schema. If we start to enforce using examples vs. example using - // a linter, we can remove the check for `example`. - // We key by statusCode so that operations with multiple success - // responses (e.g. 200 + 201) get unique keys instead of colliding. + // Status-code keys prevent collisions for operations with success responses like 200 and 201. if (mediaType.example) { examples = { [statusCode]: { @@ -335,12 +283,7 @@ export function getResponseExamples(operation: CodeSampleOperation): ResponseExa } else if (mediaType.examples) { examples = mediaType.examples } else if (parseInt(statusCode, 10) < 300) { - // Sometimes there are missing examples for say a 200 response and - // the operation also has a 304 no content status. If we don't add - // the 200 response example, even though it has not example response, - // the resulting responseExamples would only contain the 304 response. - // That would be confusing in the docs because it's expected to see the - // common or success responses by default. + // Missing success examples still render so a 304 is not the only default example. const example = { key: statusCode, response: { @@ -351,15 +294,9 @@ export function getResponseExamples(operation: CodeSampleOperation): ResponseExa responseExamples.push(example) continue } else { - // Example for this content type doesn't exist. - // We could also check if there is a fully populated example - // directly in the response schema examples properties. continue } - // There can be more than one example for a given content type. We need to - // iterate over the keys of the examples to create individual - // example objects for (const key of Object.keys(examples)) { const example = { key, @@ -368,10 +305,7 @@ export function getResponseExamples(operation: CodeSampleOperation): ResponseExa contentType, description: examples[key].summary || response.description || '', example: examples[key].value, - // Note: Including the schema significantly increases JSON file size (~4x), - // but it's necessary to support the schema/example toggle in the UI. - // Users can switch between viewing the example response and the full schema. - // example/examples annotation fields are stripped as they are not rendered. + // Schema data makes JSON about 4x larger, but the UI needs the example/schema toggle. schema: stripSchemaExamples(mediaType.schema), }, } @@ -382,12 +316,8 @@ export function getResponseExamples(operation: CodeSampleOperation): ResponseExa return responseExamples } -// Groups the operation's path parameter values by example key, in the shape: -// -// { [example key]: { [parameter name]: value } } -// -// A parameter with no examples contributes its uppercased name under the -// `default` key. +// Path parameter example values are grouped by example key, then parameter name. +// Parameters without examples use uppercased names under default so fake route values stand out. export function getParameterExamples( operation: CodeSampleOperation, ): Record> { @@ -398,9 +328,6 @@ export function getParameterExamples( const parameterExamples: Record> = {} for (const parameter of parameters) { const examples = parameter.examples - // If there are no examples, create an example from the uppercase parameter - // name, so that it is more visible that the value is fake data - // in the route path. if (!examples) { if (!parameterExamples.default) parameterExamples.default = {} parameterExamples.default[parameter.name] = parameter.name.toUpperCase() diff --git a/src/rest/scripts/utils/get-body-params.ts b/src/rest/scripts/utils/get-body-params.ts index e0061f7cf31c..d275af5b7f53 100644 --- a/src/rest/scripts/utils/get-body-params.ts +++ b/src/rest/scripts/utils/get-body-params.ts @@ -32,13 +32,11 @@ interface BodyParamProps { childParamsGroups?: TransformedParam[] } -// If there is a oneOf at the top level, then we have to present just one -// in the docs. We don't currently have a convention for showing more than one -// set of input parameters in the docs. Having a top-level oneOf is also very -// uncommon. -// Currently there aren't very many operations that require this treatment. -// As an example, the 'Add status check contexts' and 'Set status check contexts' -// operations have a top-level oneOf. +// Docs cannot display multiple input parameter sets for top-level oneOf. +// getTopLevelOneOfProperty uses the first option. The Add status check contexts +// and Set status check contexts operations need this. +// When every top-level oneOf option is an object, getTopLevelOneOfProperty merges +// all properties. With three or more options, middle required fields can lose required flags. async function getTopLevelOneOfProperty( schema: Schema, ): Promise<{ properties: Record; required: string[] }> { @@ -49,18 +47,12 @@ async function getTopLevelOneOfProperty( throw new Error('Schema requestBody oneOf property is not an array') } - // When a oneOf exists but the `type` differs, the case has historically - // been that the alternate option is an array, where the first option - // is the array as a property of the object. We need to ensure that the - // first option listed is the most comprehensive and preferred option. + // When oneOf types differ, the first option must be the comprehensive object form. const firstOneOfObject = schema.oneOf[0] const allOneOfAreObjects = schema.oneOf.every((elem) => elem.type === 'object') let required = firstOneOfObject.required || [] let properties = firstOneOfObject.properties || {} - // When all of the oneOf objects have the `type: object` we - // need to display all of the parameters. - // This merges all of the properties and required values. if (allOneOfAreObjects) { required = [] properties = {} @@ -76,7 +68,6 @@ async function getTopLevelOneOfProperty( return { properties, required } } -// Handles a oneOf whose items are all objects. Returns [] for anything else. async function handleObjectOnlyOneOf( param: Schema, paramType: string[], @@ -89,15 +80,19 @@ async function handleObjectOnlyOneOf( return [] } -// Gets the body parameters for a schema, recursively. +// OpenAPI 3.0 allows one type value, while OpenAPI 3.1 also allows an array, so getBodyParams +// normalizes type values to arrays before it builds the rendered type string. +// For child parameters, getBodyParams reads object-valued additionalProperties recursively. +// The Create a snapshot of dependencies for a repository and Update a gist operations need +// that dictionary shape. Object-only oneOf alternatives also recurse into child parameters, +// while mixed oneOf adds types and descriptions without creating child parameter groups. export async function getBodyParams(schema: Schema, topLevel = false): Promise { const bodyParametersParsed: TransformedParam[] = [] const schemaObject = schema.oneOf && topLevel ? await getTopLevelOneOfProperty(schema) : schema const properties = schemaObject.properties || {} const required = schemaObject.required || [] - // Most operation requestBody schemas are objects. When the type is an array, - // there will not be properties on the `schema` object. + // Top-level array schemas have no properties on the schema object. if (topLevel && schema.type === 'array') { const childParamsGroups: TransformedParam[] = [] if (!schema.items) { @@ -118,11 +113,6 @@ export async function getBodyParams(schema: Schema, topLevel = false): Promise t !== undefined, ) @@ -134,13 +124,6 @@ export async function getBodyParams(schema: Schema, topLevel = false): Promise (item as Schema).type === 'object', @@ -248,7 +226,7 @@ export async function getBodyParams(schema: Schema, topLevel = false): Promise { const { paramKey, required, childParamsGroups } = props const paramDecorated: TransformedParam = {} as TransformedParam - // Supports backwards compatibility for OpenAPI 3.0 - // In 3.1 a nullable type is part of the param.type array and - // the property param.nullable does not exist. + // OpenAPI 3.0 stores nullable separately from OpenAPI 3.1 type arrays. if (param.nullable) paramType.push('null') paramDecorated.type = Array.from(new Set(paramType.filter(Boolean))).join(' or ') paramDecorated.name = paramKey || '' @@ -284,8 +260,7 @@ async function getTransformedParam( paramDecorated.isRequired = true } if (childParamsGroups && childParamsGroups.length > 0 && !param.oneOfObject) { - // allOf can contribute the same property more than once. Drop the - // duplicates by name, keeping whichever one has isRequired set. + // Drop duplicate allOf child params by name, preferring required entries. const mergedChildParamsGroups = Array.from( childParamsGroups .reduce((childParam, obj) => { diff --git a/src/rest/scripts/utils/tests/get-body-params.test.ts b/src/rest/scripts/utils/tests/get-body-params.test.ts index 826ad950bbc9..ba2e1aecba23 100644 --- a/src/rest/scripts/utils/tests/get-body-params.test.ts +++ b/src/rest/scripts/utils/tests/get-body-params.test.ts @@ -2,15 +2,12 @@ import { describe, expect, it, vi } from 'vitest' import { getBodyParams, type Schema } from '@/rest/scripts/utils/get-body-params' -// Mock render-content so tests don't require the full content-render pipeline +// renderContent returns input because these tests cover schema transformation, not rendering. vi.mock('../render-content', () => ({ renderContent: async (template: string) => template, })) describe('getBodyParams — OAS 3.1 nullable handling', () => { - // ── Bug #3 ────────────────────────────────────────────────────────────────── - // anyOf: [{type:"null"}, {type:"object"}] → type should render as "object or null" - it('renders anyOf [{type:"null"},{type:"object"}] as "object or null"', async () => { const schema = { type: 'object', @@ -58,8 +55,7 @@ describe('getBodyParams — OAS 3.1 nullable handling', () => { expect(params[0].type).toBe('object or null') }) - it('renders anyOf [{type:"string"},{type:"null"}] as "string" using the first option fallback', async () => { - // When anyOf has no object, it uses the existing fallback: param.anyOf[0].type + it('renders anyOf [{type:"string"},{type:"null"}] as "string" (no object, falls back to anyOf[0])', async () => { const schema = { type: 'object', properties: { @@ -71,7 +67,6 @@ describe('getBodyParams — OAS 3.1 nullable handling', () => { const params = await getBodyParams(schema, false) expect(params).toHaveLength(1) expect(params[0].name).toBe('label') - // No object found in anyOf → falls back to anyOf[0].type = 'string' expect(params[0].type).toBe('string') }) @@ -109,10 +104,6 @@ describe('getBodyParams — OAS 3.1 nullable handling', () => { ]) }) - // ── OAS 3.1 type: ["string", "null"] scalar ───────────────────────────── - // This is already handled by existing code (paramType array normalization). - // These tests verify the existing OAS 3.1 scalar nullable path still works. - it('renders type: ["string", "null"] as "string or null"', async () => { const schema = { type: 'object', @@ -144,7 +135,6 @@ describe('getBodyParams — OAS 3.1 nullable handling', () => { expect(params[0].type).toBe('integer or null') }) - // ── anyOf without null ──────────────────────────────────────────────────── it('renders anyOf [{type:"object"}] without null (no hasNull) as just "object"', async () => { const schema = { type: 'object', @@ -165,11 +155,9 @@ describe('getBodyParams — OAS 3.1 nullable handling', () => { const params = await getBodyParams(schema, false) expect(params).toHaveLength(1) expect(params[0].type).toBe('object') - // Confirm "null" is NOT in the type expect(params[0].type).not.toContain('null') }) - // ── Existing OAS 3.0 nullable path still works ─────────────────────────── it('still handles OAS 3.0 nullable: true', async () => { const schema = { type: 'object', @@ -186,7 +174,6 @@ describe('getBodyParams — OAS 3.1 nullable handling', () => { expect(params[0].type).toBe('string or null') }) - // ── Normal non-nullable object body params ─────────────────────────────── it('renders a plain string param without null', async () => { const schema = { type: 'object', diff --git a/src/search/lib/ai-search-constants.ts b/src/search/lib/ai-search-constants.ts index 15f0a6e297c0..2f4f93c5f55e 100644 --- a/src/search/lib/ai-search-constants.ts +++ b/src/search/lib/ai-search-constants.ts @@ -1,10 +1,7 @@ -// Maximum query length (chars) we will forward to cse-copilot. Larger -// payloads are almost always pasted docs pages or unrelated content -// and either time out or return no-answer. See github/cse-copilot#1214. +// Forward at most 500 chars to cse-copilot because 15k-60k pasted inputs +// caused timeouts and no-answer responses. export const MAX_QUERY_LENGTH = 500 -// cse-copilot returns HTTP 400 with this code in `detail.code` when Azure's -// Responsible AI input content filter rejects a query. These are expected, -// user-triggered rejections, so they are tracked separately and kept out of -// the AI search error-rate monitor. See github/cse-copilot#1214. +// Azure Responsible AI input filtering returns this detail.code for expected +// user-triggered rejections, so the error-rate monitor excludes them. export const RAI_CONTENT_FILTER_CODE = 'RAI_INPUT_CONTENT_POLICY_BREACH_ERROR' diff --git a/src/search/lib/ai-search-proxy.ts b/src/search/lib/ai-search-proxy.ts index 84f083a69ae7..8b6d4761b850 100644 --- a/src/search/lib/ai-search-proxy.ts +++ b/src/search/lib/ai-search-proxy.ts @@ -10,9 +10,8 @@ import { MAX_QUERY_LENGTH, RAI_CONTENT_FILTER_CODE } from '@/search/lib/ai-searc const logger = createLogger(import.meta.url) -// Maximum time (ms) to wait for the initial response from the upstream -// AI search service. Streaming may take longer once the connection is -// established, but the connect + first-byte must complete within this window. +// The timeout covers connection and first byte; streaming may take longer +// after the response arrives. const AI_SEARCH_TIMEOUT_MS = 9_000 type ContentFilterCandidate = { @@ -159,7 +158,7 @@ export const aiSearchProxy = async (req: ExtendedRequest, res: Response) => { } const totalResponseTime = Date.now() - startTime // in ms - const charPerMsRatio = totalResponseTime > 0 ? totalChars / totalResponseTime : 0 // chars per ms + const charPerMsRatio = totalResponseTime > 0 ? totalChars / totalResponseTime : 0 statsd.gauge('ai-search.total_response_time', totalResponseTime, diagnosticTags) statsd.gauge('ai-search.response_chars_per_ms', charPerMsRatio, diagnosticTags) @@ -196,7 +195,6 @@ export const aiSearchProxy = async (req: ExtendedRequest, res: Response) => { res.status(500).json({ errors: [{ message: 'Internal server error' }] }) } } finally { - // Ensure reader lock is always released if (reader) { reader.releaseLock() } diff --git a/src/search/lib/elasticsearch-indexes.ts b/src/search/lib/elasticsearch-indexes.ts index ec4be64c5a61..e75ad30a0b36 100644 --- a/src/search/lib/elasticsearch-indexes.ts +++ b/src/search/lib/elasticsearch-indexes.ts @@ -13,15 +13,13 @@ export type SearchIndex = { type: string } -// The source of truth for Docs Elasticsearch indexes. -// -// There are two top-level categories: -// 1. General search, populated from all of our Docs pages. -// 2. AI autocomplete, populated with human-readable questions from a GPT -// query in docs-internal-data. +// Docs Elasticsearch indexes have two categories: general search, populated +// from all Docs pages, and AI autocomplete, populated with human-readable +// questions from a GPT query in docs-internal-data. // // Index names take the form ___, -// e.g. github-docs_general-search_fpt_en. is "tests_" in tests. +// for example github-docs_general-search_fpt_en. Tests use tests_ as +// . const prefix = 'github-docs' const indexes: SearchIndexes = { generalSearch: { @@ -34,7 +32,6 @@ const indexes: SearchIndexes = { }, } -// Source of truth for determining the index name for the Elastic Search index given a version and language export function getElasticSearchIndex( type: SearchTypes, version: string, @@ -61,17 +58,15 @@ export function getElasticSearchIndex( ) } - // e.g. free-pro-team becomes fpt for the index name + // free-pro-team maps to fpt in index names. let indexVersion = versionToIndexVersionMap[version] - // For AI Search autocomplete, we use the latest GHES version for all GHES versions. - // This provides AI search functionality across all supported GHES versions without - // requiring separate indexes for each version. + // AI autocomplete shares the latest GHES index across all supported GHES versions. if (type === 'aiSearchAutocomplete' && indexVersion.startsWith('ghes')) { indexVersion = versionToIndexVersionMap['enterprise-server'] } - // In the index-test-fixtures.sh script, we use the tests_ prefix index for testing + // index-test-fixtures.sh expects the tests_ prefix. const testPrefix = process.env.NODE_ENV === 'test' ? 'tests_' : '' if (manualPrefix && !manualPrefix.endsWith('_')) { diff --git a/src/search/lib/elasticsearch-versions.ts b/src/search/lib/elasticsearch-versions.ts index 256437c17dc7..69c4e4ea6958 100644 --- a/src/search/lib/elasticsearch-versions.ts +++ b/src/search/lib/elasticsearch-versions.ts @@ -1,17 +1,6 @@ -// The source of truth for versioning in the context of Elasticsearch. It maps -// every accepted version identifier to the version segment of an index name. -// Several identifiers can share one segment. -// -// Example versions (these may not be up to date): -// -// 1. free-pro-team@latest, previously known as "dotcom", the default version. -// Short name: fpt -// 2. enterprise-cloud@latest. Short name: ghec -// 3. enterprise-server@X, the source of the complexity because the version is -// dynamic. Short name: ghes-X -// -// For (3) someone might pass `&version=3.5` in the request query string, which -// maps to `ghes-3.5`. +// Elasticsearch index names use compact version segments. Multiple accepted +// request identifiers can share one segment, such as free-pro-team, dotcom, +// and free-pro-team@latest all mapping to fpt. import { allVersions } from '@/versions/lib/all-versions' @@ -20,51 +9,44 @@ import { allVersions } from '@/versions/lib/all-versions' // free-pro-team -> fpt // dotcom -> fpt // enterprise-cloud@latest -> ghec -// enterprise-server@3.5 -> ghes-3.5 -// 3.5 -> ghes-3.5 +// enterprise-server@X -> ghes-X +// X -> ghes-X export const versionToIndexVersionMap: { [key: string]: string } = {} -// For each potential input (from request query string, CLI, etc), map it to the appropriate index version for (const versionSource of Object.values(allVersions)) { if (versionSource.hasNumberedReleases) { - // Map version number to corresponding release, e.g. `3.14` -> `ghes-3.14` versionToIndexVersionMap[versionSource.currentRelease] = versionSource.miscVersionName - // Map full release name to corresponding release, e.g. `enterprise-server@3.14` -> `ghes-3.14` versionToIndexVersionMap[versionSource.version] = versionSource.miscVersionName - // Map shortname or plan, e.g. `ghes` or `enterprise-server` to the latest release, e.g. `ghes-3.14` if (versionSource.latestRelease === versionSource.currentRelease) { + // The plan or short name, such as ghes or enterprise-server, maps to the latest release only. versionToIndexVersionMap[versionSource.plan] = versionSource.miscVersionName versionToIndexVersionMap[versionSource.shortName] = versionSource.miscVersionName } } else { versionToIndexVersionMap[versionSource.version] = versionSource.shortName versionToIndexVersionMap[versionSource.miscVersionName] = versionSource.shortName - // The next two lines map things like `?version=free-pro-team` -> `?version=fpt` + // Plan names accepted by request queries map to compact index names. versionToIndexVersionMap[versionSource.plan] = versionSource.shortName versionToIndexVersionMap[versionSource.shortName] = versionSource.shortName } } -// Add the values to the keys as well so that the map value -> value works for versions that are already conformed to the indexVersion syntax +// Compact-name aliases let already-normalized requests validate. for (const [, value] of Object.entries(versionToIndexVersionMap)) { versionToIndexVersionMap[value] = value } -// All of the possible keys that can be input to access a version export const allIndexVersionKeys = Array.from( new Set([...Object.keys(versionToIndexVersionMap), ...Object.keys(allVersions)]), ) -// These should be the only possible values that an ES index will use (source of truth) -// allIndexVersionOptions example: -// fpt, ghec, ghes-3.14, ghes-3.13, ghes-3.12, ghes-3.11, ghes-3.10 +// Elasticsearch indexes accept only compact segments such as fpt, ghec, and ghes-X. export const allIndexVersionOptions = Array.from( new Set([...Object.values(versionToIndexVersionMap)]), ) -// Autocomplete only supports 3 "versions": free-pro-team, enterprise-cloud, and enterprise-server -// docs-internal-data stores data under directories with these names. It does not account for individual enterprise-server versions -// These are the "plan" names on the allVersions object +// Autocomplete data lives under plan directories: free-pro-team, +// enterprise-cloud, and enterprise-server. It does not split by GHES release. const allVersionPlans: string[] = [] for (const version of Object.values(allVersions)) { if (version.plan) { @@ -73,8 +55,8 @@ for (const version of Object.values(allVersions)) { } export const supportedAutocompletePlanVersions = Array.from(new Set(allVersionPlans)) -// Returns the plan name for the given version -// Needed because {version} in the docs-internal-data paths use the version's 'plan' name, e.g. `free-pro-team` instead of `fpt` +// docs-internal-data paths require plan names such as free-pro-team, not +// compact index names such as fpt. export function getPlanVersionFromIndexVersion(indexVersion: string): string { const planVersion = Object.values(allVersions).find( @@ -92,8 +74,7 @@ export function getPlanVersionFromIndexVersion(indexVersion: string): string { return planVersion } -// Gets the matching key from allVersions for the given index version -// This is needed for scraping since the pages use the 'allVersions' key as their version +// Scraping uses allVersions keys for page versions, not compact index names. export function getAllVersionsKeyFromIndexVersion(indexVersion: string): string { const key = Object.keys(allVersions).find( (versionKey) => diff --git a/src/search/lib/get-elasticsearch-results/ai-search-autocomplete.ts b/src/search/lib/get-elasticsearch-results/ai-search-autocomplete.ts index c6aec94628fd..0303ffe5347e 100644 --- a/src/search/lib/get-elasticsearch-results/ai-search-autocomplete.ts +++ b/src/search/lib/get-elasticsearch-results/ai-search-autocomplete.ts @@ -9,7 +9,6 @@ import type { } from '@/search/lib/get-elasticsearch-results/types' import type { estypes } from '@elastic/elasticsearch' -// Query Elasticsearch for AI Search autocomplete results export async function getAISearchAutocompleteResults({ indexName, query, @@ -22,12 +21,12 @@ export async function getAISearchAutocompleteResults({ const searchQuery: estypes.SearchRequest = { index: indexName, size, - // Send absolutely minimal from Elasticsearch to here. Less data => faster. + // Request only term values to keep autocomplete payloads small. _source_includes: ['term'], } const trimmedQuery = query.trim() - // When the query is empty, we want to return the top `size` most popular terms + // Empty queries return the most popular terms. if (trimmedQuery === '') { searchQuery.query = { match_all: {} } searchQuery.sort = [{ popularity: { order: 'desc' } }] @@ -120,7 +119,6 @@ function getAISearchAutocompleteMatchQueries( }, }) - // Add fuzzy matching for typos and variations if (query.length > fuzzy.minLength && query.length < fuzzy.maxLength) { matchQueries.push({ fuzzy: { diff --git a/src/search/lib/get-elasticsearch-results/general-search.ts b/src/search/lib/get-elasticsearch-results/general-search.ts index a6a987d58685..6fb21b2ba1d8 100644 --- a/src/search/lib/get-elasticsearch-results/general-search.ts +++ b/src/search/lib/get-elasticsearch-results/general-search.ts @@ -18,7 +18,6 @@ type getGeneralSearchResultsParams = { searchParams: ComputedSearchQueryParamsMap['generalSearch'] } -// Query Elasticsearch for general search results export async function getGeneralSearchResults( args: getGeneralSearchResultsParams, ): Promise { @@ -69,7 +68,7 @@ export async function getGeneralSearchResults( const matchBool: estypes.QueryDslBoolQuery = { should: matchQueries, - // This allows filtering by toplevel later. + // Filters make should clauses optional, so require a match before toplevel filters apply. minimum_should_match: 1, } const matchQuery: estypes.QueryDslQueryContainer = { @@ -88,7 +87,7 @@ export async function getGeneralSearchResults( } const highlightFields = Array.from(highlights || DEFAULT_HIGHLIGHT_FIELDS) - // These acts as an alias convenience + // content_explicit mirrors content highlight requests. if (highlightFields.includes('content')) { highlightFields.push('content_explicit') } @@ -102,12 +101,7 @@ export async function getGeneralSearchResults( from, size, aggs, - - // Since we know exactly which fields from the source we're going - // need we can specify that here. It's an inclusion list. - // We can save precious network by not having to transmit fields - // stored in Elasticsearch to here if it's not going to be needed - // anyway. + // Only requested source fields cross the network. _source_includes: ['title', 'url', 'breadcrumbs', 'popularity', 'toplevel'], } @@ -118,8 +112,7 @@ export async function getGeneralSearchResults( } if (sort === 'best') { - // To sort by a function score, you need to wrap the primary - // match query into a bool operation. + // function_score multiplies relevance by popularity for best-first ranking. searchQuery.query = { bool: { must: [ @@ -143,10 +136,7 @@ export async function getGeneralSearchResults( }, } } else if (sort === 'relevance') { - // Do nothing, it's the default. - // We could have a secondary sort on the 'popularity' but the - // chances of this ever doing anything is very weak because of the - // floating point almost always being different. + // Relevance sort skips popularity because near-unique scores make it ineffective. searchQuery.query = matchQuery } else { throw new Error(`Unrecognized sort enum '${sort}'`) @@ -222,6 +212,10 @@ interface GetMatchQueriesOptions { } } +// For autocomplete, getMatchQueries skips match_phrase_prefix on content. +// match_phrase_prefix matches preceding terms and expands the last word, so +// short category pages that list titles over-rank: +// https://www.elastic.co/guide/en/elasticsearch/reference/7.17/query-dsl-match-query-phrase-prefix.html#match-phrase-prefix-query-notes function getMatchQueries( query: string, { usePrefixSearch, fuzzy }: GetMatchQueriesOptions, @@ -232,40 +226,16 @@ function getMatchQueries( const BOOST_CONTENT = 1.0 const BOOST_AND = 2.5 const BOOST_EXPLICIT = 6.5 - // Number doesn't matter so much but just make sure it's - // boosted low. Because we only really want this to come into - // play if nothing else matches. E.g. a search for `AcIons` - // which wouldn't find anything else anyway. + // Fuzzy title matches get a low boost so exact and phrase matches dominate ranking. const BOOST_FUZZY = 0.1 const matchQueries: estypes.QueryDslQueryContainer[] = [] - // If the query input is multiple words, it's good to know because you can - // make the query do `match_phrase` and you can make `match` query - // with the `AND` operator (`OR` is the default). + // Spaces or hyphens enable phrase and AND-operator matching. const isMultiWordQuery = query.includes(' ') || query.includes('-') if (isMultiWordQuery) { - // If the query contains spaces, prioritize a "match phrase" query - // beyond a regular "match" query. - // Basically, that means if you search for 'foo bar' we'd rather - // rank: - // "A common term is foo bar which is often used" - // above: - // "Some people use foo" - // "Bar is also a common term" - // - // So that, when all are matched you get this rank: - // 1. "A common term is foo bar which is often used" - // 2. "Some people use foo" - // 3. "Bar is also a common term" - // - // But note, a "match phrase" isn't the holy panacea of matches. - // In particular, just because there exists a document whose *content* - // contains the phrase "... foo bar ..." we might still prefer the - // matches on title that contains the words *separately*. This - // is why a 'match_phrase' on 'content' has a lesser boost - // that a 'match' on 'title'. + // Ordinary title word matches beat ordinary content phrase matches. const matchPhraseStrategy = usePrefixSearch ? 'match_phrase_prefix' : 'match_phrase' matchQueries.push( ...[ @@ -283,13 +253,6 @@ function getMatchQueries( { [matchPhraseStrategy]: { headings: { boost: BOOST_PHRASE * BOOST_HEADINGS, query } } }, ], ) - // If the content is short, it is given a disproportionate advantage - // in search ranking. For example, our category and subcategory pages - // often includes a list of other document titles but because it's so - // short it thinks that content is really relevant. This only applies - // when you use `match_phrase_prefix` which first makes a search - // all preceeding terms and then manually appends matches on the last word. - // See https://www.elastic.co/guide/en/elasticsearch/reference/7.17/query-dsl-match-query-phrase-prefix.html#match-phrase-prefix-query-notes if (!usePrefixSearch) { matchQueries.push( ...[ @@ -304,7 +267,7 @@ function getMatchQueries( } } - // Unless the query was something like `"foo bar"` search on each word + // Quoted multi-word queries skip per-word matching so phrase search stays strict. if (!(isMultiWordQuery && query.startsWith('"') && query.endsWith('"'))) { const matchStrategy = usePrefixSearch ? 'match_bool_prefix' : 'match' if (isMultiWordQuery) { @@ -373,10 +336,7 @@ function getMatchQueries( ) } - // Add a fuzzy query if it's not too short or too long. - // Might consider only enabling this when there's no space in the query - // because something like "githob actions" will overwhelmingly - // match on the "actions" part with the regular 'match' query. + // Fuzzy matching applies only within the configured length bounds. if (query.length > fuzzy.minLength && query.length < fuzzy.maxLength) { matchQueries.push({ fuzzy: { @@ -385,22 +345,20 @@ function getMatchQueries( }) } - // If the query is just a single no-space word... + // Single-token URL searches also match page paths. if (query.split(/\s/g).length === 1) { - // E.g. someone searched for `/en/site-policy/github-company-policies` + // A path query such as /en/site-policy/github-company-policies matches url. if (query.startsWith('/')) { matchQueries.push({ match: { url: query.split('?')[0].split('#')[0] }, }) } else if (query.startsWith('http')) { - // E.g. `https://docs.github.com/en/some/page?foo=bar` - // will become a search on `{url: '/en/some/page'}` + // Full docs.github.com URLs match their pathname, such as /en/some/page. let pathname: string | undefined try { pathname = new URL(query).pathname } catch { - // If it failed, it can't be initialized with the `URL` constructor - // so we can deem it *not* a valid URL. + // Invalid URL strings do not add a url match. } if (pathname) { matchQueries.push({ @@ -433,14 +391,7 @@ function getHits( { indexName, debug = false, highlightFields, include }: GetHitsOptions, ): GeneralSearchHit[] { return hits.map((hit) => { - // Return `hit.highlights[...]` based on the highlight fields requested. - // So if you searched with `&highlights=headings&highlights=content` - // this will become: - // { - // content: [...], - // headings: [...] - // } - // even if there was a match on 'title'. + // Requested highlight fields get keys even when empty, so the response matches the request. const hitHighlights: Record = {} for (const key of highlightFields) { hitHighlights[key] = (hit.highlight && hit.highlight[key]) || [] diff --git a/src/search/lib/get-elasticsearch-results/helpers/elasticsearch-highlight-config.ts b/src/search/lib/get-elasticsearch-results/helpers/elasticsearch-highlight-config.ts index 72cd57d03d18..4d247ecad46b 100644 --- a/src/search/lib/get-elasticsearch-results/helpers/elasticsearch-highlight-config.ts +++ b/src/search/lib/get-elasticsearch-results/helpers/elasticsearch-highlight-config.ts @@ -14,7 +14,6 @@ export type HighlightFields = { [key in HighlightOptions]: HighlightConfig } -// When we query Elasticsearch, we can specify a highlight configuration export function getHighlightConfiguration( query: string, highlightsFields: HighlightOptions[], @@ -22,7 +21,7 @@ export function getHighlightConfiguration( const fields = {} as HighlightFields if (highlightsFields.includes('title')) { fields.title = { - // fvh requires the field to be indexed with {term_vector: 'with_positions_offsets'}. + // fvh requires term_vector: with_positions_offsets on the indexed field. type: 'fvh', fragment_size: 200, number_of_fragments: 1, @@ -30,11 +29,11 @@ export function getHighlightConfiguration( } if (highlightsFields.includes('content')) { fields.content = { - // fvh requires the field to be indexed with {term_vector: 'with_positions_offsets'}. + // fvh requires term_vector: with_positions_offsets on the indexed field. type: 'fvh', fragment_size: 150, number_of_fragments: 1, - // So we can at least display something if there was no highlight match within the content. + // Fallback snippets show text even when Elasticsearch finds no content highlight. no_match_size: 150, highlight_query: { @@ -46,7 +45,7 @@ export function getHighlightConfiguration( }, } fields.content_explicit = { - // fvh requires the field to be indexed with {term_vector: 'with_positions_offsets'}. + // fvh requires term_vector: with_positions_offsets on the indexed field. type: 'fvh', fragment_size: 150, number_of_fragments: 1, @@ -63,7 +62,7 @@ export function getHighlightConfiguration( } if (highlightsFields.includes('term')) { fields.term = { - // fvh requires the field to be indexed with {term_vector: 'with_positions_offsets'}. + // fvh requires term_vector: with_positions_offsets on the indexed field. type: 'fvh', } } diff --git a/src/search/lib/helpers/cse-copilot-docs-versions.ts b/src/search/lib/helpers/cse-copilot-docs-versions.ts index 486b2b5fb16e..cc9654538679 100644 --- a/src/search/lib/helpers/cse-copilot-docs-versions.ts +++ b/src/search/lib/helpers/cse-copilot-docs-versions.ts @@ -1,4 +1,4 @@ -// Versions used by cse-copilot +// cse-copilot accepts this small docs version set. import { versionToIndexVersionMap } from '../elasticsearch-versions' const CSE_COPILOT_DOCS_VERSIONS = ['dotcom', 'ghec', 'ghes'] @@ -8,7 +8,7 @@ export function getCSECopilotSource(version: (typeof CSE_COPILOT_DOCS_VERSIONS)[ } let mappedVersion = versionToIndexVersionMap[version] - // CSE-Copilot uses 'dotcom' as the version name for free-pro-team + // cse-copilot expects dotcom for free-pro-team. if (mappedVersion === 'fpt') { mappedVersion = 'dotcom' } @@ -18,6 +18,6 @@ export function getCSECopilotSource(version: (typeof CSE_COPILOT_DOCS_VERSIONS)[ `Invalid 'version' in request body: '${version}'. Must be one of: ${CSE_COPILOT_DOCS_VERSIONS.join(', ')}`, ) } - // cse-copilot uses version names in the form `docs_`, e.g. `docs_ghes-3.16` + // cse-copilot docs sources use docs_ plus the mapped index version, such as docs_ghes-X. return `docs_${mappedVersion}` } diff --git a/src/search/lib/helpers/external-search-analytics.ts b/src/search/lib/helpers/external-search-analytics.ts index d1f0409e2b20..c532d77b077f 100644 --- a/src/search/lib/helpers/external-search-analytics.ts +++ b/src/search/lib/helpers/external-search-analytics.ts @@ -5,9 +5,7 @@ import { createLogger } from '@/observability/logger' const logger = createLogger(import.meta.url) -// Validates client_name and sends analytics for external requests. Returns null -// when the request should continue, or an error response object when validation -// failed. +// Return null to continue; return an error object when client_name validation fails. export async function handleExternalSearchAnalytics( req: ExtendedRequest, searchContext: string, @@ -19,31 +17,28 @@ export async function handleExternalSearchAnalytics( let client_name = req.query.client_name || req.body?.client_name - // Rule 1: Skip analytics for browser requests from our own frontend + // Skip analytics for docs.github.com frontend browser requests. if (!isLikelyExternalAPI && client_name === 'docs.github.com-client') { return null } - // Rule 2: Send analytics for any request with a client_name that's not 'docs.github.com-client' - // (This includes partner APIs and other external clients) + // Partner APIs and other external clients send analytics with their client_name. if (client_name && client_name !== 'docs.github.com-client') { - // Analytics will be sent at the end of this function - } - // Rule 3: For requests without client_name, require it for external API requests - else if (!client_name) { + // Later code sends analytics for external client_name values. + } else if (!client_name) { if (isLikelyExternalAPI) { return { status: 400, error: "Missing required parameter 'client_name' for external requests", } } - // For browser requests without client_name to internal environments, skip analytics + // Internal browser hosts without client_name skip analytics. else if (normalizedHost.endsWith('.github.net') || normalizedHost.endsWith('.githubapp.com')) { return null } } - // For localhost, ensure we have a client_name for analytics + // localhost gets a synthetic client_name so analytics records a client identifier. if (normalizedHost === 'localhost' && !client_name) { client_name = 'localhost' } @@ -95,7 +90,6 @@ export async function handleExternalSearchAnalytics( function sanitizeUserAgent(userAgent: string | undefined): string { if (!userAgent) return 'unknown' - // Extract common client types while removing version numbers and detailed info const patterns = [ { regex: /^curl/i, name: 'curl' }, { regex: /^wget/i, name: 'wget' }, @@ -135,20 +129,18 @@ function isExternalAPIRequest(req: ExtendedRequest): boolean { const prefersJson = acceptHeader.includes('application/json') && !acceptHeader.includes('text/html') - // Common API user agents (not exhaustive, but catches common cases) + // Common API user agents cover common cases, not every client. const userAgent = headers['user-agent'] || '' const hasAPIUserAgent = userAgentRegex.test(userAgent) - // If it has browser-specific headers, it's likely a browser if (hasSecFetchHeaders || hasClientHints) { return false } - // If it prefers JSON or has a common API user agent, it's likely an API if (prefersJson || hasAPIUserAgent) { return true } - // Default to treating it as a browser request to be conservative + // Ambiguous requests default to browser handling to avoid false external-client errors. return false } diff --git a/src/search/lib/helpers/get-client.ts b/src/search/lib/helpers/get-client.ts index f29220d96412..19e11c606f8e 100644 --- a/src/search/lib/helpers/get-client.ts +++ b/src/search/lib/helpers/get-client.ts @@ -31,7 +31,7 @@ function getElasticsearchURL(overrideURL = ''): string { } let node = overrideURL || process.env.ELASTICSEARCH_URL || '' - // Allow the user to lazily set it to `localhost:9200` for example. + // Accept localhost:9200 as shorthand for http://localhost:9200. if (!node.startsWith('http') && !node.startsWith('://') && node.split(':').length === 2) { node = `http://${node}` } diff --git a/src/search/lib/helpers/get-cse-copilot-auth.ts b/src/search/lib/helpers/get-cse-copilot-auth.ts index a636852452e7..642553c1c7be 100644 --- a/src/search/lib/helpers/get-cse-copilot-auth.ts +++ b/src/search/lib/helpers/get-cse-copilot-auth.ts @@ -3,7 +3,7 @@ import crypto from 'crypto' // github/cse-copilot's API requires an HMAC-SHA256 signature with each request export function getHmacWithEpoch() { const epochTime = getEpochTime().toString() - // CSE_COPILOT_SECRET needs to be set for the api-ai-search tests to work + // Test runs get a mock secret so api-ai-search tests can sign requests. if (process.env.NODE_ENV === 'test') { process.env.CSE_COPILOT_SECRET = 'mock-secret' } @@ -14,7 +14,7 @@ export function getHmacWithEpoch() { return `${epochTime}.${hmac}` } -// In seconds +// cse-copilot signatures require Unix seconds. function getEpochTime(): number { return Math.floor(Date.now() / 1000) } diff --git a/src/search/lib/helpers/time.ts b/src/search/lib/helpers/time.ts index 7b21d8361959..0c6bf7faf5e4 100644 --- a/src/search/lib/helpers/time.ts +++ b/src/search/lib/helpers/time.ts @@ -13,9 +13,7 @@ export function formatTime(ms: number) { return `${seconds.toFixed(1)}s` } -// Return '20220719012012' if the current date is -// 2022-07-19T01:20:12.172Z. Note how the 6th month (July) becomes -// '07'. All numbers become 2 character zero-padding strings individually. +// Formats the current UTC time as YYYYMMDDHHmmss, such as 20220719012012. export function utcTimestamp() { const d = new Date() @@ -28,13 +26,13 @@ export function utcTimestamp() { d.getUTCMinutes(), d.getUTCSeconds(), ] - // If it's a number make it a zero-padding 2 character string + // Numeric UTC parts need zero-padding before joining. .map((x) => (typeof x === 'number' ? `0${x}`.slice(-2) : x)) .join('') ) } -// Formats seconds as "HH:mm:ss". 5445 becomes "01:30:45". Wraps at 24 hours. +// Formats seconds as HH:mm:ss. 5445 becomes 01:30:45. Wraps at 24 hours. export function formatSecondsToHHMMSS(seconds: number): string { return new Date(seconds * 1000).toISOString().substr(11, 8) } diff --git a/src/search/lib/routes/ai-search-autocomplete-route.ts b/src/search/lib/routes/ai-search-autocomplete-route.ts index 00a1eb63f687..48831cb963ef 100644 --- a/src/search/lib/routes/ai-search-autocomplete-route.ts +++ b/src/search/lib/routes/ai-search-autocomplete-route.ts @@ -10,9 +10,7 @@ import { getSearchFromRequestParams } from '@/search/lib/search-request-params/g import { handleGetSearchResultsError } from '@/search/middleware/search-routes' export async function aiSearchAutocompleteRoute(req: Request, res: Response) { - // If no query is provided, we want to return the top 5 most popular terms - // This is a special case for AI search autocomplete - // So we use `force` to allow the query to be empty without the usual validation error + // force lets empty autocomplete queries bypass validation and return popular terms. const force: { query?: string } = {} if (!req.query.query) { force.query = '' diff --git a/src/search/lib/routes/combined-search-route.ts b/src/search/lib/routes/combined-search-route.ts index b8743af4e535..b36f56930e47 100644 --- a/src/search/lib/routes/combined-search-route.ts +++ b/src/search/lib/routes/combined-search-route.ts @@ -18,8 +18,7 @@ export async function combinedSearchRoute(req: Request, res: Response) { validationErrors: aiValidationErrors, searchParams: { query: aiQuery, debug }, } = getSearchFromRequestParams(req, 'aiSearchAutocomplete', { - // Force query to override validation to allow empty string - // Because if query is empty, we still return top AI suggestions + // force lets empty autocomplete queries return the top AI suggestions. query: typeof req.query.query !== 'string' ? '' : req.query.query, }) @@ -51,7 +50,7 @@ export async function combinedSearchRoute(req: Request, res: Response) { debug, }) - // If query is empty for general search, we don't include general results + // Empty queries skip Elasticsearch and use an empty general-search fallback. let generalSearchPromise = {} as Promise if (generalQuery !== '') { generalSearchPromise = getGeneralSearchResults({ @@ -60,7 +59,6 @@ export async function combinedSearchRoute(req: Request, res: Response) { query: generalQuery, size: GENERAL_RESULTS_SIZE, debug: debug || false, - // Keys below are hard-coded and consistent sort: 'best', aggregate: ['toplevel'], autocomplete: false, @@ -77,8 +75,7 @@ export async function combinedSearchRoute(req: Request, res: Response) { meta: { found: { value: 0, relation: 'eq' }, took: { query_msec: 0, total_msec: 0 }, - // Mirror the requested page size so downstream page-count math - // (which divides by meta.size) stays finite for the empty branch. + // Use the requested size so page-count math stays finite. size: GENERAL_RESULTS_SIZE, page: 1, }, diff --git a/src/search/lib/routes/general-search-route.ts b/src/search/lib/routes/general-search-route.ts index 713fbaa20d1a..7d5f230ac623 100644 --- a/src/search/lib/routes/general-search-route.ts +++ b/src/search/lib/routes/general-search-route.ts @@ -13,7 +13,7 @@ export async function generalSearchRoute(req: Request, res: Response) { 'generalSearch', ) if (validationErrors.length) { - // We only send the first validation error to the user + // Return only the first validation error. return res.status(400).json(validationErrors[0]) } @@ -33,9 +33,7 @@ export async function generalSearchRoute(req: Request, res: Response) { if (process.env.NODE_ENV !== 'development') { searchCacheControl(res) - // We can cache this without purging it after every deploy - // because the API search is only used as a proxy for local - // and review environments. + // Manual surrogate keys keep API search cache entries out of deploy purges. setFastlySurrogateKey(res, SURROGATE_ENUMS.MANUAL) } diff --git a/src/search/lib/sanitize-search-query.ts b/src/search/lib/sanitize-search-query.ts index 4307e04166f4..afc375d354a8 100644 --- a/src/search/lib/sanitize-search-query.ts +++ b/src/search/lib/sanitize-search-query.ts @@ -1,54 +1,52 @@ -// Remove PII from search queries before logging -// Redacts common PII patterns like emails, tokens, and other sensitive data - +// Redact PII and secrets from search queries before logging. +// Stateless ghs tokens contain dots; GitHub's token pattern also allows hyphens. +// The token class includes . and -: +// https://github.blog/changelog/2026-05-15-github-app-installation-tokens-per-request-override-header/ export function sanitizeSearchQuery(query: string): string { if (!query) return query let sanitized = query - // Redact email addresses + // Email addresses can identify users in logs. sanitized = sanitized.replace(/\b[A-Za-z0-9._%+-]+@[A-Za-z0-9.-]+\.[A-Z|a-z]{2,}\b/g, '[EMAIL]') - // Redact GitHub tokens (all formats) - // Classic tokens: ghp_, gho_, ghu_, ghs_, ghr_ - // See https://github.blog/changelog/2026-05-15-github-app-installation-tokens-per-request-override-header/ + // GitHub token prefixes include ghp, gho, ghu, ghs, and ghr. sanitized = sanitized.replace(/(ghp|gho|ghu|ghs|ghr)_[A-Za-z0-9._-]{20,}/gi, '[TOKEN]') - // Fine-grained personal access tokens: github_pat_ + // github_pat identifies fine-grained personal access tokens. sanitized = sanitized.replace(/\bgithub_pat_[A-Za-z0-9_]{20,}\b/gi, '[TOKEN]') - // OAuth tokens: gho_ + // gho identifies OAuth tokens. sanitized = sanitized.replace(/\bgho_[A-Za-z0-9]{20,}\b/gi, '[TOKEN]') - // Redact UUIDs + // UUIDs can identify private resources in logs. sanitized = sanitized.replace( /\b[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\b/gi, '[UUID]', ) - // Redact JWT tokens (format: xxx.yyy.zzz where each part is base64url) + // JWTs have three base64url segments. sanitized = sanitized.replace( /\bey[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\b/g, '[JWT]', ) - // Redact IP addresses (with proper validation for 0-255 range) + // Validate each IP octet to avoid over-redacting dotted numbers. sanitized = sanitized.replace( /\b(?:(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\.){3}(?:25[0-5]|2[0-4][0-9]|[01]?[0-9][0-9]?)\b/g, '[IP]', ) - // Redact SSH private key headers + // Private key headers reveal pasted secrets. sanitized = sanitized.replace(/-----BEGIN( [A-Z]+)? PRIVATE KEY-----/g, '[SSH_KEY]') - // Redact potential API keys (long strings of hex or base64-like characters) - // This catches high-entropy strings that might be secrets + // Long mixed-character strings often indicate API keys or other secrets. sanitized = sanitized.replace(/\b[A-Za-z0-9_-]{40,}\b/g, (match) => { - // Only redact if it looks like high entropy (mixed case, numbers) + // Mixed case and numbers avoid over-redacting ordinary long words. const hasLowerCase = /[a-z]/.test(match) const hasUpperCase = /[A-Z]/.test(match) const hasNumbers = /[0-9]/.test(match) const entropyIndicators = [hasLowerCase, hasUpperCase, hasNumbers].filter(Boolean).length - // If it has at least 2 of the 3 character types, it's likely a secret + // Two character classes meet the secret heuristic. if (entropyIndicators >= 2) { return '[SECRET]' } diff --git a/src/search/lib/search-request-params/get-search-from-request-params.ts b/src/search/lib/search-request-params/get-search-from-request-params.ts index 110e7df82478..de4391180aa0 100644 --- a/src/search/lib/search-request-params/get-search-from-request-params.ts +++ b/src/search/lib/search-request-params/get-search-from-request-params.ts @@ -18,9 +18,9 @@ type ForceParams = { [K in keyof ComputedSearchQueryParams]?: ComputedSearchQueryParams[K] } -// Fetches the Search Params Object based on the type of request and uses that object to validate the passed in request parameters -// For example, if the request is a general search request, the general search params object expects a `page` key, e.g. ?page=1 on the request -// If that key is not present, it will be added to the validation errors array which will result in a 400 to the user. +// Each search type owns its query-parameter schema. API callers turn validation +// errors into 400s; /search middleware renders them. +// General search defaults missing page values to 1. export function getSearchFromRequestParams( req: Request, type: Type, diff --git a/src/search/lib/search-request-params/search-params-objects.ts b/src/search/lib/search-request-params/search-params-objects.ts index 89fa47adfabe..56e53bbb5e77 100644 --- a/src/search/lib/search-request-params/search-params-objects.ts +++ b/src/search/lib/search-request-params/search-params-objects.ts @@ -1,14 +1,12 @@ -// A request to a /search endpoint carries query parameters, e.g. -// ?query=foo&version=free-pro-team, which have to be validated and parsed. This -// file configures which parameters to expect for each type of search request -// (general search vs autocomplete search) and how to validate them. +// Search request schemas define parameters, defaults, casts, and validation +// rules for general search and AI autocomplete endpoints. +// Example request query strings include ?query=foo&version=free-pro-team. import languages from '@/languages/lib/languages-server' import { allIndexVersionKeys, versionToIndexVersionMap } from '@/search/lib/elasticsearch-versions' import { SearchTypes } from '@/search/types' import type { SearchRequestQueryParams } from '@/search/lib/search-request-params/types' -// Entry to this file, returns the query parameters to expect based on the type of search request export function getSearchRequestParamsObject(type: SearchTypes): SearchRequestQueryParams[] { if (type === 'aiSearchAutocomplete') { return AI_SEARCH_AUTOCOMPLETE_PARAMS_OBJ @@ -26,9 +24,7 @@ const DEFAULT_SORT = POSSIBLE_SORTS[0] const MAX_PAGE = 10 const V1_AGGREGATES = ['toplevel'] as const export const POSSIBLE_HIGHLIGHT_FIELDS = ['title', 'content'] as const -// This needs to match what we *use* in the `` component. -// For example, if we don't display "headings" we shouldn't request -// highlights for it either. +// Keep this list in sync with SearchResults; only request highlights the UI displays. export const DEFAULT_HIGHLIGHT_FIELDS: readonly string[] = ['title', 'content'] export const V1_ADDITIONAL_INCLUDES = ['intro', 'headings', 'toplevel'] as const diff --git a/src/search/lib/search-request-params/types.ts b/src/search/lib/search-request-params/types.ts index e24132e0fde2..1012726854ca 100644 --- a/src/search/lib/search-request-params/types.ts +++ b/src/search/lib/search-request-params/types.ts @@ -11,7 +11,7 @@ export interface ComputedSearchQueryParams { size: number version: string language: string - // These are optional, so we need to use ComputedSearchQueryParamsMap in functions to get the exact types per Search Type + // ComputedSearchQueryParamsMap narrows optional fields for each search type. page?: number sort?: string highlights?: HighlightOptions[] diff --git a/src/search/middleware/ai-search-local-proxy.ts b/src/search/middleware/ai-search-local-proxy.ts index 790a0e42c786..d218b35746e3 100644 --- a/src/search/middleware/ai-search-local-proxy.ts +++ b/src/search/middleware/ai-search-local-proxy.ts @@ -1,4 +1,4 @@ -// When in local development we want to proxy to the ai-search route at docs.github.com +// Local development proxies AI search requests to docs.github.com. import { Router, Request, Response, NextFunction } from 'express' import { fetchStream } from '@/frame/lib/fetch-utils' @@ -26,7 +26,6 @@ function filterRequestHeaders(src: Request['headers']) { if (!value) continue const k = key.toLowerCase() if (hopByHop.has(k) || k === 'cookie' || k === 'host') continue - // Convert array values to string out[key] = Array.isArray(value) ? value[0] : value } out['accept'] = 'application/x-ndjson' @@ -50,20 +49,17 @@ router.post('/ai-search/v1', async (req: Request, res: Response, next: NextFunct }, ) - // Set status code res.status(response.status || 500) - // Forward response headers for (const [k, v] of response.headers.entries()) { if (!v) continue const key = k.toLowerCase() - // Never forward hop-by-hop; fetch already handles chunked → strip content-length + // Never forward hop-by-hop or content-length; fetch handles chunked responses. if (hopByHop.has(key) || key === 'content-length') continue res.setHeader(k, v) } res.flushHeaders?.() - // Convert fetch ReadableStream to Node.js Readable stream for pipeline if (!response.body) { if (!res.headersSent) res.status(502).end('Bad Gateway') return @@ -99,7 +95,6 @@ router.post('/ai-search/v1', async (req: Request, res: Response, next: NextFunct logger.error('[ai-search proxy] request failed', { error: err }) next(err) } finally { - // Ensure reader lock is always released if (reader) { reader.releaseLock() } diff --git a/src/search/middleware/ai-search.ts b/src/search/middleware/ai-search.ts index 0ff0aff98edd..2dbeedcb28c0 100644 --- a/src/search/middleware/ai-search.ts +++ b/src/search/middleware/ai-search.ts @@ -14,7 +14,6 @@ router.post( }), ) -// Redirect to most recent version router.post('/', (req, res) => { res.safeRedirect(307, req.originalUrl.replace('/ai-search', '/ai-search/v1')) }) diff --git a/src/search/middleware/general-search-middleware.ts b/src/search/middleware/general-search-middleware.ts index 556f8305243d..2a1604e6fdb3 100644 --- a/src/search/middleware/general-search-middleware.ts +++ b/src/search/middleware/general-search-middleware.ts @@ -1,10 +1,5 @@ -/* -This file & middleware is for when a user requests our /search page e.g. 'docs.github.com/search?query=foo' - We make whatever search is in the ?query= parameter and attach it to req.search - req.search is then consumed by the search component in 'src/search/pages/search.tsx' - -When a user directly hits our API e.g. /api/search/v1?query=foo, they will hit the routes in ./search-routes.ts -*/ +// /search page requests attach general-search results for search-results.tsx. +// /api/search/v1 requests use search-routes.ts instead. import { fetchWithRetry } from '@/frame/lib/fetch-utils' import { Request, Response, NextFunction } from 'express' @@ -37,6 +32,11 @@ interface CustomRequest extends Request { context: Context } +// contextualizeGeneralSearch includes toplevel for category chips. Elasticsearch +// already returns toplevel in _source_includes, so this needs no mapping change +// or reindex. The default include array comes from module-level default_: [], so +// in-place mutation would leak toplevel into later requests, including public +// /api/search/v1. export default async function contextualizeGeneralSearch( req: CustomRequest<'generalSearch'>, res: Response, @@ -47,11 +47,10 @@ export default async function contextualizeGeneralSearch( return next() } - // Since this is a middleware language & version are already set in req.context via a prior middleware + // Earlier middleware sets language and version on req.context. const { indexName, searchParams, validationErrors } = getSearchFromRequestParams( req, 'generalSearch', - // Force the version and language keys to be set from the `req.context` object { version: req.context.currentVersion, language: req.context.currentLanguage, @@ -62,20 +61,14 @@ export default async function contextualizeGeneralSearch( if (Array.isArray(searchParams.query)) { searchParams.query = searchParams.query[0] } else if (!searchParams.query) { - searchParams.query = '' // If 'undefined' we need to cast to string + // Cast missing query to an empty string so search page rendering gets a string. + searchParams.query = '' } } searchParams.aggregate = ['toplevel'] - // Each result row renders a category chip (Docs 2026). `toplevel` is already in the - // Elasticsearch `_source_includes`, so this only asks getHits() to copy it onto the - // returned hit — no mapping change and no reindex. - // - // Assign a new array rather than pushing: when `include` is absent from the query - // string, getSearchFromRequestParams hands back the module-level `default_: []` *by - // reference*, so mutating it here would leak `toplevel` into every later request that - // omits the parameter — including the public /api/search/v1. + // Category chips need toplevel; assign a new include array to avoid shared defaults. if (!searchParams.include.includes('toplevel')) { searchParams.include = [...searchParams.include, 'toplevel'] } @@ -86,10 +79,10 @@ export default async function contextualizeGeneralSearch( } if (!validationErrors.length && searchParams.query) { - // In local dev ELASTICSEARCH_URL may not be set, so we proxy the search to prod + // Local development proxies to production when ELASTICSEARCH_URL is unset. if (!process.env.ELASTICSEARCH_URL) { if (searchParams.aggregate && searchParams.toplevel && searchParams.toplevel.length > 0) { - // Do 2 searches. One without filtering to get the aggregations + // Fetch unfiltered aggregations separately when toplevel filters apply. const searchWithoutFilter = Object.fromEntries( Object.entries(searchParams).filter(([key]) => key !== 'toplevel'), ) @@ -120,7 +113,7 @@ export default async function contextualizeGeneralSearch( } try { if (searchParams.aggregate && searchParams.toplevel && searchParams.toplevel.length > 0) { - // Do 2 searches. One without filtering to get the aggregations + // Fetch unfiltered aggregations separately when toplevel filters apply. const searchWithoutFilter = Object.fromEntries( Object.entries(searchParams).filter(([key]) => key !== 'toplevel'), ) @@ -135,7 +128,7 @@ export default async function contextualizeGeneralSearch( req.context.search.results = await timed(getGeneralSearchArgs) } } catch (error) { - // If the Elasticsearch sends a 4XX we want the user to see a 500 + // Rethrow Elasticsearch response errors as plain errors so users get a 500. if (error instanceof errors.ResponseError) { logger.error('Error calling getSearchResults', { indexName, @@ -162,12 +155,10 @@ const SEARCH_KEYS_TO_QUERY_STRING: (keyof ComputedSearchQueryParamsMap['generalS 'aggregate', 'toplevel', 'size', - // Without this, the proxied search (used whenever ELASTICSEARCH_URL is unset) silently - // drops `include`, so hits come back without `toplevel` and the category chips vanish. + // Local proxied search must forward include so category chips receive toplevel. 'include', ] -// Proxy the API endpoint with the relevant search params async function getProxySearch( search: ComputedSearchQueryParamsMap['generalSearch'], ): Promise { @@ -186,7 +177,7 @@ async function getProxySearch( url.searchParams.set(key, value) } } - // Add client_name for external API requests + // client_name marks local proxy requests as first-party for analytics validation. url.searchParams.set('client_name', 'docs.github.com-client') logger.info('Proxying search', { url: url.toString() }) diff --git a/src/search/middleware/search-routes.ts b/src/search/middleware/search-routes.ts index 5e24e4c6d39a..b01b90d3f557 100644 --- a/src/search/middleware/search-routes.ts +++ b/src/search/middleware/search-routes.ts @@ -1,8 +1,4 @@ -/* - This file and the routes included are for the /search endpoint of our API - - For general search (client searches on docs.github.com) we use the middleware in ./general-search-middleware to get the search results -*/ +// API /search routes. The /search page uses general-search-middleware.ts instead. import express, { Request, Response } from 'express' import FailBot from '@/observability/lib/failbot' @@ -23,8 +19,7 @@ router.get('/v1', catchMiddlewareError(generalSearchRoute)) router.get('/ai-search-autocomplete/v1', catchMiddlewareError(aiSearchAutocompleteRoute)) -// Route used by our frontend to fetch ai autocomplete search suggestions + general search results in a single request -// Combining this into a single request results in less overall requests to the server +// Combined search avoids a second frontend request for autocomplete suggestions. router.get('/combined-search/v1', catchMiddlewareError(combinedSearchRoute)) export async function handleGetSearchResultsError( @@ -51,8 +46,7 @@ export async function handleGetSearchResultsError( const reports = FailBot.report(errorForReport, extra) if (reports) await Promise.all(reports) } - // Avoid "Cannot set headers after they are sent to the client" error - // if response was already partially sent before the error occurred + // Skip writing a JSON 500 after any response has already started. if (!res.headersSent) { res.status(500).json({ error: errorMessage }) } else { @@ -63,7 +57,6 @@ export async function handleGetSearchResultsError( } } -// Redirects search routes to their latest versions router.get('/', (req: Request, res: Response) => { res.safeRedirect(307, req.originalUrl.replace('/search', '/search/v1')) }) diff --git a/src/search/pages/search-results.tsx b/src/search/pages/search-results.tsx index 024cd1f50ce6..2aec26a65a68 100644 --- a/src/search/pages/search-results.tsx +++ b/src/search/pages/search-results.tsx @@ -39,20 +39,13 @@ export const getServerSideProps: GetServerSideProps = async (context) => addUINamespaces(req, mainContext.data.ui, ['search_results']) if (!req.context?.search) { - // This should have been done by the middleware. + // Middleware populates req.context.search before this page renders. throw new Error('Expected req.context to be populated with .search') } const searchObject = (req.context?.search ?? {}) as SearchOnReqObject<'generalSearch'> - // The `req.context.search` is similar to what's needed to React - // render the search result page. - // But it contains information (from the contextualizing) that is - // not needed to display search results. - // For example, the `req.context.search.search` contains things like - // `page` and `indexName` which was useful when it made the actual - // Elasticsearch query. But it's not needed to render the results. - // We explicitly pick out the parts that are needed, only. + // Only query and debug go into searchParams; result metadata stays in results.meta. const search: SearchContextT['search'] = { searchParams: { query: searchObject.searchParams.query, @@ -60,16 +53,12 @@ export const getServerSideProps: GetServerSideProps = async (context) => }, validationErrors: searchObject.validationErrors, } - // If there are no results (e.g. /en/search?query=) from the - // contextualizing, then `req.context.search.results` will - // be `undefined` which can't be serialized as a prop, using JSON.stringify. + // Empty searches omit results because Next.js cannot serialize undefined. if (searchObject.results) { search.results = { meta: searchObject.results.meta, hits: searchObject.results.hits, - // Use `null` instead of `undefined` for JSON serialization. - // The only reason it would ever not be truthy is if the aggregates - // functionality is not enabled for this version. + // Normalize missing aggregations to null because Next.js cannot serialize undefined. aggregations: searchObject.results.aggregations || null, } } diff --git a/src/search/types.ts b/src/search/types.ts index 58ab73c01108..4f52ee474d71 100644 --- a/src/search/types.ts +++ b/src/search/types.ts @@ -7,7 +7,6 @@ import type { export type SearchTypes = 'generalSearch' | 'aiSearchAutocomplete' -// Responses to API routes export interface GeneralSearchResponse { meta: SearchResultsMeta & { page: number @@ -26,7 +25,6 @@ export interface CombinedSearchResponse { generalSearchResults: GeneralSearchResponse } -// Response to middleware /search route export interface SearchOnReqObject { searchParams: ComputedSearchQueryParamsMap[Type] validationErrors: SearchValidationErrorEntry[] @@ -39,7 +37,6 @@ export interface SearchValidationErrorEntry { field?: string } -// - - - Types for building the search responses - - - export interface GeneralSearchHitWithoutIncludes { id: string url: string diff --git a/src/secret-scanning/components/SecretScanningTable.module.scss b/src/secret-scanning/components/SecretScanningTable.module.scss new file mode 100644 index 000000000000..d8fd63d197b7 --- /dev/null +++ b/src/secret-scanning/components/SecretScanningTable.module.scss @@ -0,0 +1,3 @@ +.filterDropdown [role="menu"] { + min-width: 100%; +} diff --git a/src/secret-scanning/components/SecretScanningTable.tsx b/src/secret-scanning/components/SecretScanningTable.tsx index b1d0fbf0295d..6d32e0c8e53e 100644 --- a/src/secret-scanning/components/SecretScanningTable.tsx +++ b/src/secret-scanning/components/SecretScanningTable.tsx @@ -1,13 +1,14 @@ import React, { useState, useMemo, useEffect, useRef, useCallback } from 'react' import { DataTable, Table } from '@primer/react/experimental' -import { TextInput, ActionMenu, ActionList } from '@primer/react' -import { Pagination, Button } from '@primer/react-brand' +import { TextInput, ActionMenu, Pagination, Button } from '@primer/react-brand' import { debounce } from 'lodash-es' import { useTranslation } from '@/languages/components/useTranslation' import { sendEvent } from '@/events/components/events' import { EventType } from '@/events/types' import { sanitizeSearchQuery } from '@/search/lib/sanitize-search-query' +import { onActionMenuItemKeyDownCapture } from '@/frame/components/lib/action-menu' import type { SecretScanningData } from '@/types' +import styles from './SecretScanningTable.module.scss' const PAGE_SIZE = 25 @@ -203,6 +204,7 @@ export function SecretScanningTable({ data }: { data: SecretScanningData[] }) { )} void }) { const { t } = useTranslation('secret_scanning') + const selectedLabel = + value === 'all' ? t('filter_all') : value === 'yes' ? t('filter_yes') : t('filter_no') return ( - - - {label}:{' '} - {value === 'all' ? t('filter_all') : value === 'yes' ? t('filter_yes') : t('filter_no')} - - - +
    + onChange(selectedValue as 'all' | 'yes' | 'no')} + > + + {label}: {selectedLabel} + + {(['all', 'yes', 'no'] as const).map((opt) => ( - onChange(opt)}> + {opt === 'all' ? t('filter_all') : opt === 'yes' ? t('filter_yes') : t('filter_no')} - + ))} - - - + + +
    ) } diff --git a/src/workflows/action-context.ts b/src/workflows/action-context.ts index c3df3e693686..694866efb633 100644 --- a/src/workflows/action-context.ts +++ b/src/workflows/action-context.ts @@ -1,6 +1,5 @@ import { readFileSync } from 'fs' -// Parses the action event payload sets repo and owner to an object from runner environment export function getActionContext() { if (!process.env.GITHUB_EVENT_PATH) { if (!process.env.CI) { diff --git a/src/workflows/benchmark-pages.ts b/src/workflows/benchmark-pages.ts index d27ea86ccf69..39531863c241 100644 --- a/src/workflows/benchmark-pages.ts +++ b/src/workflows/benchmark-pages.ts @@ -1,24 +1,13 @@ -/** - * Benchmarks page load times across the local docs server. - * - * Hits every page via the article API and/or HTML routes, across - * configurable languages and versions. Reports errors and slow pages. - * - * Assumes the server is already running on --port (default 4000). - * - * Usage: - * npx tsx src/workflows/benchmark-pages.ts [options] - * - * Options: - * --port Server port (default: 4000) - * --langs Comma-separated language codes, or "all" (default: en) - * --versions Comma-separated version slugs, or "all" (default: free-pro-team@latest) - * --modes Comma-separated: article-body, article-meta, html, or "all" (default: article-body) - * --sample Random sample size per lang/version (default: all pages) - * --slow Threshold in ms to flag as slow (default: 500) - * --concurrency Max concurrent requests (default: 1) - * --json Write JSON results to file (for CI consumption) - */ +// Benchmarks latency against an already-running local docs server. +// +// Usage: +// npx tsx src/workflows/benchmark-pages.ts [options] +// +// Defaults: port 4000, language en, version free-pro-team@latest, mode +// article-body, slow threshold 500 ms, concurrency 1. +// Modes: article-body, article-meta, html, or all. +// Use --langs and --versions for comma-separated lists or all, --sample to limit +// pages per language and version, and --json to write CI-readable results. import { parseArgs } from 'node:util' diff --git a/src/workflows/check-content-type.ts b/src/workflows/check-content-type.ts index 8cd6f8a1da5f..f57b5b1b461b 100755 --- a/src/workflows/check-content-type.ts +++ b/src/workflows/check-content-type.ts @@ -7,9 +7,7 @@ const { CHANGED_FILE_PATHS, CONTENT_TYPE } = process.env main() async function main() { - // CHANGED_FILE_PATHS is a string of space-separated - // file paths. For example: - // 'content/path/foo.md content/path/bar.md' + // CHANGED_FILE_PATHS is space-separated: content/actions/index.md content/admin/index.md const filePaths = CHANGED_FILE_PATHS?.split(' ') || [] const containsRai = checkContentType(filePaths, CONTENT_TYPE || '') if (containsRai.length === 0) { diff --git a/src/workflows/content-changes-table-comment-cli.ts b/src/workflows/content-changes-table-comment-cli.ts index f92cb53f01c2..5feb1199b430 100644 --- a/src/workflows/content-changes-table-comment-cli.ts +++ b/src/workflows/content-changes-table-comment-cli.ts @@ -1,24 +1,10 @@ -// [start-readme] +// Runs the content-changes-table code locally instead of waiting on a PR run of +// .github/workflows/review-comment.yml, which runs it on pull_request_target. +// Requires GITHUB_TOKEN with content and pull request read access, plus APP_URL +// set to the review environment or production. // -// For testing the GitHub Action that executes -// src/workflows/content-changes-table-comment.ts but doing it -// locally. -// This is more convenient and faster than relying on seeing that the -// Action produces in a PR. Especially since -// .github/workflows/comment-content-changes-table.yml only runs -// on `pull_request_target`. -// -// To try it you need to generate a local `GITHUB_TOKEN` that has read-access -// "content" and "pull requests" on the repo. -// You also need to set an APP_URL which can be the domain of the -// review environment or just the production domain. Example: -// -// -// export GITHUB_TOKEN=github_pat_11AAAG..... -// export APP_URL=https://docs.github.com -// tsx src/workflows/content-changes-table-comment-cli.ts github docs-internal main 4a0b0f2 -// -// [end-readme] +// Usage: +// npx tsx src/workflows/content-changes-table-comment-cli.ts github docs-internal main 4a0b0f2 import { program } from 'commander' import main from '@/workflows/content-changes-table-comment' diff --git a/src/workflows/content-changes-table-comment.ts b/src/workflows/content-changes-table-comment.ts index 2edc5a4d12ab..8d1d8494a171 100755 --- a/src/workflows/content-changes-table-comment.ts +++ b/src/workflows/content-changes-table-comment.ts @@ -1,6 +1,4 @@ -// To test this locally, outside of Actions, run -// src/workflows/content-changes-table-comment-cli.ts. Its file header has the -// instructions. +// Run src/workflows/content-changes-table-comment-cli.ts to test this outside Actions. import fs from 'node:fs' import path from 'node:path' @@ -21,15 +19,12 @@ import { inLiquid } from './lib/in-liquid' const { GITHUB_TOKEN, APP_URL, BASE_SHA, HEAD_SHA } = process.env const context = github.context -// Max table size in characters. peter-evans/create-or-update-comment allows a -// 2^16 character comment, but the table measures itself near the end of -// rendering, before the key is added, so this stays at 2^15 for headroom. See -// github/docs-engineering#1849 and peter-evans/create-or-update-comment#271. +// peter-evans/create-or-update-comment allows 65,536-character comments. This +// table measures itself before adding the key, so 32,768 leaves headroom. const MAX_COMMENT_SIZE = 32768 const PROD_URL = 'https://docs.github.com' -// When this file is invoked directly from action as opposed to being imported if (import.meta.url.endsWith(process.argv[1])) { const baseOwner = context.payload.pull_request!.base.repo.owner.login const baseRepo = context.payload.pull_request!.base.repo.name @@ -51,7 +46,7 @@ async function main(owner: string, repo: string, baseSHA: string, headSHA: strin const octokit = retryingGithub(GITHUB_TOKEN) - // The list of file changes, which works even for a head commit from a fork. + // Compare through the base repo so forked head commits work. const response = await octokit.rest.repos.compareCommitsWithBasehead({ owner, repo, @@ -102,12 +97,11 @@ async function main(owner: string, repo: string, baseSHA: string, headSHA: strin const fileName = file.filename.slice(pathPrefix.length) const fileUrl = fileName.replace('/index.md', '').replace(/\.md$/, '') - // this script is called from the main branch, so we need the API call to get the contents from the branch, instead + // The workflow runs from main, so request the file from the changed branch. const fileContents = await getContents( owner, repo, - // `getContents()` 404s on a file that no longer exists, so for a - // removed file read the base sha to get metadata about what it was. + // Removed files need the base SHA because getContents 404s at the head SHA. file.status === 'removed' ? baseSHA : headSHA, file.filename, ) @@ -199,33 +193,29 @@ function makeRow({ contentCell += `[\`${fileName}\`](${sourceUrl})` try { - // getApplicableVersions() throws on missing, invalid or unsupported - // versions frontmatter. Remove the try/catch once - // github/docs-engineering#1821 is fixed. + // getApplicableVersions throws for missing, invalid, or unsupported versions frontmatter. const fileVersions: string[] = getApplicableVersions(data?.versions) for (const plan in allVersionShortnames) { - // `plan` is the short name, e.g. fpt, used as the link label. - // allVersionShortnames[plan] is the plan name, e.g. free-pro-team, used - // to pick the file's matching versions. Most plans link differently. + // Plan shortnames, for example fpt, label links; full names match versions. const versions = fileVersions.filter((fileVersion) => fileVersion.includes(allVersionShortnames[plan]), ) if (versions.length === 1) { if (versions.toString() === nonEnterpriseDefaultVersion) { - // omit version from fpt url + // Default free-pro-team URLs omit the version segment. reviewCell += `[${plan}](${APP_URL}/${fileUrl})
    ` prodCell += `[${plan}](${PROD_URL}/${fileUrl})
    ` } else { - // for non-versioned releases (ghec) use full url + // Other single-version releases use the full version URL. reviewCell += `[${plan}](${APP_URL}/${versions}/${fileUrl})
    ` prodCell += `[${plan}](${PROD_URL}/${versions}/${fileUrl})
    ` } } else if (versions.length) { - // for ghes releases, link each version + // GHES releases link each matching version. reviewCell += `${plan}@ ` prodCell += `${plan}@ ` @@ -246,8 +236,7 @@ function makeRow({ let note = '' if (file.status === 'removed') { note = 'removed' - // If the file was removed, the `reviewCell` no longer makes sense - // since it was based on looking at the base sha. + // Removed files do not exist in the review environment, so review links do not apply. reviewCell = 'n/a' } else if (fromReusable) { note += 'from reusable' diff --git a/src/workflows/delete-orphan-translation-files.ts b/src/workflows/delete-orphan-translation-files.ts index c15497f251c1..4f038221ec8a 100644 --- a/src/workflows/delete-orphan-translation-files.ts +++ b/src/workflows/delete-orphan-translation-files.ts @@ -1,23 +1,3 @@ -/** - * This script will delete files from a translation repo of files that - * only exist there and not "here". Here being the docs repo. - * It will only look at *.md files in `content/` and - * only look at *.md and *.yml files in `data/`. - * - * If executed with `--dry-run` it will only print what it would delete. - * - * To avoid deleting too many files at once, which can make PRs too big, - * there's a `--max ` options which is defaulted to 100. - * - * To run this locally, check out a translation repo and then run it like this: - * - * git clone git@github.com:github/docs-internal.ja-jp.git /tmp/docs-internal.ja-jp - * npm run delete-orphan-translation-files -- /tmp/docs-internal.ja-jp - * - * Note that it doesn't execute `git rm ...` for you. Just regular - * file deletion. It's up to you now to commit and push. - */ - import fs from 'fs' import path from 'path' @@ -25,6 +5,15 @@ import { program } from 'commander' import walkFiles from '@/workflows/walk-files' import { ROOT } from '@/frame/lib/constants' +// Deletes orphaned translation content files from a checked-out translation repo. +// It deletes files from the working tree only; it does not stage removals with git rm. +// +// Usage: +// npm run delete-orphan-translation-files -- /tmp/docs-internal.ja-jp +// +// Use --dry-run to print deletions without removing files. --max defaults to 100 +// so one run does not create an oversized PR. + program .description('Delete orphan translation files') .option('--dry-run', 'Just print what it would delete') @@ -86,27 +75,9 @@ function main(root: string, options: Options) { ) } +// Walk content only. Translated content can still use {% data variables.x %} after English +// deletes data/variables/x.yml, so translated data files may be orphans on purpose. function getContentAndDataFiles(root: string) { - // The reason we're only looking at content files, and not data files, - // is because data files can be *included* in content files. - // Best illustrated with an imaginary example: - // - // Suppose there exists, in English, a `content/some-page.md` and - // a `data/variables/some-var.yml`. - // The English content contains: `{% data variables.some-var.some-thing %}` - // Soon enough, this is present in the translations too. - // Then, the English writer decides to stop referencing that variable - // in `content/some-page.md`. And additionally, since no content references - // the file, they also decide to `git rm data/variables/some-var.yml`. - // At this point, there's technically an "orphan" file in the translation - // repo that doesn't have an equivalent in the English repo. But! The - // translation's copy of `content/some-page.md` might still *refer* - // to `{% data variables.some-var.some-thing %}` since it hasn't yet - // picked up that the English content changed. - // - // In conclusion, we need to be OK with the data files in translations - // being potentially "full of orphans" because they might still be - // referred to the in the content files. return walkFiles(path.join(root, 'content'), ['.md']) } diff --git a/src/workflows/find-past-built-pr.ts b/src/workflows/find-past-built-pr.ts index 9906e9a8983a..1c47cf60345f 100644 --- a/src/workflows/find-past-built-pr.ts +++ b/src/workflows/find-past-built-pr.ts @@ -4,18 +4,16 @@ import github from './github' import { getActionContext } from './action-context' import { octoSecondaryRatelimitRetry } from './secondary-ratelimit-retry' -// Marker used to dedupe the "gone to production" comment across reruns and -// across the previous workflow that posted it as github-actions[bot]. +// This marker dedupes production comments even when another account posted them. export const GONE_TO_PRODUCTION_MARKER = '' -// The merge queue can batch multiple PRs into a single deploy. We walk back -// from the deployed HEAD commit through `main`'s (linear, squash-merged) -// ancestry to find every PR in the batch. This is bounded by the deployed SHA, -// so it can only ever include already-deployed PRs, never a later batch. +// The merge queue can batch multiple PRs into a single deploy. Walk back from +// the deployed HEAD commit through main's linear squash-merged ancestry to find +// every PR in the batch. The deployed SHA bounds the search, so it cannot include +// a later batch. // -// Default to the merge queue's max batch size. Over-reaching into a previous, -// already-notified batch is harmless: those PRs are genuinely in production and -// the marker dedupe skips any that already have the comment. +// Default to the merge queue's max batch size. Overshooting into an already +// notified batch is harmless because marker dedupe skips existing comments. const BATCH_MAX_COMMITS = parseInt(process.env.DEPLOY_BATCH_MAX_COMMITS || '5', 10) export const COMMENT_BODY = `${GONE_TO_PRODUCTION_MARKER} @@ -27,8 +25,7 @@ If you don't see updates when expected, try adding a random query string to the If that shows the expected content, it would indicate that the CDN is "overly caching" the page still. It will eventually update, but it can take a while. ` -// GitHub appends "(#1234)" to the title of a squash-merge commit. Grab the last -// such reference on the title line, which is the merged PR number. +// GitHub appends the merged PR number to squash-merge commit titles. Grab the last one. export function extractPrNumber(commitMessage: string): number | null { const title = commitMessage.split('\n')[0] const matches = [...title.matchAll(/\(#(\d+)\)/g)] @@ -39,7 +36,6 @@ export function extractPrNumber(commitMessage: string): number | null { return parseInt(last[1], 10) } -// Returns the PR numbers in the deploy batch, newest first, deduplicated. export async function findBatchPrNumbers( octokit: Octokit, owner: string, @@ -63,9 +59,8 @@ export async function findBatchPrNumbers( export type CommentResult = 'created' | 'exists' | 'locked' -// Posts the production comment on a single PR, unless it is locked or already -// has the comment (detected by the marker, regardless of which account authored -// it). Idempotent: safe to call on every rerun. +// Posts the production comment on unlocked PRs unless the marker already exists +// under any author. Idempotent across reruns. export async function ensureProductionComment( octokit: Octokit, owner: string, diff --git a/src/workflows/fm-utils.ts b/src/workflows/fm-utils.ts index 3009a6c4309d..242e31ce5082 100644 --- a/src/workflows/fm-utils.ts +++ b/src/workflows/fm-utils.ts @@ -1,7 +1,6 @@ import { existsSync, readFileSync } from 'fs' import matter from '@gr2m/gray-matter' -// The file paths whose frontmatter `contentType` matches `contentType`. export function checkContentType(filePaths: string[], contentType: string) { const unallowedChangedFiles = [] for (const filePath of filePaths) { diff --git a/src/workflows/fr-add-docs-reviewers-requests.ts b/src/workflows/fr-add-docs-reviewers-requests.ts index 2191ea386503..bfd51778d1f9 100644 --- a/src/workflows/fr-add-docs-reviewers-requests.ts +++ b/src/workflows/fr-add-docs-reviewers-requests.ts @@ -107,11 +107,7 @@ async function getAllOpenPRs() { async function run() { const prData = await getAllOpenPRs() - // Get the PRs that are: - // - not draft - // - not a train - // - are requesting a review by docs-reviewers - // - have not already been reviewed on behalf of docs-reviewers + // Keep PRs that still need the requested docs-reviewers review. const prs = prData.filter( (pr) => !pr.isDraft && @@ -177,10 +173,7 @@ async function run() { const projectID = projectData.organization.projectV2.id - // Get the IDs of the last 100 items on the board. - // Until we have a way to check from a PR whether the PR is in a project, - // this is how we (roughly) avoid overwriting PRs that are already on the board. - // If we are overwriting items, query for more items. + // The last 100 board items approximate membership; query more if fields get overwritten. const existingItemIDs = projectData.organization.projectV2.items.nodes.map( (node: { id: string }) => node.id, ) @@ -200,10 +193,7 @@ async function run() { const itemIDs = await addItemsToProject(prIDs, projectID) - // If an item already existed on the project, the existing ID will be returned. - // Exclude existing items going forward. - // Until we have a way to check from a PR whether the PR is in a project, - // this is how we (roughly) avoid overwriting PRs that are already on the board + // Existing project items reuse their IDs, so skip them before populating fields. const newItemIDs: string[] = [] const newItemAuthors: string[] = [] for (let index = 0; index < itemIDs.length; index++) { @@ -219,7 +209,7 @@ async function run() { return } - // for...of rather than forEach because the body awaits. + // Use for...of because the body awaits. for (const [index, itemID] of newItemIDs.entries()) { const updateProjectV2ItemMutation = generateUpdateProjectV2ItemFieldMutation({ item: itemID, @@ -241,7 +231,7 @@ async function run() { contributorTypeID, contributorType, sizeTypeID, - sizeType: sizeMediumID, // We need to provide something here, defaulting to 'medium' or 'M' + sizeType: sizeMediumID, // The board requires size, so default to M. featureID, authorID, headers: { diff --git a/src/workflows/generate-llms-txt.ts b/src/workflows/generate-llms-txt.ts index 30e4971e2c40..fddab6606fed 100644 --- a/src/workflows/generate-llms-txt.ts +++ b/src/workflows/generate-llms-txt.ts @@ -1,15 +1,16 @@ // Generates an llms.txt file from popularity data and the docs page catalog. // // Usage: -// npm run generate-llms-txt -- --config data/llms-txt/config-docs.yml --output data/llms-txt/docs.md -// npm run generate-llms-txt -- --config data/llms-txt/config-monolith.yml --output /tmp/monolith.md +// npm run generate-llms-txt -- \ +// --config data/llms-txt/config-docs.yml \ +// --output data/llms-txt/docs.md +// npm run generate-llms-txt -- \ +// --config data/llms-txt/config-monolith.yml \ +// --output /tmp/monolith.md // -// Both targets layer overrides on top of data/llms-txt/config-default.yml. -// Writers can edit categories, pinned pages, thresholds, and copy in the -// configs without touching this script. -// -// Requires DOCS_BOT_PAT_BASE for fetching popularity data from -// github/docs-internal-data. +// Each target overrides data/llms-txt/config-default.yml, so writers can change +// categories, pinned pages, thresholds, and copy without editing this script. +// Requires DOCS_BOT_PAT_BASE because popularity data lives in github/docs-internal-data. import fs from 'fs' diff --git a/src/workflows/get-env-inputs.ts b/src/workflows/get-env-inputs.ts index d0e3c4a00937..75e609b27ef3 100644 --- a/src/workflows/get-env-inputs.ts +++ b/src/workflows/get-env-inputs.ts @@ -1,4 +1,3 @@ -// Validates the named environment variables and returns them as an object. export function getEnvInputs(options: string[]) { return Object.fromEntries( options.map((envVarName) => { @@ -11,9 +10,7 @@ export function getEnvInputs(options: string[]) { ) } -// Reads an environment variable as a boolean. 'true' and '1' are true; '', '0' -// and 'false' are false. Anything else throws, so a typo like -// `export FOO=falsee` can't be read as truthy. +// true and 1 are true; empty, 0, and false are false. Other values throw. export function boolEnvVar(key: string) { const value = process.env[key] || '' if (value === '' || value === 'false' || value === '0') return false diff --git a/src/workflows/git-utils.ts b/src/workflows/git-utils.ts index 3dbfd6ac074f..a2db408756f2 100644 --- a/src/workflows/git-utils.ts +++ b/src/workflows/git-utils.ts @@ -20,7 +20,7 @@ export async function getCommitSha(owner: string, repo: string, ref: string) { } } -// based on https://docs.github.com/rest/reference/git#get-a-reference +// https://docs.github.com/rest/reference/git#get-a-reference export async function hasMatchingRef(owner: string, repo: string, ref: string) { try { await github.git.getRef({ @@ -151,7 +151,6 @@ export async function createIssueComment( } } -// The paths of files in the repo containing any of the given strings. export async function getPathsWithMatchingStrings( strArr: string[], org: string, @@ -233,7 +232,6 @@ async function searchCode( } } -// Recursively gets the contents of a directory within a repo. export async function getDirectoryContents( owner: string, repo: string, diff --git a/src/workflows/github.ts b/src/workflows/github.ts index 33f48b665af4..0b5a0c1fbc6c 100644 --- a/src/workflows/github.ts +++ b/src/workflows/github.ts @@ -8,16 +8,11 @@ if (!process.env.GITHUB_TOKEN) { const RetryingOctokit = Octokit.plugin(retry) -// this module needs to work in development, production, and GitHub Actions -// -// GITHUB_TOKEN comes from one of the following sources: -// 1. set in the .env file (development) -// 2. set as a Heroku config var (staging and production) -// 3. an installation token granted via GitHub Actions +// GITHUB_TOKEN can come from .env, a Heroku config var, or a GitHub Actions +// installation token, because this module runs in every environment. const apiToken = process.env.GITHUB_TOKEN -// See https://github.com/octokit/rest.js/issues/1207 -// Pass `token` to authenticate as something other than GITHUB_TOKEN. +// token overrides GITHUB_TOKEN for workflows that need a different installation token. export default function github(token?: string) { return new Octokit({ auth: `token ${token || apiToken}`, @@ -30,11 +25,8 @@ export function retryingGithub(token?: string) { }) } -// Duck-typing instead of `instanceof RequestError` on purpose. -// `@octokit/request` throws a RequestError built from its own nested copy of -// `@octokit/request-error`, which is a different module instance than the -// top-level one. The two classes are not identical, so `instanceof` always -// returns false across that boundary. +// Duck-type instead of using instanceof because nested Octokit dependencies can +// construct RequestError classes from different module instances. export function isRequestError( error: unknown, status?: number, diff --git a/src/workflows/husky/pre-commit b/src/workflows/husky/pre-commit index 68ad34b7abb5..56b9a1f1e3cf 100755 --- a/src/workflows/husky/pre-commit +++ b/src/workflows/husky/pre-commit @@ -1,18 +1,5 @@ #!/bin/sh [ -n "$CI" ] && exit 0 -# The --verbose flag tells lint-staged to *not* swallow the stdout -# output even if it passes. -# For example, if `npm run lint-content` only had warnings, and no errors, -# now it will display those warnings in the terminal. -# In pseudo-code, `lint-staged` works like this: -# -# for command in find_commands_to_execute(git_staged_files): -# output, exitCode = executeCommand(command, git_staged_files) -# if exitCode != 0: -# print(output) -# exit(exitCode) -# else if verbose: -# print(output) -# +# --verbose prints passing command output, so lint-content warnings are visible before commit. npx lint-staged --verbose diff --git a/src/workflows/issue-report.ts b/src/workflows/issue-report.ts index 68bb81dbcfe8..95e7eae28498 100644 --- a/src/workflows/issue-report.ts +++ b/src/workflows/issue-report.ts @@ -66,7 +66,7 @@ export async function linkReports({ repo, creator: reportAuthor, labels: reportLabel, - state: 'all', // We want to get the previous report, even if it is closed + state: 'all', // Include closed reports so the new report can link to the previous one. sort: 'created', direction: 'desc', per_page: 25, @@ -83,7 +83,7 @@ export async function linkReports({ return } - // 2nd report should be most recent previous report + // Index 0 is the new report, so index 1 is the previous report. const previousReport = previousReports[1] try { @@ -104,7 +104,7 @@ export async function linkReports({ continue } - // If an old report is not assigned to someone we close it + // Close unassigned old reports so owners can keep assigned reports open. const shouldClose = !oldReport.assignees?.length let body = `➡️ [Newer report](${newReport.html_url})` if (shouldClose) { diff --git a/src/workflows/labeler.ts b/src/workflows/labeler.ts index 46f96ded6c4b..9f3af4e3db9b 100755 --- a/src/workflows/labeler.ts +++ b/src/workflows/labeler.ts @@ -1,9 +1,3 @@ -// [start-readme] -// -// This script adds labels to issues or pull requests. -// -// [end-readme] - import { program } from 'commander' import label from '../../.github/actions/labeler/labeler' import { getCoreInject } from '@/links/scripts/action-injections' diff --git a/src/workflows/lib/in-liquid.ts b/src/workflows/lib/in-liquid.ts index a8648e4292d1..f34641b4020e 100644 --- a/src/workflows/lib/in-liquid.ts +++ b/src/workflows/lib/in-liquid.ts @@ -2,7 +2,6 @@ import { getLiquidTokens } from '@/content-linter/lib/helpers/liquid-utils' import type { TagToken } from 'liquidjs' import { TokenKind } from 'liquidjs' -// Type guard to check if a token is a TagToken function isTagToken(token: unknown): token is TagToken { return ( token !== null && diff --git a/src/workflows/measure-instruction-budget.ts b/src/workflows/measure-instruction-budget.ts index ec8581fc8a5d..967e1786592d 100644 --- a/src/workflows/measure-instruction-budget.ts +++ b/src/workflows/measure-instruction-budget.ts @@ -1,22 +1,8 @@ -/** - * @purpose Writer tool - * @description Measure the always-on Copilot instruction budget for a content interaction. - * - * Scans `.github/instructions/*.instructions.md`, reads each file's `applyTo` - * frontmatter, works out which files load for a representative content path, - * and reports the combined budget against soft guardrails. - * - * The primary number is the discrete **rule count** (instruction-following - * degrades with the number of discrete instructions). The **token count** is a - * mechanical backstop. Both guardrails are soft: the script warns, it does not - * fail, unless you pass `--strict`. - * - * Usage: - * npm run measure-instruction-budget - * npm run measure-instruction-budget -- --path content/get-started/foo.md - * npm run measure-instruction-budget -- --json - * npm run measure-instruction-budget -- --strict # exit 1 if over budget - */ +// @purpose Writer tool +// @description Measure the always-on Copilot instruction budget for a content interaction. +// Scans instruction files for a representative path and reports rule and token budgets. +// Passing --strict turns budget warnings into failures. +// Usage: npm run measure-instruction-budget -- [--path content/get-started/foo.md] [--json] import fs from 'fs' import path from 'path' @@ -25,10 +11,10 @@ import { encode } from 'gpt-tokenizer/encoding/o200k_base' import readFrontmatter from '@/frame/lib/read-frontmatter' -// Single source of truth for the guardrails. Keep these in sync with the -// instruction architecture doc (github/technical-content) and any future CI check. -// Derived in github/technical-content#6829: frontier models stay near-perfect to -// ~150 discrete instructions; ~45 tokens/rule puts the token backstop at ~6,500. +// Single source of truth for guardrails. Keep these in sync with the instruction +// architecture doc in github/technical-content and any future CI check. Frontier +// models stay near-perfect near 150 discrete instructions, and 45 tokens per rule +// sets the token backstop near 6,500. const RULE_BUDGET = 150 const TOKEN_BUDGET = 6500 @@ -82,8 +68,7 @@ function main(options: Options): void { process.exit(2) } - // Normalize to POSIX separators so backslash paths (e.g. on Windows) still - // match the forward-slash applyTo globs. + // Normalize to POSIX separators so Windows paths match forward-slash applyTo globs. const simulatedPath = options.path.replace(/\\/g, '/') const loaded: FileReport[] = [] @@ -91,8 +76,7 @@ function main(options: Options): void { const raw = fs.readFileSync(path.join(dir, name), 'utf-8') const { content, data, errors } = readFrontmatter(raw, { filepath: name }) if (errors && errors.length > 0) { - // Don't silently fall back to applyTo '**' (which matches everything) and - // distort the budget. Skip the file and warn so the numbers stay trustworthy. + // Invalid applyTo must warn and skip the file because fallback ** distorts the budget. console.warn( `Warning: skipping ${name} because its frontmatter could not be parsed ` + `(${errors.map((e) => e.reason).join('; ')}).`, @@ -105,9 +89,7 @@ function main(options: Options): void { loaded.push({ file: name, applyTo, - // Count the frontmatter-stripped body: the `applyTo` frontmatter is - // metadata that governs when the file loads, not text injected into the - // prompt, so including it would systematically overcount. + // Count the body only; applyTo controls loading but is not injected into the prompt. tokens: encode(body).length, rules: countRules(body), }) @@ -145,16 +127,14 @@ function main(options: Options): void { } } -// Count discrete list-item rules (a proxy for "number of instructions"), -// ignoring fenced code blocks so example code is not counted as instructions. +// Count list-item rules as an instruction proxy, ignoring fenced code examples. export function countRules(body: string): number { const withoutCode = body.replace(/```[\s\S]*?```/g, '') const matches = withoutCode.match(/^\s*([-*]|\d+\.)\s/gm) return matches ? matches.length : 0 } -// True if any comma-separated glob in `applyTo` matches `filePath`. Both sides -// are normalized to POSIX separators so backslash paths still match. +// Match comma-separated applyTo globs after normalizing Windows separators. export function matchesPath(applyTo: string, filePath: string): boolean { const normalizedPath = filePath.replace(/\\/g, '/') return applyTo @@ -164,9 +144,9 @@ export function matchesPath(applyTo: string, filePath: string): boolean { .some((pattern) => globToRegExp(pattern).test(normalizedPath)) } -// Convert a VS Code-style applyTo glob to a RegExp. `**/` matches zero or more -// path segments (so `**/*.md` matches both `README.md` and `dir/README.md`), -// a standalone `**` matches across segments, and `*` matches within a segment. +// Convert a VS Code-style applyTo glob to a RegExp. Double-star slash matches +// zero or more path segments, standalone double-star matches across segments, +// and single-star matches within a segment. export function globToRegExp(glob: string): RegExp { let out = '' let i = 0 diff --git a/src/workflows/prevent-pushes-to-main.ts b/src/workflows/prevent-pushes-to-main.ts index e89c9265af6f..2854c40bd6d7 100644 --- a/src/workflows/prevent-pushes-to-main.ts +++ b/src/workflows/prevent-pushes-to-main.ts @@ -1,9 +1,4 @@ -// [start-readme] - -// This script is intended to be used as a git "prepush" hook. -// If the current branch is main, it will exit unsuccessfully and prevent the push. - -// [end-readme] +// Pre-push hooks use this script to block accidental pushes to main. import { execSync } from 'child_process' diff --git a/src/workflows/projects.ts b/src/workflows/projects.ts index b6271bf2fb22..074d2d75bc66 100644 --- a/src/workflows/projects.ts +++ b/src/workflows/projects.ts @@ -1,16 +1,7 @@ import { graphql } from '@octokit/graphql' -// Shared functions for managing projects (memex) - -/** - * The team whose members count as "Docs team" on the review board. - * - * Renamed from `docs` to `technical-content`. GraphQL looks teams up by slug, so a rename - * silently turns the lookup into `null` rather than erroring, which is why the old slug - * kept "working" right up until it didn't. Numeric team IDs survive renames, but the - * GraphQL `team` field only accepts a slug, so this has to be updated by hand if the team - * is renamed again. - */ +// GraphQL resolves teams by slug, not numeric ID. A rename returns null, so +// update this by hand if the Docs team is renamed. const DOCS_TEAM_SLUG = 'technical-content' export interface ProjectV2FieldNode { @@ -106,8 +97,7 @@ export function findSingleSelectID( } } -// Adds the PRs/issues to the project and returns their project item IDs. An -// item already on the board keeps its existing ID. +// Existing project items keep their item IDs instead of creating duplicates. export async function addItemsToProject(items: string[], project: string) { console.log(`Adding ${items} to project ${project}`) @@ -137,8 +127,6 @@ export async function addItemsToProject(items: string[], project: string) { }, }) - // The mutation returns {"item_0":{"item":{"id":ID!}},...}. - const newItemIDs = Object.entries(newItems).map((item) => item[1].item.id) return newItemIDs @@ -153,8 +141,7 @@ export async function addItemToProject(item: string, project: string) { } export async function isDocsTeamMember(login: string) { - // docs-bot and copilot bypass the check so their PRs are treated as though a - // docs team member opened them. + // docs-bot and copilot count as Docs team members without GraphQL lookup. if (login === 'docs-bot' || login === 'copilot') { return true } @@ -180,10 +167,7 @@ export async function isDocsTeamMember(login: string) { }, ) - // `team` is null when the slug no longer resolves, which is what a rename looks like from - // here. Dereferencing it threw and killed the whole job *after* the PR had already been - // added to the board, leaving an item with no fields populated. Fall through to the - // hubber fallback instead so the board stays usable, and say why. + // A renamed team returns null, so fall back instead of leaving the project item unpopulated. const team = data.organization.team if (!team) { console.warn( @@ -224,9 +208,9 @@ export function formatDateForProject(date: Date) { return date.toISOString() } -// `turnaround` days from `datePosted`, plus two days if posted on a Thursday -// or Friday and one if posted on a Saturday. With the default turnaround of 2 -// that lands on a weekday; a larger turnaround can still land on a weekend. +// Due dates add turnaround days, plus two days from Thursday or Friday and one +// from Saturday. With the default turnaround of 2, that lands on a weekday. +// Larger turnaround values can still land on a weekend. // Holidays are not considered. export function calculateDueDate(datePosted: Date, turnaround = 2) { let daysUntilDue @@ -248,11 +232,8 @@ export function calculateDueDate(datePosted: Date, turnaround = 2) { return dueDate } -// A GraphQL mutation that populates these fields on one project item: -// - "Status", "Contributor type" and "Size", passed as request variables -// - "Date posted", today -// - "Review due date", see calculateDueDate -// - "Feature" and "Contributor" +// This mutation populates status, contributor type, size, date posted, review +// due date, feature, and contributor on one project item. export function generateUpdateProjectV2ItemFieldMutation({ item, author, @@ -267,8 +248,7 @@ export function generateUpdateProjectV2ItemFieldMutation({ const datePosted = new Date() const dueDate = calculateDueDate(datePosted, turnaround) - // Builds the mutation for a single field. literal=true means the value is a - // string rather than a variable reference. + // Literal values write strings directly instead of variable references. function generateMutationToUpdateField({ item: itemId, fieldID, @@ -284,8 +264,7 @@ export function generateUpdateProjectV2ItemFieldMutation({ }) { const parsedValue = literal ? `${fieldType}: "${value}"` : `${fieldType}: ${value}` - // Anything outside [a-z0-9] in the mutation ID is a GraphQL parse error, - // so strip it. The result is still unique in practice. + // GraphQL mutation IDs reject characters outside a-z0-9, so strip them. return ` set_${fieldID.slice(1)}_item_${itemId.replaceAll( /[^a-z0-9]/g, @@ -369,7 +348,6 @@ export function generateUpdateProjectV2ItemFieldMutation({ return mutation } -// Guesses the affected docs sets from the files the PR changed. export function getFeature(data: ItemData) { if (data.item.__typename !== 'PullRequest') { return '' @@ -377,9 +355,7 @@ export function getFeature(data: ItemData) { const paths = data.item.files.nodes.map((node) => node.path) - // For docs, docs-internal and docs-early-access, take the docs sets from the - // directories under `content/` that changed. Changes to data files are - // ignored. + // Docs repos derive docs sets from changed content directories and ignore data files. if ( process.env.REPO === 'github/docs-internal' || process.env.REPO === 'github/docs' || @@ -429,7 +405,6 @@ export function getFeature(data: ItemData) { return '' } -// Guesses the size of an item. export function getSize(data: ItemData) { // An issue has no files to measure, so guess small. if (data.item.__typename !== 'PullRequest') { diff --git a/src/workflows/purge-fastly-changed-content.ts b/src/workflows/purge-fastly-changed-content.ts index 3a1cdb46bdad..c2c0ec6ab1b2 100644 --- a/src/workflows/purge-fastly-changed-content.ts +++ b/src/workflows/purge-fastly-changed-content.ts @@ -5,17 +5,13 @@ import { makePageSurrogateKey } from '@/frame/middleware/set-fastly-surrogate-ke import github from './github' import { getActionContext } from './action-context' -// Purges the English content pages whose source changed in a production deploy, by surrogate key. -// Each page has one `language:,path:` covering all versions. -// Fastly's batch purge takes up to 256 keys per request. -// Uses hard purge instead of soft for PR authors to see their changes more quickly. -// `data/` changes and AUTOTITLE produce too many keys. -// Translations only rebuild once per day. +// Purges changed English content pages by surrogate key after production deploys. +// Hard purge lets PR authors see changes quickly. Data changes and AUTOTITLE +// produce too many keys, and translations rebuild daily. const CONTENT_PREFIX = 'content/' -// We only purge English pages: content/*.md is the English source, and -// translations lag behind it, so an English deploy shouldn't evict them. +// Purge only English pages: content/*.md is the English source, and translations lag behind it. const PURGE_LANGUAGE = 'en' // Fastly's batch surrogate-key purge accepts at most 256 keys per request. @@ -27,34 +23,28 @@ const MAX_KEYS_PER_PURGE = 256 // everything rather than paginating through a huge change set. const COMPARE_FILE_LIMIT = 300 -// When Fastly rate-limits us (HTTP 429), retry the batch this many times before -// giving up on it. +// Retry rate-limited HTTP 429 batches this many times before giving up. const PURGE_MAX_RATE_LIMIT_RETRIES = 5 // Every key is purged twice because of Fastly shielding. A purge doesn't reach // every POP at the same instant, so a request arriving in between can repopulate // an already-purged edge node from the not-yet-purged shield, leaving the edge -// holding pre-deploy content again. The second pass evicts that copy. Same -// reasoning as the double purge in purge-fastly.ts; see the "Race conditions" -// section of +// holding pre-deploy content again. The second pass evicts that copy. // https://www.fastly.com/documentation/guides/concepts/cache/purging#race-conditions const PURGE_PASSES = 2 -// How long to wait before the second pass. It has to be long enough that any -// re-populated edge copy already exists, otherwise the second purge runs too -// early and the re-population happens after it. purge-fastly.ts uses the same -// 20s for the same reason: Fastly suggests ~2s, but that has been too short in -// practice. Unlike purge-fastly.ts we don't stagger keys within a pass, because -// that spacing exists to keep whole-language purges from stampeding the backend -// and we only purge the handful of pages that actually changed. +// The second pass waits long enough for repopulated edge copies to exist. +// Otherwise the second purge runs too early, and re-population happens after it. +// purge-fastly.ts uses the same 20s because Fastly's suggested 2s has been too +// short in practice. This does not stagger keys within a pass because spacing +// protects the backend during whole-language purges, and this only purges changed pages. const DELAY_BEFORE_SECOND_PURGE = 20 * 1000 -// Jitter ceiling (ms) added to each backoff so retries that saw the same reset -// timestamp don't wake in lockstep and re-burst. +// Jitter ceiling in ms keeps retries with the same reset timestamp from re-bursting. const PURGE_JITTER_MS = 150 -// Backoff bounds for retrying a rate-limited (429) purge. Additive (linear) -// growth from BASE per attempt, capped at MAX. Fastly's rate-limit window resets +// Backoff bounds for retrying rate-limited purges. Additive linear growth from +// BASE per attempt is capped at MAX. Fastly's rate-limit window resets // on the order of a second, so a batch just needs to wait for the next window. // This backoff also floors any server-provided hint so a hint that resolves to // ~0 can't collapse the retry to 0ms, and MAX caps any server-provided delay so @@ -66,7 +56,7 @@ function sleep(ms: number): Promise { return new Promise((resolve) => setTimeout(resolve, ms)) } -// How long to wait before retrying a rate-limited (429) purge. Prefers Fastly's +// How long to wait before retrying a rate-limited purge. Prefers Fastly's // own hints (Retry-After in seconds or as an HTTP date; else Fastly-RateLimit- // Reset as a Unix timestamp), but floors that hint at the additive backoff so a // stale or current-second reset (which computes to <= 0) can't produce a 0ms @@ -108,10 +98,9 @@ type ChangedFile = { status: string } -// The most recent production deployment that was actually live before `headSha`. -// We diff against this to find what changed in the current deploy. The merge -// queue can batch several PRs into one deploy, so this range can span multiple -// merge commits. That's intentional: we want every changed file in the batch. +// The most recent production deployment that was live before headSha. Diff +// against this to find the current deploy's changes. The merge queue can batch +// several PRs into one deploy, so this range can span multiple merge commits. export async function resolvePreviousProductionSha( octokit: Octokit, owner: string, @@ -133,11 +122,7 @@ export async function resolvePreviousProductionSha( deployment_id: deployment.id, per_page: 30, }) - // Require evidence the sha was actually live: a `success` status. The - // previous live deploy keeps its `success` status in history even after a - // newer deploy marks it `inactive`, so this still finds it. We deliberately - // do NOT accept `inactive` alone, since a deploy that failed and was later - // auto-inactivated never served traffic and would give a wrong base. + // Require success because inactive alone can come from a failed deploy that never served. if (statuses.some((status) => status.state === 'success')) { return deployment.sha } @@ -150,7 +135,7 @@ export async function resolvePreviousProductionSha( // by the redirect that replaces a removed/renamed page plus the short max-age, // so it isn't enumerated here. // -// Note: compareCommitsWithBasehead uses three-dot (merge-base) semantics. For +// compareCommitsWithBasehead uses three-dot merge-base semantics. For // normal forward-moving deploys that equals the tree diff. For a rollback (head // is an ancestor of, or diverged from, the previous live sha) it can miss the // reverted files; those simply fall back to the short max-age refresh, so it's @@ -185,11 +170,11 @@ export async function getChangedContentFiles( } // Map changed content files to their per-page surrogate keys, deduped. One key -// per source page, and that single key covers every version-URL of the page -// (fpt, ghec, each ghes release), so a page that changed once is purged once -// regardless of how many versions it fans out to. The key is derived purely from -// the language and the path under content/, matching what the response -// middleware emits, so we don't need to warm the server or resolve permalinks. +// per source page, and that single key covers every version URL of the page, +// including fpt, ghec, and each GHES release. A page that changed once is purged +// once regardless of how many versions it fans out to. The key is derived purely +// from the language and the path under content/, matching what the response +// middleware emits, so the purge need not warm the server or resolve permalinks. export function contentFilesToPageKeys( changedFiles: ChangedFile[], langCode: string = PURGE_LANGUAGE, @@ -203,7 +188,6 @@ export function contentFilesToPageKeys( return [...keys] } -// Split a list into chunks of at most `size`. export function chunk(items: T[], size: number): T[][] { const batches: T[][] = [] for (let index = 0; index < items.length; index += size) { @@ -212,7 +196,7 @@ export function chunk(items: T[], size: number): T[][] { return batches } -// Hard-purge one batch (<= 256) of surrogate keys. Fastly's batch endpoint is +// Hard-purge one batch of at most 256 surrogate keys. Fastly's batch endpoint is // service-scoped; omitting the soft-purge header makes it a hard purge, so every // object tagged with any listed key is evicted and the next request is a fresh // miss. Retries on HTTP 429, honoring Fastly's rate-limit hint. @@ -239,8 +223,7 @@ async function hardPurgeKeyBatch( ) if (response.ok) return - // Fastly rate limit. fetchWithRetry doesn't retry 429 when throwHttpErrors - // is false, so back off and retry the batch ourselves, honoring Fastly's hint. + // fetchWithRetry does not retry 429 here, so retry the batch and honor Fastly's hint. if (response.status === 429 && attempt < PURGE_MAX_RATE_LIMIT_RETRIES) { const waitMs = rateLimitDelayFn(response, attempt) console.warn( @@ -266,9 +249,9 @@ async function hardPurgeKeyBatch( } } -// Hard-purge every key in batches of <= 256, one batch at a time, then do it all -// again after a delay to clear anything the origin shield re-populated (see -// PURGE_PASSES). Collects failures so one bad batch doesn't drop the rest, then +// Hard-purge every key in batches of at most 256, one batch at a time, then do +// it all again after a delay to clear anything the origin shield repopulated. +// Collects failures so one bad batch doesn't drop the rest, then // throws at the end if any failed so the workflow's failure alerting fires. export async function hardPurgeSurrogateKeys( keys: string[], @@ -299,8 +282,7 @@ export async function hardPurgeSurrogateKeys( } for (let pass = 1; pass <= PURGE_PASSES; pass++) { - // A failed first pass still gets a second one: the later attempt may well - // succeed, and giving up here would guarantee stale content. + // A failed first pass still gets a second one because giving up guarantees stale content. if (pass > 1) { console.log(`Waiting ${DELAY_BEFORE_SECOND_PURGE}ms before pass ${pass}...`) await sleepFn(DELAY_BEFORE_SECOND_PURGE) @@ -334,8 +316,7 @@ async function main() { const baseSha = await resolvePreviousProductionSha(octokit, owner, repo, headSha) if (!baseSha) { - // First-ever deploy, or we couldn't find a prior production deploy. The short - // max-age still refreshes everything, so just no-op rather than fail. + // The short max-age refreshes everything when there is no prior production deploy. console.warn('No previous production deployment found; skipping targeted purge.') return } diff --git a/src/workflows/purge-fastly.ts b/src/workflows/purge-fastly.ts index 6c00ce9c00f4..e70f1dc5b51d 100644 --- a/src/workflows/purge-fastly.ts +++ b/src/workflows/purge-fastly.ts @@ -4,15 +4,9 @@ import { fetchWithRetry } from '@/frame/lib/fetch-utils' import { languageKeys } from '@/languages/lib/languages-server' import { makeLanguageSurrogateKey } from '@/frame/middleware/set-fastly-surrogate-key' -// Single entry point for purging Fastly. It runs in one of three modes: -// -// - --everything -> hard purge the ENTIRE cache via `purge_all`. -// - --surrogate-key -> double-purge that one surrogate key. Search uses -// this for `api-search:`. -// - otherwise -> double-purge `no-language` + each `language:` -// key, the routine post-deploy / manual purge. -// -// --hard forces a hard purge; --everything ignores it and is always hard. +// Purges Fastly by mode: entire cache, one surrogate key, or no-language plus +// every language key. --hard forces hard purges for targeted modes, and +// --everything always hard-purges. const { FASTLY_TOKEN, FASTLY_SERVICE_ID } = process.env @@ -53,6 +47,11 @@ type Options = { everything?: boolean } +// The --everything branch calls purge_all, which ignores soft purge, evicts every +// object, and can spike origin traffic. +// It clears content, no-language, manual-purge assets, search, and every other +// surrogate key. Use targeted purges unless they cannot reach the stale content. +// https://www.fastly.com/documentation/reference/api/purging/ await main(program.opts()) async function main(options: Options) { @@ -62,15 +61,7 @@ async function main(options: Options) { if (!FASTLY_SERVICE_ID) { throw new Error('FASTLY_SERVICE_ID not detected; refusing to purge') } - if (options.everything) { - // NOTE: Fastly's purge_all is always a HARD purge. The `fastly-soft-purge` - // header has no effect here, so every object is evicted immediately and - // origin sees a traffic spike while the cache refills. It clears every - // surrogate key: content, no-language, manual-purge assets, search, and so - // on. A targeted surrogate-key purge cannot reach those, so only reach for - // this when a targeted purge will not do. - // https://www.fastly.com/documentation/reference/api/purging/ console.log('Attempting hard purge of the entire cache...') const result = await fastlyPurge('purge_all') console.log('Fastly purge_all result:', result.status) @@ -85,17 +76,13 @@ async function main(options: Options) { } function languageSurrogateKeys(languagesInput?: string): string[] { - // Put `en` first because contributors write mostly in English and are most - // likely waiting to see their landed changes appear in production. Build a - // new array rather than sorting `languageKeys` in place, which is shared - // state. + // Put en first without mutating shared languageKeys, because contributors often wait on English. const trimmed = languagesInput?.trim() const languages = trimmed ? languagesFromString(trimmed) : ['en', ...languageKeys.filter((lang) => lang !== 'en')] - // The leading no-language key, an empty `makeLanguageSurrogateKey()`, covers - // things like `/api/webhooks` which aren't language specific. + // The empty no-language key covers routes such as /api/webhooks that are not language-specific. return [ makeLanguageSurrogateKey(), ...languages.map((language) => makeLanguageSurrogateKey(language)), @@ -118,44 +105,19 @@ function languagesFromString(str: string): string[] { type PurgePhase = 'first' | 'second' type PurgeOutcome = { key: string; phase: PurgePhase; error?: unknown } -/** - * Double-purge a set of surrogate keys. Per-deploy / manual purges pass - * `no-language` plus each `language:` key; the search reindex passes a - * single `api-search:` key. - * - * Each key is purged twice because of Fastly shielding: the first purge clears - * the edge nodes, the second clears the origin shield. Two delays shape the run. - * DELAY_BETWEEN_KEYS spaces out each key's first purge to avoid a stampeding - * herd on the backend. DELAY_BEFORE_SECOND_PURGE waits long enough for the now- - * stale content to be re-fetched and re-shielded before the second purge clears - * it. Fastly suggests ~2s between surrogate-key purges, but that's been too short - * in practice, so we use a larger margin. See the "Race conditions" section of - * https://www.fastly.com/documentation/guides/concepts/cache/purging#race-conditions - * Its 30s figure is for purge-all, which we don't use. - * - * To avoid serializing the run, we schedule every purge against one wall-clock - * timeline up front instead of blocking on each second purge. Because the second- - * purge delay is a multiple of the between-keys delay, a key's second purge - * shares a slot with a later key's first purge: - * - * t=0 no-language (1st) - * t=10 en (1st) - * t=20 es (1st) + no-language (2nd) - * t=30 ja (1st) + en (2nd) - * t=40 pt (1st) + es (2nd) - * ... - * - * A single-key purge is the degenerate case: first at t=0, second at t=20s. - */ +// purgeKeys double-purges surrogate keys to clear Fastly edge nodes first and the +// origin shield after stale content can be re-fetched. DELAY_BETWEEN_KEYS spaces +// first purges to avoid a backend traffic spike. DELAY_BEFORE_SECOND_PURGE must +// remain a multiple of that delay so second purges share later first-purge slots. +// A single-key purge runs at 0s and 20s. Fastly's 30s figure applies to +// purge_all, not these targeted purges. +// https://www.fastly.com/documentation/guides/concepts/cache/purging#race-conditions async function purgeKeys(surrogateKeys: string[], soft: boolean) { - // One wall-clock start time so the cadence doesn't drift with per-purge network - // latency and each second purge aligns with a later first purge, as above. + // One wall-clock start time keeps network latency from drifting the purge cadence. const startTime = Date.now() const purges: Promise[] = [] - // Each call resolves to an outcome and never rejects: the try/catch keeps a - // failed purge from becoming an unhandled rejection while later scheduled - // purges are still pending. Failures are surfaced after all purges settle. + // Each call resolves to an outcome so later scheduled purges can still finish. async function runPurge( key: string, phase: PurgePhase, @@ -188,14 +150,12 @@ async function purgeKeys(surrogateKeys: string[], soft: boolean) { } } -// Low-level Fastly purge. `endpoint` is appended to the service path, e.g. -// `purge/` or `purge_all`. Returns the response; on a non-2xx throws with -// the body best-effort, since Fastly puts permission/feature details there. -// -// Soft purge marks the object stale and serves stale-while-revalidate; hard -// purge evicts it outright. Soft can fail to clear content whose origin returns -// `304 Not Modified` on revalidation, since a 304 just extends the stale object, -// so use hard then. `purge_all` ignores the soft header and is always hard. +// fastlyPurge appends endpoint to the service path, such as purge/ or +// purge_all. Non-2xx responses throw with the body best-effort because Fastly +// puts permission and feature details there. Soft purge marks the object stale +// and serves stale-while-revalidate; hard purge evicts it outright. Soft can +// fail to clear content whose origin returns 304 Not Modified on revalidation, +// since a 304 extends the stale object. purge_all ignores the soft header. async function fastlyPurge(endpoint: string, { soft = false }: { soft?: boolean } = {}) { const headers: Record = { 'fastly-key': FASTLY_TOKEN as string, diff --git a/src/workflows/ready-for-docs-review.ts b/src/workflows/ready-for-docs-review.ts index bc01a13b4d67..c906242097cc 100644 --- a/src/workflows/ready-for-docs-review.ts +++ b/src/workflows/ready-for-docs-review.ts @@ -13,7 +13,7 @@ import { type ItemData, } from './projects' -// Whether copilot-swe-agent authored the PR, and its first other assignee. +// Copilot PRs use their first non-Copilot assignee as the human contributor. function getCopilotAuthorInfo(data: ItemData): { isCopilotAuthor: boolean copilotAssignee: string @@ -135,14 +135,10 @@ async function run() { const size = getSize(data) const sizeType = findSingleSelectID(size, 'Size', data) - // Check if the author is a bot account (e.g. dependabot[bot], github-actions[bot]). - // GitHub bot logins end with '[bot]' and cannot be resolved as regular GitHub users, - // so we skip any user-specific GraphQL queries for them. + // Bot logins such as dependabot[bot] cannot be resolved as regular GitHub users. const isBotAuthor = (process.env.AUTHOR_LOGIN || '').endsWith('[bot]') - // If this is the OS repo, determine if this is a first time contributor - // If yes, set the author to 'first time contributor' instead of to the author login - // Bot accounts (e.g. dependabot[bot]) are not resolvable as GitHub users, so skip this check. + // github/docs PRs from new non-bot contributors use first time contributor instead of the login. let firstTimeContributor if (!isBotAuthor && process.env.REPO === 'github/docs') { const contributorData: Record = await graphql( @@ -225,7 +221,7 @@ async function run() { let contributorType if (isCopilotAuthor || isBotAuthor) { - // Treat Copilot and bot-authored PRs (e.g. dependabot[bot]) as Docs team + // Treat Copilot and bot-authored PRs as Docs team. contributorType = docsMemberTypeID } else if (await isDocsTeamMember(process.env.AUTHOR_LOGIN || '')) { contributorType = docsMemberTypeID @@ -234,7 +230,7 @@ async function run() { } else if (process.env.REPO === 'github/docs') { contributorType = osContributorTypeID } else { - // use hubber as the fallback so that the PR doesn't get lost on the board + // Fall back to hubber so the PR stays visible on the board. contributorType = hubberTypeID } diff --git a/src/workflows/secondary-ratelimit-retry.ts b/src/workflows/secondary-ratelimit-retry.ts index 33049015f401..4c7cd8cf0329 100644 --- a/src/workflows/secondary-ratelimit-retry.ts +++ b/src/workflows/secondary-ratelimit-retry.ts @@ -3,9 +3,7 @@ import { isRequestError } from '@/workflows/github' const DEFAULT_SLEEPTIME = parseInt(process.env.SECONDARY_RATELIMIT_RETRY_SLEEPTIME || '30000', 10) const DEFAULT_ATTEMPTS = parseInt(process.env.SECONDARY_RATELIMIT_RETRY_ATTEMPTS || '5', 10) -// Secondary rate limits are responded with a 403. The message will contain -// "You have exceeded a secondary rate limit". -// More info about what they are here: +// Secondary rate limits return 403 with "You have exceeded a secondary rate limit." // https://docs.github.com/en/rest/using-the-rest-api/rate-limits-for-the-rest-api?apiVersion=2022-11-28#about-secondary-rate-limits export async function octoSecondaryRatelimitRetry( fn: () => Promise, diff --git a/src/workflows/unallowed-contribution-filters.yml b/src/workflows/unallowed-contribution-filters.yml index 209f066a8d21..5bdd0ce97c16 100644 --- a/src/workflows/unallowed-contribution-filters.yml +++ b/src/workflows/unallowed-contribution-filters.yml @@ -12,6 +12,6 @@ notAllowed: - 'content/README.md' contentTypes: - 'content/**' -# allows getting a list of just added files from the dorny/paths-filter action +# Lets dorny/paths-filter return only added content files. added: - added: 'content/**' diff --git a/src/workflows/unallowed-contributions.ts b/src/workflows/unallowed-contributions.ts index be643e0dc823..010b6d2bc6f2 100755 --- a/src/workflows/unallowed-contributions.ts +++ b/src/workflows/unallowed-contributions.ts @@ -26,17 +26,13 @@ main() async function main() { const unallowedChangedFiles = [...JSON.parse(FILE_PATHS_NOT_ALLOWED || '')] - // Content files that are added in a forked repo won't be in the - // `github/docs` repo, so we don't need to check them. They will be - // reviewed manually by a content writer. + // Added content from forks is absent from github/docs, so content writers review it manually. const contentFilesToCheck: string[] = difference( JSON.parse(CHANGED_FILE_PATHS || ''), JSON.parse(ADDED_CONTENT_FILES || ''), ) - // Any modifications or deletions to a file in the content directory - // could potentially have `type: rai` so each changed content file's - // frontmatter needs to be checked. + // Modified or deleted content can have contentType: rai, so check each file's frontmatter. unallowedChangedFiles.push(...checkContentType(contentFilesToCheck, 'rai')) if (unallowedChangedFiles.length === 0) return @@ -51,7 +47,7 @@ async function main() { "It looks like you've modified some files that we can't accept as contributions." let createdComment - // Add the `invalid` label so the PR gets closed + // The invalid label triggers automatic PR closure. try { await octokit.rest.issues.addLabels({ owner, diff --git a/src/workflows/walk-files.ts b/src/workflows/walk-files.ts index 9e32e4e9e452..2b973ece45a6 100644 --- a/src/workflows/walk-files.ts +++ b/src/workflows/walk-files.ts @@ -1,9 +1,3 @@ -// [start-readme] -// -// A helper that returns an array of files for a given path and file extension. -// -// [end-readme] - import walk from 'walk-sync' import fs from 'fs' diff --git a/src/workflows/writers-help-metadata.ts b/src/workflows/writers-help-metadata.ts index 482982a44b92..0d6cb87051f9 100644 --- a/src/workflows/writers-help-metadata.ts +++ b/src/workflows/writers-help-metadata.ts @@ -27,7 +27,7 @@ interface ScriptMetadata { description?: string } -// Manual entries for scripts that aren't TypeScript files with metadata +// Manual entries cover scripts that do not carry writer-tool metadata. const MANUAL_ENTRIES: WriterToolsCollection = { 'Validation and formatting': [ { name: 'prettier', description: 'Format markdown, YAML, and other files' }, @@ -41,12 +41,9 @@ const MANUAL_ENTRIES: WriterToolsCollection = { async function discoverWriterTools(): Promise { const packageJsonPath = path.join(__dirname, '..', '..', 'package.json') const packageJson = JSON.parse(readFileSync(packageJsonPath, 'utf8')) - const tools: WriterToolsCollection = { ...MANUAL_ENTRIES } // Start with manual entries + const tools: WriterToolsCollection = { ...MANUAL_ENTRIES } - // First get all files. node:fs has no `absolute` option, and its `exclude` - // patterns skip directory contents without skipping the directory entry, so - // the bare directory names are listed too. Neither library guarantees an - // order, so sort to keep the printed listing stable. + // node:fs has no absolute option, excludes leave bare directories, and ordering is unstable. const repoRoot = path.join(__dirname, '..', '..') const allFiles = ( await Array.fromAsync( @@ -67,14 +64,13 @@ async function discoverWriterTools(): Promise { .map((file) => path.resolve(repoRoot, file)) .sort() - // Then filter for .ts, .js, .sh scripts const scriptFiles = allFiles.filter((file) => { - if (file === __scriptname) return false // skip the current file + if (file === __scriptname) return false const ext = path.extname(file) if (['.ts', '.js', '.sh'].includes(ext)) return true - // For extensionless files, check if they're executable or have shebang + // Extensionless executable shell scripts count as writer tools. if (ext === '') { try { const content = readFileSync(file, 'utf8') @@ -94,12 +90,10 @@ async function discoverWriterTools(): Promise { if (metadata.isWriterTool) { metadata.category = getCategory(relativePath) - // Find corresponding npm script const scriptName = findScriptName(packageJson.scripts, relativePath) if (scriptName) { if (!tools[metadata.category]) tools[metadata.category] = [] - // Check if not already added manually const exists = tools[metadata.category].some((tool) => tool.name === scriptName) if (!exists) { tools[metadata.category].push({ @@ -110,7 +104,7 @@ async function discoverWriterTools(): Promise { } } } catch { - // Skip files that can't be read + // Unreadable files are irrelevant to writer-tool discovery. continue } } @@ -120,7 +114,8 @@ async function discoverWriterTools(): Promise { function extractMetadata(content: string): ScriptMetadata { const metadata: ScriptMetadata = {} - const lines = content.split('\n').slice(0, 20) // Only check first 20 lines + // Writer-tool metadata must appear within the first 20 lines. + const lines = content.split('\n').slice(0, 20) for (const line of lines) { if (line.includes(PURPOSE_STRING)) { @@ -139,8 +134,7 @@ function extractMetadata(content: string): ScriptMetadata { return metadata } -// Convert the DIR in src/DIR/ to a title-cased category name -// E.g. src/secret-scanning becomes Secret Scanning +// src/secret-scanning becomes Secret Scanning. function getCategory(relativePath: string): string { const directory = relativePath.split(path.sep)[1] const category = directory @@ -148,7 +142,6 @@ function getCategory(relativePath: string): string { .map((w) => w.charAt(0).toUpperCase() + w.slice(1)) .join(' ') - // Clarify some category names return category .replace('Content Render', 'Content Tasks') .replace('Ghes Releases', 'GHES release notes') @@ -156,11 +149,9 @@ function getCategory(relativePath: string): string { function findScriptName(scripts: Record, relativePath: string): string | null { for (const [scriptName, command] of Object.entries(scripts)) { - // Check if the command includes this file path if (command.includes(relativePath)) { return scriptName } - // Also check for simplified paths without the src/ prefix const simplifiedPath = relativePath.replace(/^src\//, '') if (command.includes(simplifiedPath)) { return scriptName @@ -170,7 +161,6 @@ function findScriptName(scripts: Record, relativePath: string): } function prioritizeOrder(tools: WriterToolsCollection) { - // Define priorities for specific tools const priorities = { 'move-content': 1, 'cta-builder': 2, @@ -178,26 +168,21 @@ function prioritizeOrder(tools: WriterToolsCollection) { dev: 1, } - // Assign priorities to discovered tools for (const tool of Object.values(tools).flat()) { if (priorities[tool.name as keyof typeof priorities]) { tool.priority = priorities[tool.name as keyof typeof priorities] } } - // Sort each category by priority, then alphabetically for (const category of Object.keys(tools)) { tools[category].sort((a, b) => { - // Items with priority come first if (a.priority !== undefined && b.priority === undefined) return -1 if (a.priority === undefined && b.priority !== undefined) return 1 - // Both have priority: sort by priority value if (a.priority !== undefined && b.priority !== undefined) { return a.priority - b.priority } - // Neither has priority: sort alphabetically return a.name.localeCompare(b.name) }) }