Skip to content

DevIndex Data Sync #272

DevIndex Data Sync

DevIndex Data Sync #272

name: DevIndex Data Sync
# The four collection stages, plus publication of the working set.
#
# Cadence is the only lever on throughput: each run is bounded by the GraphQL window rather than by
# `--limit 200`, admitting roughly 153 users, so the schedule sets how stale the tail of the index
# gets — about 1,836 users/day here, a full sweep of 50,000 in roughly 27 days.
on:
schedule:
- cron: '0 */2 * * *'
workflow_dispatch:
inputs:
run_collection:
# Default FALSE, so an ordinary dispatch is a credential probe with no side effects — the
# mutating stages are exactly the ones you cannot safely repeat while fixing an installation.
#
# A SCHEDULED run always collects. `inputs` does not exist on a schedule event, so gating the
# stages on `inputs.run_collection` alone would skip every one of them while reporting success.
description: 'Run the collection stages. Leave off to probe credentials only.'
type: boolean
default: false
concurrency:
group: devindex-data-sync
cancel-in-progress: false
jobs:
collect:
runs-on: ubuntu-latest
permissions:
# The derived files are published as objects and never committed, so this job needs no git write.
contents: read
# Workload Identity Federation: the runner mints an OIDC token that GCP exchanges for a
# short-lived access token, which is why no Google credential is stored here.
id-token: write
# A publish dispatches the Pages workflow, so the site serves what was just published.
actions: write
# A repository variable: the destination is a bucket id, not a credential, and keeping it unmasked is
# what makes a failed publish readable in the log. The stages read it too, to hydrate from the store.
env:
DEVINDEX_PUBLISH_BUCKET: ${{ vars.DEVINDEX_PUBLISH_BUCKET }}
steps:
# The Intake App is installed on the two intake repositories, not on this one, so they are named.
#
# `permission-contents: write` is for the STARGAZER READ, not to write anything: GitHub limits
# that endpoint to admins and collaborators and infers the status from contents-write, and
# `metadata: read` no longer carries it. There is no lesser lever, so the cost is stated rather
# than buried — this token can write code to both intake repositories, and its narrowness rests
# on the App's installation scope.
#
# Read the message, not the step name, when this fails: an `iss` claim error means the App ID is
# not a bare integer (a Client ID stored by mistake), `Not Found` means a named repository has no
# installation, and a permissions error means the grant is missing on the App.
- name: Mint Intake installation token
id: intake-token
uses: actions/create-github-app-token@v3
with:
app-id: ${{ secrets.DATA_SYNC_INTAKE_APP_ID }}
private-key: ${{ secrets.DATA_SYNC_INTAKE_PRIVATE_KEY }}
owner: neomjs
repositories: |
devindex-opt-in
devindex-opt-out
permission-contents: write
permission-issues: write
permission-metadata: read
# Nothing here pushes or compares SHAs, so no history is needed.
- name: Checkout
uses: actions/checkout@v7
with:
fetch-depth: 1
# This repository declares no `engines`; 24 is the version these services are known to run under.
- name: Setup Node
uses: actions/setup-node@v7
with:
node-version: '24'
- name: Install dependencies
run: npm ci
# Safe to repeat while an installation is being corrected, unlike the stages that would
# otherwise serve as the check.
- name: Probe intake access
env:
GH_TOKEN: ${{ steps.intake-token.outputs.token }}
run: |
set -euo pipefail
for repo in devindex-opt-in devindex-opt-out; do
gh api "repos/neomjs/$repo" --jq '"reachable: \(.full_name)"'
done
# Keyless, and NOT gated on `run_collection`: a probe that cannot exercise the publish credential
# checks only half the pipeline, and WIF authentication has no side effects. It runs BEFORE the stages
# because they hydrate from the store with its access token.
- name: Authenticate to Google Cloud
id: gcp-auth
uses: google-github-actions/auth@v3
with:
workload_identity_provider: ${{ secrets.WIF_PROVIDER }}
service_account: ${{ secrets.PUBLISH_SA_EMAIL }}
token_format: access_token
- name: Setup Cloud SDK
uses: google-github-actions/setup-gcloud@v3
# Proves what a credential check usually misses: not that the token minted, but that this
# identity can see the destination. A binding on the wrong bucket authenticates perfectly and
# fails at the first write.
- name: Probe publish access
run: |
set -euo pipefail
if [ -z "${DEVINDEX_PUBLISH_BUCKET}" ]; then
echo "::error::DEVINDEX_PUBLISH_BUCKET is not set. Add it as a repository variable."
exit 1
fi
# Three outcomes, classified rather than collapsed. A listing that matches nothing is a
# destination this identity CAN read that holds nothing yet — the state before the first
# publish, which the publish below repairs; failing it would keep the store empty forever.
# A 403 on `getAccessToken` is impersonation being refused, usually IAM propagation, while a
# 403 on `storage.objects.list` is the bucket grant: the first resolves itself, the second never will.
if OUT=$(gcloud storage ls "${DEVINDEX_PUBLISH_BUCKET}/" 2>&1); then
echo "Publish identity can read the destination."
elif echo "$OUT" | grep -q "matched no objects"; then
echo "Publish identity can read the destination; it holds nothing yet."
elif echo "$OUT" | grep -q "impersonated credentials\|getAccessToken"; then
echo "::error::Impersonation refused. If the binding was just created, wait a few minutes and re-run. If it persists, the binding or the provider condition is wrong."
echo "$OUT" | head -5
exit 1
else
echo "::error::Authenticated, but cannot list the destination — the bucket grant, not the identity."
echo "$OUT" | head -5
exit 1
fi
- name: Report probe-only run
if: ${{ github.event_name != 'schedule' && !inputs.run_collection }}
run: echo "run_collection was not set — credential probe only, no collection performed."
# The ORDER is a privacy guarantee: opt-in and opt-out are processed before discovery and
# enrichment, so somebody who asked to be removed cannot be indexed by the Spider in the same
# run that honoured the request. The first stage hydrates from the store; the later ones keep
# what the earlier ones wrote (`Storage#hydrateOncePerRun`).
- name: DevIndex Opt-In
if: ${{ github.event_name == 'schedule' || inputs.run_collection }}
env:
DEVINDEX_STORE_TOKEN: ${{ steps.gcp-auth.outputs.access_token }}
GH_TOKEN: ${{ steps.intake-token.outputs.token }}
run: npm run devindex:optin
- name: DevIndex Opt-Out
if: ${{ github.event_name == 'schedule' || inputs.run_collection }}
env:
DEVINDEX_STORE_TOKEN: ${{ steps.gcp-auth.outputs.access_token }}
GH_TOKEN: ${{ steps.intake-token.outputs.token }}
run: npm run devindex:optout
- name: DevIndex Spider
if: ${{ github.event_name == 'schedule' || inputs.run_collection }}
env:
DEVINDEX_STORE_TOKEN: ${{ steps.gcp-auth.outputs.access_token }}
GH_TOKEN: ${{ steps.intake-token.outputs.token }}
run: npm run devindex:spider -- --strategy random
# The 200-candidate argument is a ceiling, not a promise: GraphQL cost admission may allow fewer.
- name: DevIndex Updater
if: ${{ github.event_name == 'schedule' || inputs.run_collection }}
env:
DEVINDEX_STORE_TOKEN: ${{ steps.gcp-auth.outputs.access_token }}
GH_TOKEN: ${{ steps.intake-token.outputs.token }}
run: npm run devindex:update
- name: Publish the working set
if: ${{ github.event_name == 'schedule' || inputs.run_collection }}
run: node ./buildScripts/publishWorkingSet.mjs
# `workflow_dispatch` is one of the two events a GITHUB_TOKEN may start. A publish that failed never gets here.
- name: Redeploy the site
if: ${{ github.event_name == 'schedule' || inputs.run_collection }}
env:
GH_TOKEN: ${{ github.token }}
run: gh workflow run pages.yml --repo "${{ github.repository }}" --ref dev