Skip to content

Report triage

Report triage #76

# .github/workflows/release-triage.yml
#
# Manual triage for freeze week / release promotions.
# Actions tab -> "Report triage" -> Run workflow -> fill in release_tag, leave
# everything else on its default unless you know you need it.
# Creates ONE Taiga user story tagged "release-<tag>" containing one task
# per failure cluster. Re-running with the same tag adds only NEW clusters
# to the SAME story (no duplicates) thanks to the per-release state file.
# You don't need to create anything in Taiga beforehand — the story and its
# tasks are created for you. See .github/workflows/README.md for the full guide.
name: Report triage
on:
workflow_dispatch:
inputs:
release_tag:
description: 'Release tag, e.g. "2.18" — not the full build string. Same tag every day = same story.'
required: true
report_run_id:
description: 'Override: specific run id to triage (default: last scheduled daily run on main)'
required: false
group_by:
description: 'Task grouping: cluster (default, one per bug), file, or folder (snapshot diffs bundled per folder)'
required: false
default: 'cluster'
type: choice
options:
- cluster
- file
- folder
epic_ref:
description: 'Optional: Taiga epic ref (number only, e.g. 118) to link the story under'
required: false
reset_state:
description: 'Reset this release tag: forget all previous triage state (the old story in Taiga is deleted for you automatically)'
required: false
default: false
type: boolean
close_only:
description: "Only close resolved tasks, file nothing new. Don't combine with reset_state"
required: false
default: false
type: boolean
permissions:
contents: read
actions: read # required for gh run list
jobs:
triage:
runs-on: ubuntu-latest
environment: PRE # grants access to MATTERMOST_WEBHOOK_DAILY_REPORT (environment secret)
steps:
- name: Reject reset_state + close_only together
if: ${{ inputs.reset_state == true && inputs.close_only == true }}
run: |
echo "reset_state wipes triage state right before close_only would need it to find what's resolved — pick one." >&2
exit 1
- name: Sanity-check release_tag
run: |
TAG="${{ inputs.release_tag }}"
if [[ "$TAG" =~ -g[0-9a-f]{7,40}$ ]]; then
{
echo "### :warning: release_tag looks like a full build string"
echo "\`$TAG\` ends in a git commit suffix, so this doesn't look like a short release name (e.g. \`2.18\`)."
echo "That's fine if you're intentionally continuing a story that was already opened under this exact tag — otherwise, a full build string changes on every commit, so each run would open its own story instead of accumulating into one."
} >> "$GITHUB_STEP_SUMMARY"
fi
- uses: actions/checkout@v6
- uses: actions/setup-node@v6
with:
node-version: 20
- name: Configure AWS credentials
uses: aws-actions/configure-aws-credentials@v6
with:
aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }}
aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
aws-region: ${{ secrets.AWS_REGION }}
- name: Resolve report to triage
id: report
run: |
if [ -n "${{ inputs.report_run_id }}" ]; then
RUN_ID="${{ inputs.report_run_id }}"
if ! aws s3 cp "s3://kaleidos-qa-reports/run-$RUN_ID/results.json" results.json; then
echo "::error::No results.json in S3 for run $RUN_ID — double-check the id at https://github.com/${{ github.repository }}/actions/workflows/playwright_pre_daily.yml, or its report may have expired."
exit 1
fi
echo "Using manually provided run: $RUN_ID"
else
# Last finished *scheduled* daily runs on main (completed, not success:
# the test step fails the workflow on failing nights by design), newest
# first. Walk them until one whose report is actually still in S3 —
# `gh run list`'s answer has been seen to lag behind reality by several
# runs when read with the workflow's own GITHUB_TOKEN, and even without
# that, an old report can simply have expired from the bucket.
CANDIDATES=$(gh run list \
--workflow "$DAILY_WORKFLOW" \
--branch main \
--event schedule \
--status completed \
--limit 1 \
--json databaseId \
--jq '.[].databaseId')
if [ -z "$CANDIDATES" ]; then
echo "::error::No completed scheduled daily run found on main. Has 'Daily Penpot Regression Tests on PRE' ever run there?"
exit 1
fi
RUN_ID=""
for CANDIDATE in $CANDIDATES; do
if aws s3 cp "s3://kaleidos-qa-reports/run-$CANDIDATE/results.json" results.json 2>/dev/null; then
RUN_ID="$CANDIDATE"
echo "Using last scheduled daily run on main with a report in S3: $RUN_ID"
break
fi
echo "Run $CANDIDATE has no report in S3 (expired, or failed before upload) — trying the next one."
done
if [ -z "$RUN_ID" ]; then
CANDIDATE_COUNT=$(echo "$CANDIDATES" | grep -c .)
echo "::error::None of the last $CANDIDATE_COUNT completed scheduled daily run(s) on main have a report in S3 — the report may have expired, or the run failed before uploading it. Find an earlier run at https://github.com/${{ github.repository }}/actions/workflows/playwright_pre_daily.yml and re-run 'Report triage' with its id in report_run_id, or wait for tonight's run."
exit 1
fi
fi
echo "run_id=$RUN_ID" >> "$GITHUB_OUTPUT"
aws s3 cp "s3://kaleidos-qa-reports/run-$RUN_ID-enterprise/results.json" results-enterprise.json 2>/dev/null \
&& echo "Found enterprise results for this run — will be triaged alongside the standard suite." \
|| echo "No enterprise results for this run (older run, or the enterprise job didn't run/upload) — triaging standard suite only."
env:
GH_TOKEN: ${{ github.token }}
DAILY_WORKFLOW: playwright_pre_daily.yml # "Daily Penpot Regression Tests on PRE"
- name: Fetch app version for this run
id: appver
run: |
VERSION=$(aws s3 cp "s3://kaleidos-qa-reports/run-${{ steps.report.outputs.run_id }}/app-version.txt" - 2>/dev/null || echo "")
echo "version=$VERSION" >> "$GITHUB_OUTPUT"
echo "App version for this run: ${VERSION:-<not recorded>}"
- name: Pull per-release triage state
run: |
mkdir -p .triage
aws s3 cp "s3://kaleidos-qa-reports/triage/state-release-${{ inputs.release_tag }}.json" \
.triage/state.json || echo "first triage for this release"
- name: Reset per-release triage state
if: ${{ inputs.reset_state == true }}
env:
TAIGA_URL: ${{ secrets.TAIGA_URL }}
TAIGA_USERNAME: ${{ secrets.TAIGA_USERNAME }}
TAIGA_PASSWORD: ${{ secrets.TAIGA_PASSWORD }}
run: |
STORY_ID=""
[ -f .triage/state.json ] && STORY_ID=$(jq -r '.__release_story__.taigaId // empty' .triage/state.json)
if [ -n "$STORY_ID" ]; then
echo "Deleting old Taiga story $STORY_ID for release ${{ inputs.release_tag }}..."
TOKEN=$(curl -sf -X POST "$TAIGA_URL/api/v1/auth" \
-H 'Content-Type: application/json' \
-d "{\"type\":\"normal\",\"username\":\"$TAIGA_USERNAME\",\"password\":\"$TAIGA_PASSWORD\"}" \
| jq -r '.auth_token')
curl -sf -X DELETE "$TAIGA_URL/api/v1/userstories/$STORY_ID" \
-H "Authorization: Bearer $TOKEN" \
-H "User-Agent: qa-triage/1.0 (+github-actions)" \
&& echo "Deleted old story $STORY_ID." \
|| echo "Could not delete story $STORY_ID (already gone, or a permissions issue) — continuing anyway."
else
echo "No stored story for release ${{ inputs.release_tag }} yet — nothing to delete."
fi
rm -f .triage/state.json
echo "State for release ${{ inputs.release_tag }} reset — every failure will be treated as new."
- name: Run report triage
run: |
CLOSE_ONLY_FLAG=""
if [ "${{ inputs.close_only }}" = "true" ]; then
CLOSE_ONLY_FLAG="--close-only"
echo "Close-only sweep: closing resolved tasks, filing nothing new."
fi
ENTERPRISE_RESULTS_FLAG=""
if [ -f results-enterprise.json ]; then
ENTERPRISE_RESULTS_FLAG="--results results-enterprise.json"
fi
npx tsx scripts/triage.ts \
--results results.json \
$ENTERPRISE_RESULTS_FLAG \
--state .triage/state.json \
--release "${{ inputs.release_tag }}" \
--group-by "${{ inputs.group_by || 'cluster' }}" \
--epic-ref "${{ inputs.epic_ref }}" \
--app-version "${{ steps.appver.outputs.version }}" \
--run-id "${{ steps.report.outputs.run_id }}" \
--report-url "https://kaleidos-qa-reports.s3.eu-west-1.amazonaws.com/run-${{ steps.report.outputs.run_id }}/index.html" \
$CLOSE_ONLY_FLAG
env:
GITHUB_TOKEN: ${{ github.token }} # raises GitHub API rate limit for the app-repo changelog fetch
# APP_REPO: penpot/penpot # default; override if the app repo moves
TAIGA_CLOSE_ASSIGNEE: qa.integrations.bot # resolved tasks are assigned to this user when auto-closed
# TAIGA_CLOSED_STATUS: Closed # default; override if the project's closed task status is named differently
TAIGA_URL: ${{ secrets.TAIGA_URL }} # https://api.taiga.io on Taiga cloud
TAIGA_PUBLIC_URL: ${{ vars.TAIGA_PUBLIC_URL }} # https://tree.taiga.io — used for links in the digest
TAIGA_USERNAME: ${{ secrets.TAIGA_USERNAME }}
TAIGA_PASSWORD: ${{ secrets.TAIGA_PASSWORD }}
TAIGA_PROJECT: ${{ vars.TAIGA_PROJECT }}
TAIGA_TAGS: 'needs-triage' # release-<tag> is added automatically
MATTERMOST_WEBHOOK_URL: ${{ secrets.MATTERMOST_WEBHOOK_DAILY_REPORT }} # PRE environment secret -> Autotests Reports channel
# DRY_RUN: "1"
- name: Push per-release triage state
if: always()
run: |
if [ -f .triage/state.json ]; then
aws s3 cp .triage/state.json \
"s3://kaleidos-qa-reports/triage/state-release-${{ inputs.release_tag }}.json"
else
echo "No state file to push (dry run leaves no trace)"
fi
- name: Digest to job summary
if: always()
run: cat .triage/digest.md >> "$GITHUB_STEP_SUMMARY"