Repository navigation
224 lines (209 loc) · 11 KB
/
Copy pathrelease-triage.yml
File metadata and controls
224 lines (209 loc) · 11 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
# .github/workflows/release-triage.yml
#
# Manual triage for freeze week / release promotions.
# Actions tab -> "Report triage" -> Run workflow -> fill in release_tag, leave
# everything else on its default unless you know you need it.
# Creates ONE Taiga user story tagged "release-<tag>" containing one task
# per failure cluster. Re-running with the same tag adds only NEW clusters
# to the SAME story (no duplicates) thanks to the per-release state file.
# You don't need to create anything in Taiga beforehand — the story and its
# tasks are created for you. See .github/workflows/README.md for the full guide.
name: Report triage
on:
workflow_dispatch:
inputs:
release_tag:
description: 'Release tag, e.g. "2.18" — not the full build string. Same tag every day = same story.'
required: true
report_run_id:
description: 'Override: specific run id to triage (default: last scheduled daily run on main)'
required: false
group_by:
description: 'Task grouping: cluster (default, one per bug), file, or folder (snapshot diffs bundled per folder)'
required: false
default: 'cluster'
type: choice
options:
- cluster
- file
- folder
epic_ref:
description: 'Optional: Taiga epic ref (number only, e.g. 118) to link the story under'
required: false
reset_state:
description: 'Reset this release tag: forget all previous triage state (the old story in Taiga is deleted for you automatically)'
required: false
default: false
type: boolean
close_only:
description: "Only close resolved tasks, file nothing new. Don't combine with reset_state"
required: false
default: false
type: boolean
permissions:
contents: read
actions: read # required for gh run list
jobs:
triage:
runs-on: ubuntu-latest
environment: PRE # grants access to MATTERMOST_WEBHOOK_DAILY_REPORT (environment secret)
steps:
- name: Reject reset_state + close_only together
if: ${{ inputs.reset_state == true && inputs.close_only == true }}
run: |
echo "reset_state wipes triage state right before close_only would need it to find what's resolved — pick one." >&2
exit 1
- name: Sanity-check release_tag
run: |
TAG="${{ inputs.release_tag }}"
if [[ "$TAG" =~ -g[0-9a-f]{7,40}$ ]]; then
{
echo "### :warning: release_tag looks like a full build string"
echo "\`$TAG\` ends in a git commit suffix, so this doesn't look like a short release name (e.g. \`2.18\`)."
echo "That's fine if you're intentionally continuing a story that was already opened under this exact tag — otherwise, a full build string changes on every commit, so each run would open its own story instead of accumulating into one."
} >> "$GITHUB_STEP_SUMMARY"
fi
- uses: actions/checkout@v6
- uses: actions/setup-node@v6
with:
node-version: 20
- name: Configure AWS credentials
uses: aws-actions/configure-aws-credentials@v6
with:
aws-access-key-id: ${{ secrets.AWS_ACCESS_KEY_ID }}
aws-secret-access-key: ${{ secrets.AWS_SECRET_ACCESS_KEY }}
aws-region: ${{ secrets.AWS_REGION }}
- name: Resolve report to triage
id: report
run: |
if [ -n "${{ inputs.report_run_id }}" ]; then
RUN_ID="${{ inputs.report_run_id }}"
if ! aws s3 cp "s3://kaleidos-qa-reports/run-$RUN_ID/results.json" results.json; then
echo "::error::No results.json in S3 for run $RUN_ID — double-check the id at https://github.com/${{ github.repository }}/actions/workflows/playwright_pre_daily.yml, or its report may have expired."
exit 1
fi
echo "Using manually provided run: $RUN_ID"
else
# Last finished *scheduled* daily runs on main (completed, not success:
# the test step fails the workflow on failing nights by design), newest
# first. Walk them until one whose report is actually still in S3 —
# `gh run list`'s answer has been seen to lag behind reality by several
# runs when read with the workflow's own GITHUB_TOKEN, and even without
# that, an old report can simply have expired from the bucket.
CANDIDATES=$(gh run list \
--workflow "$DAILY_WORKFLOW" \
--branch main \
--event schedule \
--status completed \
--limit 1 \
--json databaseId \
--jq '.[].databaseId')
if [ -z "$CANDIDATES" ]; then
echo "::error::No completed scheduled daily run found on main. Has 'Daily Penpot Regression Tests on PRE' ever run there?"
exit 1
fi
RUN_ID=""
for CANDIDATE in $CANDIDATES; do
if aws s3 cp "s3://kaleidos-qa-reports/run-$CANDIDATE/results.json" results.json 2>/dev/null; then
RUN_ID="$CANDIDATE"
echo "Using last scheduled daily run on main with a report in S3: $RUN_ID"
break
fi
echo "Run $CANDIDATE has no report in S3 (expired, or failed before upload) — trying the next one."
done
if [ -z "$RUN_ID" ]; then
CANDIDATE_COUNT=$(echo "$CANDIDATES" | grep -c .)
echo "::error::None of the last $CANDIDATE_COUNT completed scheduled daily run(s) on main have a report in S3 — the report may have expired, or the run failed before uploading it. Find an earlier run at https://github.com/${{ github.repository }}/actions/workflows/playwright_pre_daily.yml and re-run 'Report triage' with its id in report_run_id, or wait for tonight's run."
exit 1
fi
fi
echo "run_id=$RUN_ID" >> "$GITHUB_OUTPUT"
aws s3 cp "s3://kaleidos-qa-reports/run-$RUN_ID-enterprise/results.json" results-enterprise.json 2>/dev/null \
&& echo "Found enterprise results for this run — will be triaged alongside the standard suite." \
|| echo "No enterprise results for this run (older run, or the enterprise job didn't run/upload) — triaging standard suite only."
env:
GH_TOKEN: ${{ github.token }}
DAILY_WORKFLOW: playwright_pre_daily.yml # "Daily Penpot Regression Tests on PRE"
- name: Fetch app version for this run
id: appver
run: |
VERSION=$(aws s3 cp "s3://kaleidos-qa-reports/run-${{ steps.report.outputs.run_id }}/app-version.txt" - 2>/dev/null || echo "")
echo "version=$VERSION" >> "$GITHUB_OUTPUT"
echo "App version for this run: ${VERSION:-<not recorded>}"
- name: Pull per-release triage state
run: |
mkdir -p .triage
aws s3 cp "s3://kaleidos-qa-reports/triage/state-release-${{ inputs.release_tag }}.json" \
.triage/state.json || echo "first triage for this release"
- name: Reset per-release triage state
if: ${{ inputs.reset_state == true }}
env:
TAIGA_URL: ${{ secrets.TAIGA_URL }}
TAIGA_USERNAME: ${{ secrets.TAIGA_USERNAME }}
TAIGA_PASSWORD: ${{ secrets.TAIGA_PASSWORD }}
run: |
STORY_ID=""
[ -f .triage/state.json ] && STORY_ID=$(jq -r '.__release_story__.taigaId // empty' .triage/state.json)
if [ -n "$STORY_ID" ]; then
echo "Deleting old Taiga story $STORY_ID for release ${{ inputs.release_tag }}..."
TOKEN=$(curl -sf -X POST "$TAIGA_URL/api/v1/auth" \
-H 'Content-Type: application/json' \
-d "{\"type\":\"normal\",\"username\":\"$TAIGA_USERNAME\",\"password\":\"$TAIGA_PASSWORD\"}" \
| jq -r '.auth_token')
curl -sf -X DELETE "$TAIGA_URL/api/v1/userstories/$STORY_ID" \
-H "Authorization: Bearer $TOKEN" \
-H "User-Agent: qa-triage/1.0 (+github-actions)" \
&& echo "Deleted old story $STORY_ID." \
|| echo "Could not delete story $STORY_ID (already gone, or a permissions issue) — continuing anyway."
else
echo "No stored story for release ${{ inputs.release_tag }} yet — nothing to delete."
fi
rm -f .triage/state.json
echo "State for release ${{ inputs.release_tag }} reset — every failure will be treated as new."
- name: Run report triage
run: |
CLOSE_ONLY_FLAG=""
if [ "${{ inputs.close_only }}" = "true" ]; then
CLOSE_ONLY_FLAG="--close-only"
echo "Close-only sweep: closing resolved tasks, filing nothing new."
fi
ENTERPRISE_RESULTS_FLAG=""
if [ -f results-enterprise.json ]; then
ENTERPRISE_RESULTS_FLAG="--results results-enterprise.json"
fi
npx tsx scripts/triage.ts \
--results results.json \
$ENTERPRISE_RESULTS_FLAG \
--state .triage/state.json \
--release "${{ inputs.release_tag }}" \
--group-by "${{ inputs.group_by || 'cluster' }}" \
--epic-ref "${{ inputs.epic_ref }}" \
--app-version "${{ steps.appver.outputs.version }}" \
--run-id "${{ steps.report.outputs.run_id }}" \
--report-url "https://kaleidos-qa-reports.s3.eu-west-1.amazonaws.com/run-${{ steps.report.outputs.run_id }}/index.html" \
$CLOSE_ONLY_FLAG
env:
GITHUB_TOKEN: ${{ github.token }} # raises GitHub API rate limit for the app-repo changelog fetch
# APP_REPO: penpot/penpot # default; override if the app repo moves
TAIGA_CLOSE_ASSIGNEE: qa.integrations.bot # resolved tasks are assigned to this user when auto-closed
# TAIGA_CLOSED_STATUS: Closed # default; override if the project's closed task status is named differently
TAIGA_URL: ${{ secrets.TAIGA_URL }} # https://api.taiga.io on Taiga cloud
TAIGA_PUBLIC_URL: ${{ vars.TAIGA_PUBLIC_URL }} # https://tree.taiga.io — used for links in the digest
TAIGA_USERNAME: ${{ secrets.TAIGA_USERNAME }}
TAIGA_PASSWORD: ${{ secrets.TAIGA_PASSWORD }}
TAIGA_PROJECT: ${{ vars.TAIGA_PROJECT }}
TAIGA_TAGS: 'needs-triage' # release-<tag> is added automatically
MATTERMOST_WEBHOOK_URL: ${{ secrets.MATTERMOST_WEBHOOK_DAILY_REPORT }} # PRE environment secret -> Autotests Reports channel
# DRY_RUN: "1"
- name: Push per-release triage state
if: always()
run: |
if [ -f .triage/state.json ]; then
aws s3 cp .triage/state.json \
"s3://kaleidos-qa-reports/triage/state-release-${{ inputs.release_tag }}.json"
else
echo "No state file to push (dry run leaves no trace)"
fi
- name: Digest to job summary
if: always()
run: cat .triage/digest.md >> "$GITHUB_STEP_SUMMARY"