-
Notifications
You must be signed in to change notification settings - Fork 0
346 lines (333 loc) · 14.7 KB
/
Copy pathci.yml
File metadata and controls
346 lines (333 loc) · 14.7 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
name: CI
on:
push:
branches: [main]
pull_request:
workflow_dispatch:
inputs:
suite:
description: Which live eval suite to run (manual dispatch only).
type: choice
options: [both, static, lift]
default: both
leaderboard:
description: >-
Leaderboard seeding (manual dispatch only): skip; build the board from the committed +
fetched bundles; or seed (also live-record the second seed model, spending MODELS_PAT).
type: choice
options: [skip, build, seed]
default: skip
publish_space:
description: >-
After build/seed, push the rebuilt board to the static Hugging Face Space
(needs the HF_TOKEN secret and the LEADERBOARD_SPACE_REPO variable).
type: boolean
default: false
concurrency:
group: ci-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
test:
name: test (${{ matrix.os }}, py${{ matrix.python-version }})
runs-on: ${{ matrix.os }}
strategy:
fail-fast: false
matrix:
os: [ubuntu-latest]
python-version: ['3.10', '3.11', '3.12']
include:
- os: windows-latest
python-version: '3.12'
env:
UV_PYTHON: ${{ matrix.python-version }}
# The suite mocks the embedder everywhere (FakeEmbedder / stubbed retriever), so no test loads
# the real model. Enforce that: offline mode fails fast and loud if a test ever reaches for
# HuggingFace, instead of silently downloading the model and flaking on a 429 rate limit.
HF_HUB_OFFLINE: '1'
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync --all-groups
# `not gmat` keeps the dynamic dry-run tier out of the GMAT-free matrix; it runs in the gated
# `dry-run` job below (setup-gmat + the [gmat] extra). The dry-run worker is omitted from the
# coverage gate, so this `--cov` run stays GMAT-free and still meets the 90% floor.
- run: uv run pytest -m "not eval_smoke and not gmat" --cov --cov-report=term-missing
lint:
name: lint
runs-on: ubuntu-latest
env:
UV_PYTHON: '3.12'
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync --all-groups
- run: uv run ruff check
- run: uv run ruff format --check
typecheck:
name: typecheck
runs-on: ubuntu-latest
env:
UV_PYTHON: '3.12'
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync --all-groups
- run: uv run mypy
eval-smoke:
name: eval smoke (recorded, deterministic)
runs-on: ubuntu-latest
env:
UV_PYTHON: '3.12'
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync --all-groups
# Replays recorded completions + judge verdicts. Zero model calls, zero quota.
- run: uv run pytest -m eval_smoke
gated-eval:
name: gated eval (live, GitHub Models)
# Live inference runs ONLY on manual dispatch (decision D7): per-PR CI is the deterministic
# recorded replay above, so there is no per-PR inference budget to subset. This job is the
# full-suite live run / fixture-refresh path; it needs a personal MODELS_PAT (the workflow
# GITHUB_TOKEN is unreliable for GitHub Models inference). The `suite` dispatch input selects
# which live eval(s) run, so this static suite can be refreshed without the close-the-loop run.
if: >-
github.event_name == 'workflow_dispatch'
&& (inputs.suite == 'static' || inputs.suite == 'both')
runs-on: ubuntu-latest
permissions:
models: read
contents: read
env:
UV_PYTHON: '3.12'
MODELS_PAT: ${{ secrets.MODELS_PAT }}
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync --all-groups
- run: >
uv run gmat-copilot eval --live
--prompts tests/data/eval/prompts.json
-m github:openai/gpt-4.1-mini
--judge-model openai/gpt-4.1-mini
--pace 4.5
minimal-install:
name: minimal install (no extras)
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v6
- uses: actions/setup-python@v6
with:
python-version: '3.12'
- run: python -m pip install --upgrade pip
- run: python -m pip install .
- name: import cleanly with no extras
run: python -c "import gmat_copilot; print(gmat_copilot.__version__)"
- name: provider SDKs and the GMAT stack must not be pulled
run: |
python - <<'PY'
import importlib.util as u
for mod in ("anthropic", "openai", "ollama", "gmat_run"):
assert u.find_spec(mod) is None, f"{mod} leaked into the base install"
print("base install is light: no provider SDK, no GMAT stack")
PY
extras-smoke:
name: extras resolve and import
runs-on: ubuntu-latest
env:
UV_PYTHON: '3.12'
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync --all-groups --all-extras
- run: uv run python -c "import anthropic, openai, ollama; import gmat_copilot; print('extras import OK')"
dry-run:
name: dry-run tier (setup-gmat, [gmat] extra)
# The dynamic validation tier needs a real GMAT install, supplied by setup-gmat. Kept off the
# base matrix (which is GMAT-free) and gated here; the dry-run worker is omitted from coverage,
# so this job runs the gmat-marked tests without the coverage gate. These tests use no model
# (the close-the-loop verification re-checks the recorded lift verdicts against a real GMAT), so
# the job stays inference-free — the per-PR rate limit is never touched.
runs-on: ubuntu-latest
timeout-minutes: 30
env:
UV_PYTHON: '3.12'
HF_HUB_OFFLINE: '1'
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- uses: astro-tools/setup-gmat@v0
with:
version: R2026a
cache: true
- run: uv sync --all-groups --extra gmat
- run: uv run pytest -m gmat
gated-lift-eval:
name: gated close-the-loop eval (live, GitHub Models + setup-gmat)
# The live close-the-loop eval drives a real model AND a real GMAT dry-run, so it runs ONLY on
# manual dispatch (decision D7): per-PR CI replays the recorded lift (the eval-smoke job) and
# re-checks the recorded verdicts against a real GMAT (the dry-run job), neither of which calls a
# model. This job is the on-demand live run / fixture-refresh path; it needs a personal MODELS_PAT
# and is paced to respect the free-tier daily budget. The `suite` dispatch input selects which
# live eval(s) run, so the close-the-loop suite can be refreshed without the static run.
if: >-
github.event_name == 'workflow_dispatch'
&& (inputs.suite == 'lift' || inputs.suite == 'both')
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
models: read
contents: read
env:
UV_PYTHON: '3.12'
MODELS_PAT: ${{ secrets.MODELS_PAT }}
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- uses: astro-tools/setup-gmat@v0
with:
version: R2026a
cache: true
- run: uv sync --all-groups --extra gmat
- run: >
uv run gmat-copilot eval --lift
--prompts tests/data/eval_lift/prompts.json
-m github:openai/gpt-4.1-mini
--judge-model openai/gpt-4.1-mini
--budget 2
--pace 4.5
extension:
name: vscode extension
# Builds the editor surface (decision D15) and runs its activation smoke test headless. The
# engine worker it drives is pure Python and covered by the test job above; here we only build
# the TypeScript bundle, type-check, lint, run the activation test, and validate the .vsix.
# Publishing to the Marketplace / Open VSX is a separate release concern.
runs-on: ubuntu-latest
timeout-minutes: 20
defaults:
run:
working-directory: editors/vscode
steps:
- uses: actions/checkout@v6
- uses: actions/setup-node@v6
with:
node-version: '22'
cache: npm
cache-dependency-path: editors/vscode/package-lock.json
- run: npm ci
- name: Lint, type-check, and run the activation smoke test (headless)
run: xvfb-run -a npm test
- name: Package the .vsix
run: npx @vscode/vsce package --out gmat-copilot.vsix
- uses: actions/upload-artifact@v7
with:
name: vscode-extension-vsix
path: editors/vscode/gmat-copilot.vsix
if-no-files-found: error
leaderboard:
name: leaderboard (build + verify, gated)
# The per-model leaderboard engine (decision D16). Manual dispatch only: the held-out score uses
# the LLM judge's frozen gold verdicts from a private store, and `seed` spends live GitHub Models
# quota — neither belongs on the per-PR path. `build` assembles the board from the committed
# public bundle plus any held-out fetched from the private store; `seed` additionally live-records
# the second seed model. The board is uploaded as an artifact; rendering it on the static HF Space
# is a separate surface. The committed board (leaderboard/leaderboard.json) is the gpt-4.1-mini
# public anchor; this job refreshes it with held-out scores + additional seeds when configured.
if: >-
github.event_name == 'workflow_dispatch'
&& (inputs.leaderboard == 'build' || inputs.leaderboard == 'seed')
runs-on: ubuntu-latest
timeout-minutes: 60
permissions:
models: read
contents: read
env:
UV_PYTHON: '3.12'
# MODELS_PAT drives the `seed` live record (the workflow GITHUB_TOKEN is unreliable for GitHub
# Models). HF_TOKEN + the HELDOUT_DATASET repo variable fetch the never-committed held-out from
# its private HF Dataset (decision D16); until that dataset is authored (release cut), the fetch
# is skipped and every held-out cell stays pending.
MODELS_PAT: ${{ secrets.MODELS_PAT }}
HF_TOKEN: ${{ secrets.HF_TOKEN }}
HELDOUT_DATASET: ${{ vars.LEADERBOARD_HELDOUT_DATASET }}
# The static HF Space the rebuilt board is published to (the project's one hosted artifact,
# decision D16). Until the Space is created and this variable is set, the publish step is
# skipped; the board still builds, verifies, and uploads as an artifact.
SPACE_REPO: ${{ vars.LEADERBOARD_SPACE_REPO }}
steps:
- uses: actions/checkout@v6
- uses: astral-sh/setup-uv@v7
with:
enable-cache: true
cache-dependency-glob: uv.lock
- run: uv sync --all-groups
- name: Record the second seed's public bundle (live; paced for the free daily cap)
if: inputs.leaderboard == 'seed'
continue-on-error: true
run: |
DIR=leaderboard/.cache/public/openai__gpt-4o-mini
mkdir -p "$DIR"
cp tests/data/eval/prompts.json "$DIR/prompts.json"
uv run gmat-copilot eval --record "$DIR" \
-m github:openai/gpt-4o-mini \
--judge-model openai/gpt-4.1-mini \
--pace 4.5
# `eval --record` keys judge.json by the judge model; normalise it to the flat,
# model-agnostic {prompt_id: verdicts} form the committed gold bundle uses, so the board's
# run_recorded replay scores this generation model correctly (decisions D7, D11).
uv run python -c 'import json, sys; p = sys.argv[1]; d = json.load(open(p, encoding="utf-8")); d = next(iter(d.values())) if d and all(isinstance(v, dict) for v in d.values()) else d; json.dump(d, open(p, "w", encoding="utf-8"), indent=2, sort_keys=True)' "$DIR/judge.json"
- name: Fetch the private held-out bundles (if the dataset is configured)
if: env.HF_TOKEN != '' && env.HELDOUT_DATASET != ''
# The HF Hub Python API (stable across the huggingface-cli -> hf CLI rename). HF_TOKEN
# authenticates the read of the private dataset; an empty dataset downloads to an empty dir
# and leaves every held-out cell pending.
run: |
uvx --from huggingface_hub python -c 'import os; from huggingface_hub import snapshot_download; snapshot_download(repo_id=os.environ["HELDOUT_DATASET"], repo_type="dataset", local_dir="leaderboard/.cache/heldout")'
- name: Build the board
run: |
uv run gmat-copilot leaderboard build \
--out leaderboard/leaderboard.json \
--held-out leaderboard/.cache/heldout
- name: Verify the board reproduces offline and leaks no held-out gold
run: uv run gmat-copilot leaderboard verify leaderboard/leaderboard.json
- name: Publish the board to the static HF Space
# The Space is the project's one hosted artifact (decision D16). The preceding `verify` step
# is the firewall: it fails the job if the board is not aggregate-only, so a held-out gold can
# never reach the Space. The Space payload is the committed static front end plus the freshly
# built board, pushed via the HF Hub API; the published JSON carries aggregates only.
if: inputs.publish_space && env.HF_TOKEN != '' && env.SPACE_REPO != ''
# The HF Hub Python API (stable across the huggingface-cli -> hf CLI rename). The Space payload
# is the committed static front end plus the freshly built board; the published JSON carries
# aggregates only (the preceding verify step is the firewall).
run: |
cp leaderboard/leaderboard.json leaderboard/space/leaderboard.json
uvx --from huggingface_hub python -c 'import os, json; from huggingface_hub import upload_folder; v = json.load(open("leaderboard/leaderboard.json"))["eval_protocol_version"]; upload_folder(repo_id=os.environ["SPACE_REPO"], repo_type="space", folder_path="leaderboard/space", commit_message=f"Refresh leaderboard (eval protocol {v})")'
- uses: actions/upload-artifact@v7
with:
name: leaderboard
path: leaderboard/leaderboard.json
if-no-files-found: error