Repository navigation
katas #155
This file contains hidden or bidirectional Unicode text that may be interpreted or compiled differently than what appears below. To review, open the file in an editor that reveals hidden Unicode characters.
Learn more about bidirectional Unicode characters
| # katas.yml — engine kata regression gate | |
| # | |
| # INTERIM workflow (tsc cycles #36 + #38). | |
| # This file is the v1 local implementation of the katas-in-CI gate. | |
| # It will be replaced by canonical templates landed by: | |
| # - cnos #344 Cycle B (template authoring) | |
| # - tsc cycle C-2 (template adoption) | |
| # Until then, this is the source of truth for kata CI. Both the | |
| # build-from-HEAD pre-merge gate (`run-katas`) AND the published-binary | |
| # validation job (`validate-published-binary`, added cycle #38 AC3+AC4) | |
| # will be replaced by canonical cnos #344 Cycle B templates once those | |
| # land. | |
| # | |
| # Surface contract: | |
| # - Auto-discovers every directory under katas/ (no hard-coded kata names). | |
| # - Invokes `coh --kata <id> --mode mechanical` for each kata. | |
| # - Fails the job on any non-zero exit from the kata runner. | |
| # - Caches OPAM + dune _build keyed on src/engine/ocaml/dune-project + | |
| # src/engine/ocaml/tsc_engine.opam so dep / build-config changes | |
| # invalidate cleanly. | |
| # - Persists per-kata result JSON to `.kata-results/<id>.json` and | |
| # uploads as an artifact (cycle #38 AC1); emits a markdown summary | |
| # table to $GITHUB_STEP_SUMMARY (cycle #38 AC2). | |
| # | |
| # Consolidation note: this workflow replaces the previous `kata-check` | |
| # job that lived in ci.yml (removed in cycle #36 R2). That job ran the | |
| # same `bash scripts/run-katas.sh` regression check on every push + | |
| # PR but had no OPAM/dune build cache (every run cold ~5–8 min) and | |
| # no concurrency control. Cycle #36 consolidates kata-running into | |
| # this dedicated workflow. | |
| name: katas | |
| on: | |
| push: | |
| branches: [main] | |
| pull_request: | |
| # Cycle #38 AC3 — published-binary validation triggers. | |
| # `release: published` fires once per GitHub Release publication | |
| # (covers the v* tag-push flow handled by release.yml). The weekly | |
| # cron catches infra drift (libcurl point release, runner image | |
| # change, opam-repository state) independent of code-change cadence | |
| # — issue #38 §Open question 2 recommendation: weekly Mon 06:00 UTC. | |
| # `workflow_dispatch` lets operators trigger the published-binary | |
| # job manually (e.g. after suspected infra regressions) without | |
| # waiting for the next cron tick. | |
| release: | |
| types: [published] | |
| schedule: | |
| - cron: '0 6 * * 1' | |
| workflow_dispatch: | |
| # Per issue #36 open question 2: cancel superseded PR runs, never cancel main. | |
| # Per cycle #38: release / schedule / workflow_dispatch runs target the | |
| # `validate-published-binary` job — they share this top-level concurrency | |
| # group with the pre-merge `run-katas` job, but in practice never collide | |
| # because their trigger sources never produce overlapping refs with PR / | |
| # push-to-main. Splitting concurrency per-job is unnecessary for v1. | |
| concurrency: | |
| group: katas-${{ github.ref }} | |
| cancel-in-progress: ${{ github.event_name == 'pull_request' }} | |
| jobs: | |
| run-katas: | |
| name: run-katas (auto-discovered) | |
| runs-on: ubuntu-22.04 | |
| steps: | |
| - uses: actions/checkout@v4 | |
| - name: Install system depexts (libcurl for ezcurl) | |
| # Pre-install libcurl4-openssl-dev so opam-depext / setup-ocaml | |
| # does not need to resolve the (occasionally 404ing) gnutls flavour. | |
| # Mirrors ci.yml build job; see cycle #26 β observation, #32 AC5. | |
| run: | | |
| sudo apt-get update | |
| sudo apt-get install -y --no-install-recommends libcurl4-openssl-dev pkg-config | |
| # AC3 — Build cache. Two-layer cache keyed on the engine's dune | |
| # build/dep manifests so any dep or build-config change invalidates | |
| # cleanly. OS dimension included so cache is not shared across | |
| # runner images. Both `dune-project` (generate_opam_files true) | |
| # and `tsc_engine.opam` (its generated output) move together but | |
| # we hash both because either being edited directly is in scope. | |
| - name: Cache OPAM + dune | |
| id: cache-opam-dune | |
| uses: actions/cache@v4 | |
| with: | |
| path: | | |
| ~/.opam | |
| src/engine/ocaml/_build | |
| key: katas-${{ runner.os }}-ocaml-5.2-${{ hashFiles('src/engine/ocaml/dune-project', 'src/engine/ocaml/tsc_engine.opam') }} | |
| restore-keys: | | |
| katas-${{ runner.os }}-ocaml-5.2- | |
| - name: Set up OCaml | |
| uses: ocaml/setup-ocaml@v3 | |
| with: | |
| ocaml-compiler: "5.2" | |
| - name: Install dependencies | |
| working-directory: src/engine/ocaml | |
| run: opam install . --deps-only -y | |
| - name: Build engine | |
| # Mirrors ci.yml::build — `dune build` produces the binary at | |
| # src/engine/ocaml/_build/default/bin/main.exe. | |
| # | |
| # Earlier (post-#36-merge) the step here was `opam install . -y` | |
| # to put `coh` on PATH. That step failed on the first main run | |
| # with exit 31. Root cause: opam's package-mode build (`dune | |
| # build -p name @install`) treats the extracted package source | |
| # as its root, but the bin/dune `build_version.ml` rule depends | |
| # on `../../../VERSION` which resolves to outside-the-package | |
| # in opam's build sandbox. Switching to bare `dune build` in | |
| # the engine working directory avoids the package-mode | |
| # constraint; the kata loop invokes the binary by direct path. | |
| working-directory: src/engine/ocaml | |
| run: opam exec -- dune build | |
| # AC2 — auto-discovery. The glob `katas/*/` matches *directories only*, | |
| # so `katas/README.md` is skipped naturally. Each iteration uses the | |
| # directory basename as the kata id; this matches the kata.toml `id` | |
| # field by convention (id == directory basename, asserted in | |
| # katas/README.md §Directory layout). No kata names are hard-coded | |
| # here — Phase 2 katas (#34) auto-attach when their directories land. | |
| # | |
| # Binary invocation: direct path to the dune build output rather | |
| # than relying on `coh` being on PATH. Matches the canonical | |
| # tsc.yml pattern (which uses `dune exec` but resolves the same | |
| # binary). Direct-path is slightly less portable but avoids any | |
| # dependency on opam-install vs dune-install state. | |
| # | |
| # Per-kata result JSON capture (cycle #38 AC1): the engine's | |
| # `run_kata` (src/engine/ocaml/bin/main.ml:516) emits its result JSON | |
| # to *stdout* (Printf.printf), not via `--output`. The CLI | |
| # `--output` flag exists but is only wired for --target / --files | |
| # modes (main.ml:299/358/365/371/426). For kata mode we therefore | |
| # capture stdout with `tee`. If/when the engine wires `--output` | |
| # for kata mode, this loop can be simplified to pass the flag | |
| # directly — see cycle #38 alpha-closeout for the follow-on issue. | |
| # | |
| # `set -e` is disabled inside the loop so a single failing kata | |
| # does not short-circuit before the JSON is captured for the | |
| # remaining katas. Aggregate pass/fail is enforced after the loop. | |
| - name: Run all katas (auto-discovered) | |
| run: | | |
| set +e | |
| shopt -s nullglob | |
| COH="src/engine/ocaml/_build/default/bin/main.exe" | |
| if [ ! -x "$COH" ]; then | |
| echo "::error::engine binary not found at $COH — Build engine step probably failed" | |
| exit 1 | |
| fi | |
| mkdir -p .kata-results | |
| ran=0 | |
| failed=0 | |
| for kata_dir in katas/*/; do | |
| [ -f "${kata_dir}kata.toml" ] || continue | |
| id=$(basename "$kata_dir") | |
| echo "::group::kata $id" | |
| # Capture stdout JSON to .kata-results/<id>.json; stderr stays | |
| # in the step log under the ::group:: markers. Exit code of | |
| # the engine (not tee) determines pass/fail via PIPESTATUS. | |
| "$COH" --kata "$id" --mode mechanical | tee ".kata-results/${id}.json" | |
| rc=${PIPESTATUS[0]} | |
| if [ "$rc" -eq 0 ]; then | |
| echo "kata $id: PASS" | |
| ran=$((ran + 1)) | |
| else | |
| echo "::error::kata $id: FAIL (exit $rc)" | |
| failed=$((failed + 1)) | |
| ran=$((ran + 1)) | |
| fi | |
| echo "::endgroup::" | |
| done | |
| echo "katas: $ran ran, $failed failed" | |
| if [ "$ran" -eq 0 ]; then | |
| echo "::error::no katas discovered under katas/*/ — expected at least kata-01" | |
| exit 1 | |
| fi | |
| if [ "$failed" -gt 0 ]; then | |
| exit 1 | |
| fi | |
| # AC2 — Emit a markdown table to $GITHUB_STEP_SUMMARY so the PR | |
| # check UI shows per-kata Verdict / C_Σ / Range / Status inline | |
| # without click-through to the step log. Row schema must match | |
| # the table in katas/README.md §Where to find kata results (AC5) | |
| # exactly: | Kata | Verdict | C_Σ | Range | Status |. | |
| # | |
| # Implementation: parse each `.kata-results/<id>.json` with python3 | |
| # (preinstalled on ubuntu-22.04 runners; same dependency as | |
| # tsc.yml::Display results). `kata_pass` from the JSON drives the | |
| # Status emoji; `expected_verdict`, `c_sigma`, `score_range.min`, | |
| # `score_range.max` drive the rest. Empty / unparseable JSON files | |
| # render a single explicit "no result" row rather than failing the | |
| # step (the kata-failure exit was already reported above). | |
| - name: Emit kata step-summary table | |
| if: always() | |
| run: | | |
| echo "## Kata Results" >> "$GITHUB_STEP_SUMMARY" | |
| echo "" >> "$GITHUB_STEP_SUMMARY" | |
| shopt -s nullglob | |
| results=( .kata-results/*.json ) | |
| if [ ${#results[@]} -eq 0 ]; then | |
| echo "_No kata results captured (see step log for the kata-run failure)._" >> "$GITHUB_STEP_SUMMARY" | |
| exit 0 | |
| fi | |
| echo "| Kata | Verdict | C_Σ | Range | Status |" >> "$GITHUB_STEP_SUMMARY" | |
| echo "|---|---|---|---|---|" >> "$GITHUB_STEP_SUMMARY" | |
| for f in "${results[@]}"; do | |
| id=$(basename "$f" .json) | |
| row=$(python3 - "$f" "$id" <<'PY' | |
| import json, sys | |
| path, kata_id = sys.argv[1], sys.argv[2] | |
| try: | |
| with open(path) as fh: | |
| data = json.load(fh) | |
| except Exception as exc: | |
| print(f"| {kata_id} | — | — | — | :warning: unparseable JSON ({exc.__class__.__name__}) |") | |
| sys.exit(0) | |
| verdict = data.get("expected_verdict", "—") | |
| c_sigma = data.get("c_sigma") | |
| rng = data.get("score_range") or {} | |
| mn, mx = rng.get("min"), rng.get("max") | |
| kata_pass = data.get("kata_pass") | |
| c_str = f"{c_sigma:.2f}" if isinstance(c_sigma, (int, float)) else "—" | |
| rng_str = (f"{mn:.2f}–{mx:.2f}" | |
| if isinstance(mn, (int, float)) and isinstance(mx, (int, float)) | |
| else "—") | |
| if kata_pass is True: | |
| status = ":white_check_mark:" | |
| elif kata_pass is False: | |
| status = ":x:" | |
| else: | |
| status = "—" | |
| print(f"| {kata_id} | {verdict} | {c_str} | {rng_str} | {status} |") | |
| PY | |
| ) | |
| echo "$row" >> "$GITHUB_STEP_SUMMARY" | |
| done | |
| # AC1 — Upload per-kata result JSON as an artifact, 90-day retention. | |
| # `if: always()` so failed runs (the very case where post-mortem | |
| # JSON is most valuable) still upload. Naming follows tsc.yml's | |
| # `tsc-reports-${{ github.sha }}` convention. | |
| - name: Upload kata results | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: kata-results-${{ github.sha }} | |
| path: .kata-results/ | |
| retention-days: 90 | |
| if-no-files-found: ignore | |
| # Cycle #38 AC3 + AC4 — Published-binary kata validation. | |
| # | |
| # Purpose: confirm that the binary we *ship* to users (the | |
| # `coh-linux-x64` artifact attached to the latest GitHub Release by | |
| # release.yml line 41/48 via softprops/action-gh-release@v2) still | |
| # passes every kata under today's runner / dep / infra conditions. | |
| # This is distinct from `run-katas` above — that job rebuilds from | |
| # HEAD on every PR and so only catches regressions introduced by | |
| # code change. Infra drift (libcurl point release, runner image | |
| # bump, opam-repository state) only surfaces when the *same* source | |
| # is rebuilt under newer infra, or when the released binary is | |
| # re-exercised. AC3 picks the latter — re-exercise the released | |
| # binary directly. | |
| # | |
| # AC4 — mechanism: Path B (download `coh-linux-x64` from the latest | |
| # GitHub Release). Justification (also recorded in alpha-closeout): | |
| # | |
| # 1. Release.yml at main `8e3094c` already publishes `coh-linux-x64` | |
| # reliably for every `v*` tag — no missing-artifact risk. | |
| # 2. Path B exercises the *exact bytes users download*, which is | |
| # the only mechanism that can catch source-vs-artifact drift | |
| # (the whole motivation for AC3 per issue #38 §Problem). Path A | |
| # (build-from-tag) would silently re-run release.yml's build | |
| # under whatever opam state the validation runner happens to | |
| # pick up — that's a *different* binary than the one users have. | |
| # 3. Path B is cheaper (no opam install + dune build — ~30s vs | |
| # ~3 min) so the weekly cron stays well under the runner-minute | |
| # budget even as the kata count grows. | |
| # | |
| # Failure handling: AC3 is *notify-only* for v1 (per §Open question 5 | |
| # recommendation). A non-zero kata exit fails this job, the workflow | |
| # is visible-red on the Actions tab, but no release is blocked and | |
| # no PR is gated. If reproducible drift is observed in practice, | |
| # escalation to release-blocking is a follow-on cycle. | |
| # | |
| # This job is interim per the workflow's INTERIM header — it will | |
| # be replaced by canonical cnos #344 Cycle B templates once those | |
| # land. | |
| validate-published-binary: | |
| name: validate-published-binary (latest release) | |
| # Only run on triggers that imply "we want to validate the published | |
| # binary" — release publication, weekly cron, or manual dispatch. | |
| # Pre-merge push/PR runs skip this job (their gate is `run-katas`). | |
| if: github.event_name == 'release' || github.event_name == 'schedule' || github.event_name == 'workflow_dispatch' | |
| runs-on: ubuntu-22.04 | |
| permissions: | |
| # `gh release download` needs contents:read; default GITHUB_TOKEN | |
| # has it on public repos but we declare it explicitly to keep | |
| # the surface auditable. | |
| contents: read | |
| steps: | |
| - uses: actions/checkout@v4 | |
| # Path B — download the latest release's coh-linux-x64 asset. | |
| # We pin to the release that *triggered* the run when available | |
| # (release event provides `github.event.release.tag_name`) so a | |
| # release event validates *that* release. For cron + manual | |
| # dispatch, fall back to "latest" so the weekly check always | |
| # exercises the most recently-published version. | |
| - name: Download published coh-linux-x64 | |
| env: | |
| GH_TOKEN: ${{ secrets.GITHUB_TOKEN }} | |
| run: | | |
| set -euo pipefail | |
| # Resolve which release tag to test. | |
| tag="${{ github.event.release.tag_name || '' }}" | |
| if [ -z "$tag" ]; then | |
| # cron / workflow_dispatch path: resolve "latest" via gh. | |
| tag=$(gh release view --json tagName --jq .tagName) | |
| fi | |
| echo "Validating released binary at tag: $tag" | |
| echo "RELEASE_TAG=$tag" >> "$GITHUB_ENV" | |
| gh release download "$tag" --pattern 'coh-linux-x64' --output coh-linux-x64 | |
| chmod +x coh-linux-x64 | |
| ./coh-linux-x64 --version | |
| # Run the same auto-discovery loop as `run-katas` above, but | |
| # against the downloaded binary. Capture JSON to .kata-results/ | |
| # so the same artifact-upload + step-summary surface applies. | |
| # | |
| # Note: kata directories themselves still come from this checkout | |
| # (i.e. main's view of katas/). That's a deliberate choice — we | |
| # want to test "does the *released* binary still solve today's | |
| # kata corpus", not "does the released binary still solve the | |
| # kata corpus from its own tag" (which would just retest what | |
| # CI already tested at release time). | |
| - name: Run all katas against published binary | |
| run: | | |
| set +e | |
| shopt -s nullglob | |
| COH="./coh-linux-x64" | |
| if [ ! -x "$COH" ]; then | |
| echo "::error::published binary not found at $COH" | |
| exit 1 | |
| fi | |
| mkdir -p .kata-results | |
| ran=0 | |
| failed=0 | |
| for kata_dir in katas/*/; do | |
| [ -f "${kata_dir}kata.toml" ] || continue | |
| id=$(basename "$kata_dir") | |
| echo "::group::kata $id" | |
| "$COH" --kata "$id" --mode mechanical | tee ".kata-results/${id}.json" | |
| rc=${PIPESTATUS[0]} | |
| if [ "$rc" -eq 0 ]; then | |
| echo "kata $id: PASS" | |
| ran=$((ran + 1)) | |
| else | |
| echo "::error::kata $id: FAIL (exit $rc) — against published binary $RELEASE_TAG" | |
| failed=$((failed + 1)) | |
| ran=$((ran + 1)) | |
| fi | |
| echo "::endgroup::" | |
| done | |
| echo "katas: $ran ran, $failed failed (binary=$RELEASE_TAG)" | |
| if [ "$ran" -eq 0 ]; then | |
| echo "::error::no katas discovered under katas/*/ — expected at least kata-01" | |
| exit 1 | |
| fi | |
| if [ "$failed" -gt 0 ]; then | |
| exit 1 | |
| fi | |
| # AC2 — step-summary table, mirrors the run-katas job's format | |
| # exactly so consumers see the same row schema regardless of | |
| # which job produced the artifact. Title differs so the two | |
| # sections are visually distinguishable on a single run page. | |
| - name: Emit kata step-summary table | |
| if: always() | |
| run: | | |
| echo "## Kata Results — Published Binary ($RELEASE_TAG)" >> "$GITHUB_STEP_SUMMARY" | |
| echo "" >> "$GITHUB_STEP_SUMMARY" | |
| shopt -s nullglob | |
| results=( .kata-results/*.json ) | |
| if [ ${#results[@]} -eq 0 ]; then | |
| echo "_No kata results captured (see step log for the kata-run failure)._" >> "$GITHUB_STEP_SUMMARY" | |
| exit 0 | |
| fi | |
| echo "| Kata | Verdict | C_Σ | Range | Status |" >> "$GITHUB_STEP_SUMMARY" | |
| echo "|---|---|---|---|---|" >> "$GITHUB_STEP_SUMMARY" | |
| for f in "${results[@]}"; do | |
| id=$(basename "$f" .json) | |
| row=$(python3 - "$f" "$id" <<'PY' | |
| import json, sys | |
| path, kata_id = sys.argv[1], sys.argv[2] | |
| try: | |
| with open(path) as fh: | |
| data = json.load(fh) | |
| except Exception as exc: | |
| print(f"| {kata_id} | — | — | — | :warning: unparseable JSON ({exc.__class__.__name__}) |") | |
| sys.exit(0) | |
| verdict = data.get("expected_verdict", "—") | |
| c_sigma = data.get("c_sigma") | |
| rng = data.get("score_range") or {} | |
| mn, mx = rng.get("min"), rng.get("max") | |
| kata_pass = data.get("kata_pass") | |
| c_str = f"{c_sigma:.2f}" if isinstance(c_sigma, (int, float)) else "—" | |
| rng_str = (f"{mn:.2f}–{mx:.2f}" | |
| if isinstance(mn, (int, float)) and isinstance(mx, (int, float)) | |
| else "—") | |
| if kata_pass is True: | |
| status = ":white_check_mark:" | |
| elif kata_pass is False: | |
| status = ":x:" | |
| else: | |
| status = "—" | |
| print(f"| {kata_id} | {verdict} | {c_str} | {rng_str} | {status} |") | |
| PY | |
| ) | |
| echo "$row" >> "$GITHUB_STEP_SUMMARY" | |
| done | |
| # Artifact name embeds the release tag (not the SHA) so historical | |
| # artifacts are findable by version — the principal axis a human | |
| # would search when investigating "did v0.8.0 ever fail kata-02 | |
| # under cron?". | |
| - name: Upload kata results | |
| if: always() | |
| uses: actions/upload-artifact@v4 | |
| with: | |
| name: kata-results-published-${{ env.RELEASE_TAG }}-${{ github.run_id }} | |
| path: .kata-results/ | |
| retention-days: 90 | |
| if-no-files-found: ignore |