Skip to content

katas

katas #155

Workflow file for this run

# katas.yml — engine kata regression gate
#
# INTERIM workflow (tsc cycles #36 + #38).
# This file is the v1 local implementation of the katas-in-CI gate.
# It will be replaced by canonical templates landed by:
# - cnos #344 Cycle B (template authoring)
# - tsc cycle C-2 (template adoption)
# Until then, this is the source of truth for kata CI. Both the
# build-from-HEAD pre-merge gate (`run-katas`) AND the published-binary
# validation job (`validate-published-binary`, added cycle #38 AC3+AC4)
# will be replaced by canonical cnos #344 Cycle B templates once those
# land.
#
# Surface contract:
# - Auto-discovers every directory under katas/ (no hard-coded kata names).
# - Invokes `coh --kata <id> --mode mechanical` for each kata.
# - Fails the job on any non-zero exit from the kata runner.
# - Caches OPAM + dune _build keyed on src/engine/ocaml/dune-project +
# src/engine/ocaml/tsc_engine.opam so dep / build-config changes
# invalidate cleanly.
# - Persists per-kata result JSON to `.kata-results/<id>.json` and
# uploads as an artifact (cycle #38 AC1); emits a markdown summary
# table to $GITHUB_STEP_SUMMARY (cycle #38 AC2).
#
# Consolidation note: this workflow replaces the previous `kata-check`
# job that lived in ci.yml (removed in cycle #36 R2). That job ran the
# same `bash scripts/run-katas.sh` regression check on every push +
# PR but had no OPAM/dune build cache (every run cold ~5–8 min) and
# no concurrency control. Cycle #36 consolidates kata-running into
# this dedicated workflow.
name: katas
on:
push:
branches: [main]
pull_request:
# Cycle #38 AC3 — published-binary validation triggers.
# `release: published` fires once per GitHub Release publication
# (covers the v* tag-push flow handled by release.yml). The weekly
# cron catches infra drift (libcurl point release, runner image
# change, opam-repository state) independent of code-change cadence
# — issue #38 §Open question 2 recommendation: weekly Mon 06:00 UTC.
# `workflow_dispatch` lets operators trigger the published-binary
# job manually (e.g. after suspected infra regressions) without
# waiting for the next cron tick.
release:
types: [published]
schedule:
- cron: '0 6 * * 1'
workflow_dispatch:
# Per issue #36 open question 2: cancel superseded PR runs, never cancel main.
# Per cycle #38: release / schedule / workflow_dispatch runs target the
# `validate-published-binary` job — they share this top-level concurrency
# group with the pre-merge `run-katas` job, but in practice never collide
# because their trigger sources never produce overlapping refs with PR /
# push-to-main. Splitting concurrency per-job is unnecessary for v1.
concurrency:
group: katas-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
jobs:
run-katas:
name: run-katas (auto-discovered)
runs-on: ubuntu-22.04
steps:
- uses: actions/checkout@v4
- name: Install system depexts (libcurl for ezcurl)
# Pre-install libcurl4-openssl-dev so opam-depext / setup-ocaml
# does not need to resolve the (occasionally 404ing) gnutls flavour.
# Mirrors ci.yml build job; see cycle #26 β observation, #32 AC5.
run: |
sudo apt-get update
sudo apt-get install -y --no-install-recommends libcurl4-openssl-dev pkg-config
# AC3 — Build cache. Two-layer cache keyed on the engine's dune
# build/dep manifests so any dep or build-config change invalidates
# cleanly. OS dimension included so cache is not shared across
# runner images. Both `dune-project` (generate_opam_files true)
# and `tsc_engine.opam` (its generated output) move together but
# we hash both because either being edited directly is in scope.
- name: Cache OPAM + dune
id: cache-opam-dune
uses: actions/cache@v4
with:
path: |
~/.opam
src/engine/ocaml/_build
key: katas-${{ runner.os }}-ocaml-5.2-${{ hashFiles('src/engine/ocaml/dune-project', 'src/engine/ocaml/tsc_engine.opam') }}
restore-keys: |
katas-${{ runner.os }}-ocaml-5.2-
- name: Set up OCaml
uses: ocaml/setup-ocaml@v3
with:
ocaml-compiler: "5.2"
- name: Install dependencies
working-directory: src/engine/ocaml
run: opam install . --deps-only -y
- name: Build engine
# Mirrors ci.yml::build — `dune build` produces the binary at
# src/engine/ocaml/_build/default/bin/main.exe.
#
# Earlier (post-#36-merge) the step here was `opam install . -y`
# to put `coh` on PATH. That step failed on the first main run
# with exit 31. Root cause: opam's package-mode build (`dune
# build -p name @install`) treats the extracted package source
# as its root, but the bin/dune `build_version.ml` rule depends
# on `../../../VERSION` which resolves to outside-the-package
# in opam's build sandbox. Switching to bare `dune build` in
# the engine working directory avoids the package-mode
# constraint; the kata loop invokes the binary by direct path.
working-directory: src/engine/ocaml
run: opam exec -- dune build
# AC2 — auto-discovery. The glob `katas/*/` matches *directories only*,
# so `katas/README.md` is skipped naturally. Each iteration uses the
# directory basename as the kata id; this matches the kata.toml `id`
# field by convention (id == directory basename, asserted in
# katas/README.md §Directory layout). No kata names are hard-coded
# here — Phase 2 katas (#34) auto-attach when their directories land.
#
# Binary invocation: direct path to the dune build output rather
# than relying on `coh` being on PATH. Matches the canonical
# tsc.yml pattern (which uses `dune exec` but resolves the same
# binary). Direct-path is slightly less portable but avoids any
# dependency on opam-install vs dune-install state.
#
# Per-kata result JSON capture (cycle #38 AC1): the engine's
# `run_kata` (src/engine/ocaml/bin/main.ml:516) emits its result JSON
# to *stdout* (Printf.printf), not via `--output`. The CLI
# `--output` flag exists but is only wired for --target / --files
# modes (main.ml:299/358/365/371/426). For kata mode we therefore
# capture stdout with `tee`. If/when the engine wires `--output`
# for kata mode, this loop can be simplified to pass the flag
# directly — see cycle #38 alpha-closeout for the follow-on issue.
#
# `set -e` is disabled inside the loop so a single failing kata
# does not short-circuit before the JSON is captured for the
# remaining katas. Aggregate pass/fail is enforced after the loop.
- name: Run all katas (auto-discovered)
run: |
set +e
shopt -s nullglob
COH="src/engine/ocaml/_build/default/bin/main.exe"
if [ ! -x "$COH" ]; then
echo "::error::engine binary not found at $COH — Build engine step probably failed"
exit 1
fi
mkdir -p .kata-results
ran=0
failed=0
for kata_dir in katas/*/; do
[ -f "${kata_dir}kata.toml" ] || continue
id=$(basename "$kata_dir")
echo "::group::kata $id"
# Capture stdout JSON to .kata-results/<id>.json; stderr stays
# in the step log under the ::group:: markers. Exit code of
# the engine (not tee) determines pass/fail via PIPESTATUS.
"$COH" --kata "$id" --mode mechanical | tee ".kata-results/${id}.json"
rc=${PIPESTATUS[0]}
if [ "$rc" -eq 0 ]; then
echo "kata $id: PASS"
ran=$((ran + 1))
else
echo "::error::kata $id: FAIL (exit $rc)"
failed=$((failed + 1))
ran=$((ran + 1))
fi
echo "::endgroup::"
done
echo "katas: $ran ran, $failed failed"
if [ "$ran" -eq 0 ]; then
echo "::error::no katas discovered under katas/*/ — expected at least kata-01"
exit 1
fi
if [ "$failed" -gt 0 ]; then
exit 1
fi
# AC2 — Emit a markdown table to $GITHUB_STEP_SUMMARY so the PR
# check UI shows per-kata Verdict / C_Σ / Range / Status inline
# without click-through to the step log. Row schema must match
# the table in katas/README.md §Where to find kata results (AC5)
# exactly: | Kata | Verdict | C_Σ | Range | Status |.
#
# Implementation: parse each `.kata-results/<id>.json` with python3
# (preinstalled on ubuntu-22.04 runners; same dependency as
# tsc.yml::Display results). `kata_pass` from the JSON drives the
# Status emoji; `expected_verdict`, `c_sigma`, `score_range.min`,
# `score_range.max` drive the rest. Empty / unparseable JSON files
# render a single explicit "no result" row rather than failing the
# step (the kata-failure exit was already reported above).
- name: Emit kata step-summary table
if: always()
run: |
echo "## Kata Results" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
shopt -s nullglob
results=( .kata-results/*.json )
if [ ${#results[@]} -eq 0 ]; then
echo "_No kata results captured (see step log for the kata-run failure)._" >> "$GITHUB_STEP_SUMMARY"
exit 0
fi
echo "| Kata | Verdict | C_Σ | Range | Status |" >> "$GITHUB_STEP_SUMMARY"
echo "|---|---|---|---|---|" >> "$GITHUB_STEP_SUMMARY"
for f in "${results[@]}"; do
id=$(basename "$f" .json)
row=$(python3 - "$f" "$id" <<'PY'
import json, sys
path, kata_id = sys.argv[1], sys.argv[2]
try:
with open(path) as fh:
data = json.load(fh)
except Exception as exc:
print(f"| {kata_id} | — | — | — | :warning: unparseable JSON ({exc.__class__.__name__}) |")
sys.exit(0)
verdict = data.get("expected_verdict", "—")
c_sigma = data.get("c_sigma")
rng = data.get("score_range") or {}
mn, mx = rng.get("min"), rng.get("max")
kata_pass = data.get("kata_pass")
c_str = f"{c_sigma:.2f}" if isinstance(c_sigma, (int, float)) else "—"
rng_str = (f"{mn:.2f}–{mx:.2f}"
if isinstance(mn, (int, float)) and isinstance(mx, (int, float))
else "—")
if kata_pass is True:
status = ":white_check_mark:"
elif kata_pass is False:
status = ":x:"
else:
status = "—"
print(f"| {kata_id} | {verdict} | {c_str} | {rng_str} | {status} |")
PY
)
echo "$row" >> "$GITHUB_STEP_SUMMARY"
done
# AC1 — Upload per-kata result JSON as an artifact, 90-day retention.
# `if: always()` so failed runs (the very case where post-mortem
# JSON is most valuable) still upload. Naming follows tsc.yml's
# `tsc-reports-${{ github.sha }}` convention.
- name: Upload kata results
if: always()
uses: actions/upload-artifact@v4
with:
name: kata-results-${{ github.sha }}
path: .kata-results/
retention-days: 90
if-no-files-found: ignore
# Cycle #38 AC3 + AC4 — Published-binary kata validation.
#
# Purpose: confirm that the binary we *ship* to users (the
# `coh-linux-x64` artifact attached to the latest GitHub Release by
# release.yml line 41/48 via softprops/action-gh-release@v2) still
# passes every kata under today's runner / dep / infra conditions.
# This is distinct from `run-katas` above — that job rebuilds from
# HEAD on every PR and so only catches regressions introduced by
# code change. Infra drift (libcurl point release, runner image
# bump, opam-repository state) only surfaces when the *same* source
# is rebuilt under newer infra, or when the released binary is
# re-exercised. AC3 picks the latter — re-exercise the released
# binary directly.
#
# AC4 — mechanism: Path B (download `coh-linux-x64` from the latest
# GitHub Release). Justification (also recorded in alpha-closeout):
#
# 1. Release.yml at main `8e3094c` already publishes `coh-linux-x64`
# reliably for every `v*` tag — no missing-artifact risk.
# 2. Path B exercises the *exact bytes users download*, which is
# the only mechanism that can catch source-vs-artifact drift
# (the whole motivation for AC3 per issue #38 §Problem). Path A
# (build-from-tag) would silently re-run release.yml's build
# under whatever opam state the validation runner happens to
# pick up — that's a *different* binary than the one users have.
# 3. Path B is cheaper (no opam install + dune build — ~30s vs
# ~3 min) so the weekly cron stays well under the runner-minute
# budget even as the kata count grows.
#
# Failure handling: AC3 is *notify-only* for v1 (per §Open question 5
# recommendation). A non-zero kata exit fails this job, the workflow
# is visible-red on the Actions tab, but no release is blocked and
# no PR is gated. If reproducible drift is observed in practice,
# escalation to release-blocking is a follow-on cycle.
#
# This job is interim per the workflow's INTERIM header — it will
# be replaced by canonical cnos #344 Cycle B templates once those
# land.
validate-published-binary:
name: validate-published-binary (latest release)
# Only run on triggers that imply "we want to validate the published
# binary" — release publication, weekly cron, or manual dispatch.
# Pre-merge push/PR runs skip this job (their gate is `run-katas`).
if: github.event_name == 'release' || github.event_name == 'schedule' || github.event_name == 'workflow_dispatch'
runs-on: ubuntu-22.04
permissions:
# `gh release download` needs contents:read; default GITHUB_TOKEN
# has it on public repos but we declare it explicitly to keep
# the surface auditable.
contents: read
steps:
- uses: actions/checkout@v4
# Path B — download the latest release's coh-linux-x64 asset.
# We pin to the release that *triggered* the run when available
# (release event provides `github.event.release.tag_name`) so a
# release event validates *that* release. For cron + manual
# dispatch, fall back to "latest" so the weekly check always
# exercises the most recently-published version.
- name: Download published coh-linux-x64
env:
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set -euo pipefail
# Resolve which release tag to test.
tag="${{ github.event.release.tag_name || '' }}"
if [ -z "$tag" ]; then
# cron / workflow_dispatch path: resolve "latest" via gh.
tag=$(gh release view --json tagName --jq .tagName)
fi
echo "Validating released binary at tag: $tag"
echo "RELEASE_TAG=$tag" >> "$GITHUB_ENV"
gh release download "$tag" --pattern 'coh-linux-x64' --output coh-linux-x64
chmod +x coh-linux-x64
./coh-linux-x64 --version
# Run the same auto-discovery loop as `run-katas` above, but
# against the downloaded binary. Capture JSON to .kata-results/
# so the same artifact-upload + step-summary surface applies.
#
# Note: kata directories themselves still come from this checkout
# (i.e. main's view of katas/). That's a deliberate choice — we
# want to test "does the *released* binary still solve today's
# kata corpus", not "does the released binary still solve the
# kata corpus from its own tag" (which would just retest what
# CI already tested at release time).
- name: Run all katas against published binary
run: |
set +e
shopt -s nullglob
COH="./coh-linux-x64"
if [ ! -x "$COH" ]; then
echo "::error::published binary not found at $COH"
exit 1
fi
mkdir -p .kata-results
ran=0
failed=0
for kata_dir in katas/*/; do
[ -f "${kata_dir}kata.toml" ] || continue
id=$(basename "$kata_dir")
echo "::group::kata $id"
"$COH" --kata "$id" --mode mechanical | tee ".kata-results/${id}.json"
rc=${PIPESTATUS[0]}
if [ "$rc" -eq 0 ]; then
echo "kata $id: PASS"
ran=$((ran + 1))
else
echo "::error::kata $id: FAIL (exit $rc) — against published binary $RELEASE_TAG"
failed=$((failed + 1))
ran=$((ran + 1))
fi
echo "::endgroup::"
done
echo "katas: $ran ran, $failed failed (binary=$RELEASE_TAG)"
if [ "$ran" -eq 0 ]; then
echo "::error::no katas discovered under katas/*/ — expected at least kata-01"
exit 1
fi
if [ "$failed" -gt 0 ]; then
exit 1
fi
# AC2 — step-summary table, mirrors the run-katas job's format
# exactly so consumers see the same row schema regardless of
# which job produced the artifact. Title differs so the two
# sections are visually distinguishable on a single run page.
- name: Emit kata step-summary table
if: always()
run: |
echo "## Kata Results — Published Binary ($RELEASE_TAG)" >> "$GITHUB_STEP_SUMMARY"
echo "" >> "$GITHUB_STEP_SUMMARY"
shopt -s nullglob
results=( .kata-results/*.json )
if [ ${#results[@]} -eq 0 ]; then
echo "_No kata results captured (see step log for the kata-run failure)._" >> "$GITHUB_STEP_SUMMARY"
exit 0
fi
echo "| Kata | Verdict | C_Σ | Range | Status |" >> "$GITHUB_STEP_SUMMARY"
echo "|---|---|---|---|---|" >> "$GITHUB_STEP_SUMMARY"
for f in "${results[@]}"; do
id=$(basename "$f" .json)
row=$(python3 - "$f" "$id" <<'PY'
import json, sys
path, kata_id = sys.argv[1], sys.argv[2]
try:
with open(path) as fh:
data = json.load(fh)
except Exception as exc:
print(f"| {kata_id} | — | — | — | :warning: unparseable JSON ({exc.__class__.__name__}) |")
sys.exit(0)
verdict = data.get("expected_verdict", "—")
c_sigma = data.get("c_sigma")
rng = data.get("score_range") or {}
mn, mx = rng.get("min"), rng.get("max")
kata_pass = data.get("kata_pass")
c_str = f"{c_sigma:.2f}" if isinstance(c_sigma, (int, float)) else "—"
rng_str = (f"{mn:.2f}–{mx:.2f}"
if isinstance(mn, (int, float)) and isinstance(mx, (int, float))
else "—")
if kata_pass is True:
status = ":white_check_mark:"
elif kata_pass is False:
status = ":x:"
else:
status = "—"
print(f"| {kata_id} | {verdict} | {c_str} | {rng_str} | {status} |")
PY
)
echo "$row" >> "$GITHUB_STEP_SUMMARY"
done
# Artifact name embeds the release tag (not the SHA) so historical
# artifacts are findable by version — the principal axis a human
# would search when investigating "did v0.8.0 ever fail kata-02
# under cron?".
- name: Upload kata results
if: always()
uses: actions/upload-artifact@v4
with:
name: kata-results-published-${{ env.RELEASE_TAG }}-${{ github.run_id }}
path: .kata-results/
retention-days: 90
if-no-files-found: ignore