-
Notifications
You must be signed in to change notification settings - Fork 6
773 lines (710 loc) · 40.4 KB
/
Copy pathtest.yml
File metadata and controls
773 lines (710 loc) · 40.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
name: Test
permissions:
contents: read
# The cheap gates (lint, cross-compile matrix, unit tests) run for BOTH
# develop and main — develop is where the actual work merges, and guarding
# only main left every develop PR untested. The heavy e2e lifecycle job stays
# main-bound (see its `if`), so develop traffic doesn't monopolize the
# self-hosted runners.
on:
pull_request:
branches: [main, develop]
push:
branches: [main, develop]
workflow_dispatch:
concurrency:
group: test-${{ github.head_ref || github.ref }}
# A superseded PR run is dead weight on the self-hosted runners — cancel it.
# Runs on main/develop pushes still queue (never cancel a branch-tip run).
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
# =============================================================================
# JOBS
# =============================================================================
jobs:
lint:
name: Lint
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- name: Checkout code
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- name: Set up Go
uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6.5.0
with:
go-version-file: 'go.mod'
cache: true
- name: golangci-lint
# Blocking gate: the audit backlog is cleared (`golangci-lint run ./...`
# reports 0 issues), so any NEW finding fails the pipeline.
uses: golangci/golangci-lint-action@ba0d7d2ec06a0ea1cb5fa41b2e4a3ab91d21278a # v9.3.0
with:
version: latest
- name: 'Check: go.mod/go.sum are tidy'
run: |
echo "==================================================================="
echo "=== TEST: go.mod/go.sum are tidy (go mod tidy -diff)"
echo "==================================================================="
go mod tidy -diff
release-build-matrix:
# Compile-only guard for the FULL GoReleaser platform matrix. The three
# e2e runner legs already compile linux/amd64, windows/amd64 and
# darwin/arm64 (including their test files via `make test-unit`), but the
# release ships SIX combos — a GOOS/GOARCH break on the un-runnered three
# (linux/arm64, windows/arm64, darwin/amd64) would otherwise surface only
# when the release itself fails. For those three, `go vet` also
# cross-compiles the TEST files, catching windows/darwin-only breakage in
# test code that no runner ever builds.
name: Release build matrix (compile-only)
runs-on: ubuntu-latest
timeout-minutes: 15
steps:
- name: Checkout code
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- name: Set up Go
uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6.5.0
with:
go-version-file: 'go.mod'
cache: true
- name: 'Build: all six GoReleaser platform combos'
run: |
echo "==================================================================="
echo "=== TEST: cross-compile the full release matrix (6 GOOS/GOARCH)"
echo "==================================================================="
# Keep in sync with .goreleaser.yml (goos: linux,windows,darwin ×
# goarch: amd64,arm64).
for target in linux/amd64 linux/arm64 windows/amd64 windows/arm64 darwin/amd64 darwin/arm64; do
echo "--- go build ${target}"
GOOS="${target%/*}" GOARCH="${target#*/}" CGO_ENABLED=0 go build ./...
done
- name: 'Vet (incl. test files) for combos without a runner'
run: |
echo "==================================================================="
echo "=== TEST: go vet for the un-runnered combos (compiles test code)"
echo "==================================================================="
# The runner legs vet-compile their own combos; these three are never
# built anywhere else.
for target in linux/arm64 windows/arm64 darwin/amd64; do
echo "--- go vet ${target}"
GOOS="${target%/*}" GOARCH="${target#*/}" go vet ./...
done
# Unit tests as their own fast job: they used to be the LAST steps of the
# e2e job below, which meant (a) a unit-test failure surfaced only after the
# ~50-minute cluster lifecycle, and (b) an e2e flake earlier in the job
# skipped them entirely, masking their result. Same three OS legs, no
# Docker/cluster needed.
unit:
name: Unit tests on ${{ matrix.os }}-${{ matrix.arch }}
runs-on: ${{ matrix.runner }}
timeout-minutes: 25
strategy:
fail-fast: false
matrix:
include:
- os: linux
arch: amd64
runner: Linux-x64
- os: windows
arch: amd64
runner: Windows-x64
- os: darwin
arch: arm64
runner: macos-26
steps:
- name: Checkout code
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- name: Set up Go
uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6.5.0
with:
go-version-file: 'go.mod'
cache: true
- name: Run unit tests
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: unit tests (root, cmd, internal, tests/testutil, tests/scripts)"
echo "==================================================================="
make test-unit
- name: Run unit tests with race detector
# Linux only (race detector needs CGO/gcc). A blocking gate: the
# codebase is race-clean (the pterm spinner was replaced with a
# synchronized wrapper in internal/shared/ui/spinner).
if: matrix.os == 'linux'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: unit tests with -race (linux only)"
echo "==================================================================="
make test-race
test-cli:
name: Test CLI on ${{ matrix.os }}-${{ matrix.arch }}
runs-on: ${{ matrix.runner }}
timeout-minutes: 70
# The full e2e lifecycle (real k3d cluster, 20-minute ArgoCD install) is
# main-bound: it guards what ships, while develop PRs get the cheap gates
# above. Lint gates it — never burn an hour of self-hosted runner time on
# code that fails a 3-minute lint. The fork guard keeps pull_request code
# off the self-hosted runners unless the PR comes from this repo itself.
needs: [lint]
if: >-
github.event_name == 'workflow_dispatch' ||
(github.event_name == 'push' && github.ref == 'refs/heads/main') ||
(github.event_name == 'pull_request' && github.base_ref == 'main' &&
github.event.pull_request.head.repo.full_name == github.repository)
env:
# Path to the built binary for this matrix leg. On Windows it forwards the
# whole CLI into WSL; on Linux it runs natively. Used by every test step.
OF_BIN: ./build/openframe-${{ matrix.os }}-${{ matrix.arch }}${{ matrix.os == 'windows' && '.exe' || '' }}
# Cluster used by the lifecycle steps; its k3d kube-context is k3d-<name>.
OF_CLUSTER: openframe-test
OF_CONTEXT: k3d-openframe-test
strategy:
fail-fast: false
matrix:
include:
- os: linux
arch: amd64
runner: Linux-x64
- os: windows
arch: amd64
runner: Windows-x64
- os: darwin
arch: arm64
runner: macos-26
steps:
- name: Checkout code
uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
with:
persist-credentials: false
- name: Set up Go
uses: actions/setup-go@924ae3a1cded613372ab5595356fb5720e22ba16 # v6.5.0
with:
go-version-file: 'go.mod'
cache: true
# The Windows binary re-runs the whole CLI inside WSL2, so the runner needs
# a WSL distro that can run Docker (k3d needs a Docker daemon). Alpine is the
# lightest official distro; the CLI's Docker installer now supports it (apk).
# Only `bash` is pre-installed (the launcher uses `bash -lc`) — Docker itself
# is installed by the CLI (`prerequisites install`) below, exercising the new
# apk path; the daemon is then started explicitly (WSL has no OpenRC init).
- name: Install Alpine in WSL
if: matrix.os == 'windows'
uses: Vampire/setup-wsl@d1da7f2c0322a5ee4f24975344f67fc0f5baf364 # v7.0.0
with:
distribution: Alpine-3.23
set-as-default: 'true'
additional-packages: bash
# --- Build & stage (all platforms, incl. darwin) -----------------------
- name: Build & stage CLI
shell: bash
run: |
echo "==================================================================="
echo "=== SETUP: build CLI for ${{ matrix.os }}/${{ matrix.arch }}"
echo "==================================================================="
make build
if [ "${{ matrix.os }}" = "windows" ]; then
# The cluster and the Kubernetes client live in WSL, so the CLI needs a
# *Linux* openframe binary there. `make build` is a dev version with no
# published release to auto-download, so build the Linux binary and
# stream it into the default WSL distro's ~/.openframe/bin via stdin —
# no Windows→WSL path translation, which is where the launcher looks.
GOOS=linux GOARCH=amd64 CGO_ENABLED=0 go build -o build/openframe-linux-amd64 .
wsl -- bash -c 'mkdir -p "$HOME/.openframe/bin" && cat > "$HOME/.openframe/bin/openframe" && chmod +x "$HOME/.openframe/bin/openframe"' < build/openframe-linux-amd64
fi
# --- Simplest: no cluster, read-only (all platforms, incl. darwin) -----
- name: 'CLI: version & help'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: version, help & completion (read-only, no cluster)"
echo "==================================================================="
"$OF_BIN" --version
"$OF_BIN" --help
# Per-command --help is exercised exhaustively in the next step.
echo "--- completion: every supported shell generates cleanly"
# Shell-completion generation is pure (no cluster/network) — smoke every shell.
for sh in bash zsh fish powershell; do "$OF_BIN" completion "$sh" >/dev/null; done
echo "--- update check (text + json; network-tolerant)"
# Self-update check needs neither Docker nor k3d, so it runs everywhere
# (incl. darwin); tolerate transient network issues on the runner.
"$OF_BIN" update check || echo "update check skipped (network)"
# json mode must keep stdout machine-clean (no spinner) and stay parseable.
"$OF_BIN" update check -o json || echo "update check -o json skipped (network)"
echo "--- update rollback with no prior update"
# Rollback with no prior update must exit cleanly with a "nothing to roll
# back" notice (offline, no download), not error out or hang.
"$OF_BIN" update rollback
# Exhaustive --help over the WHOLE command tree, on the real (built/staged)
# binary. Complements the hermetic cmd/help_matrix_test.go by covering the
# actual binary (ldflags, embeds, and on Windows the WSL forward). Discovery
# is dynamic — a newly added command is covered with no hand-maintained list.
# Pure and non-interactive (no cluster, no network), so it runs on every OS.
- name: 'CLI: every command --help (exhaustive)'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: --help on EVERY command (real binary, non-interactive)"
echo "==================================================================="
fail() { echo "::error::$1"; exit 1; }
walk_help() {
local path="$1"
# $path is a space-separated command path ("cluster create") that MUST
# word-split into separate argv entries — intentional, so quiet SC2086.
# shellcheck disable=SC2086
"$OF_BIN" $path --help >/tmp/h.out 2>&1 || { cat /tmp/h.out; fail "'openframe $path --help' exited non-zero"; }
grep -q '^Usage:' /tmp/h.out || { cat /tmp/h.out; fail "'openframe $path --help' printed no Usage section"; }
echo " OK: openframe $path --help"
# Leaf commands list no subcommands, so grep matching nothing is fine.
local subs
subs=$(sed -n '/Available Commands:/,/^$/p' /tmp/h.out | grep -E '^ [a-z]' | awk '{print $1}' || true)
for s in $subs; do
[ "$s" = "help" ] && continue
walk_help "${path:+$path }$s"
done
}
walk_help ""
# Self-update against the REAL latest release: the only test that
# exercises the live trust chain end to end (GitHub release lookup →
# cosign signature of checksums.txt against the pinned identity →
# SHA256 of the archive → atomic swap → --version smoke test → rollback).
# Runs on a COPY of the binary, never the one under test.
#
# Not on Windows: the native launcher forwards the whole CLI into WSL, so
# this would exercise the Linux binary there, not the Windows one.
- name: 'Update: apply the real latest release, then roll back'
if: matrix.os != 'windows'
shell: bash
env:
# GH_TOKEN for the curl below; GITHUB_TOKEN is what `openframe update`
# itself reads. Exporting only GH_TOKEN left the CLI unauthenticated,
# which rate-limited to HTTP 403 on shared macOS runner IPs.
GH_TOKEN: ${{ secrets.GITHUB_TOKEN }}
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
echo "==================================================================="
echo "=== TEST: openframe update (real release, cosign-verified) + rollback"
echo "==================================================================="
LATEST="$(curl -fsSL -H "Authorization: Bearer $GH_TOKEN" \
"https://api.github.com/repos/${GITHUB_REPOSITORY}/releases/latest" | jq -r .tag_name)"
[ -n "$LATEST" ] && [ "$LATEST" != "null" ] || { echo "::error::could not resolve the latest release"; exit 1; }
echo "latest published release: $LATEST"
# A dev build refuses to self-update, so stamp an old version in.
WORK="$(mktemp -d)"; export HOME="$WORK/home"; mkdir -p "$HOME"
go build -ldflags "-X github.com/flamingo-stack/openframe-cli/cmd.version=0.0.1" -o "$WORK/openframe" .
OF="$WORK/openframe"
[ "$("$OF" --version | head -n1 | cut -d' ' -f1)" = "0.0.1" ] || { echo "::error::ldflags version injection broken"; exit 1; }
echo "--- update to the latest release (verifies the cosign bundle)"
"$OF" update --yes
got="$("$OF" --version | head -n1 | cut -d' ' -f1)"
[ "$got" = "$LATEST" ] || { echo "::error::after update --version is $got, want $LATEST"; exit 1; }
echo "updated 0.0.1 -> $got"
echo "--- rollback restores the previous binary (offline)"
"$OF" update rollback --yes
back="$("$OF" --version | head -n1 | cut -d' ' -f1)"
[ "$back" = "0.0.1" ] || { echo "::error::after rollback --version is $back, want 0.0.1"; exit 1; }
echo "--- rollback again: nothing left to restore, clean exit"
"$OF" update rollback --yes
echo "--- explicit tag, both spellings (ForTag tolerates the v prefix)"
for spelling in "$LATEST" "v${LATEST#v}"; do
W2="$(mktemp -d)"; HOME="$W2/home"; mkdir -p "$HOME"
go build -ldflags "-X github.com/flamingo-stack/openframe-cli/cmd.version=0.0.1" -o "$W2/openframe" .
HOME="$W2/home" "$W2/openframe" update "$spelling" --yes
v="$(HOME="$W2/home" "$W2/openframe" --version | head -n1 | cut -d' ' -f1)"
[ "$v" = "$LATEST" ] || { echo "::error::update $spelling landed on $v, want $LATEST"; exit 1; }
echo "OK: update $spelling -> $v"
done
echo "--- an unsigned/nonexistent release is refused"
W3="$(mktemp -d)"; HOME="$W3/home"; mkdir -p "$HOME"
go build -ldflags "-X github.com/flamingo-stack/openframe-cli/cmd.version=0.0.1" -o "$W3/openframe" .
if HOME="$W3/home" "$W3/openframe" update 99.99.99 --yes; then
echo "::error::update to a nonexistent release must fail"; exit 1
fi
echo "update + rollback OK"
# Windows only: run the CLI's own prerequisite installer inside WSL. This
# exercises the apk Docker install path (installAlpine) and installs k3d +
# helm via verified download. The Docker daemon is started separately below
# (installAlpine's OpenRC start is a no-op under WSL's initless environment).
- name: 'Prereqs: install (WSL, exercises apk Docker install)'
if: matrix.os == 'windows'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: prerequisites install inside WSL (apk Docker path)"
echo "==================================================================="
"$OF_BIN" prerequisites install || echo "prereq install returned nonzero (daemon is started next)"
# Start the Docker daemon inside WSL. Alpine WSL has no OpenRC init, so mount
# cgroups and launch dockerd directly (vfs storage driver avoids overlayfs
# issues under the WSL2 kernel). Dump the log if it never comes up.
- name: 'Start Docker in WSL'
if: matrix.os == 'windows'
shell: bash
run: |
echo "==================================================================="
echo "=== SETUP: start dockerd inside WSL (Alpine, initless)"
echo "==================================================================="
# rc-service is unreliable in initless WSL (it returns 0 without actually
# bringing up the daemon), so start dockerd directly and detached
# (setsid, </dev/null) so it survives this shell and keeps the WSL VM and
# daemon alive for later steps. vfs avoids overlayfs issues under WSL2.
wsl -u root -- sh -c '
command -v dockerd >/dev/null || { echo "dockerd not installed"; exit 1; }
rc-service docker stop >/dev/null 2>&1 || true
pkill dockerd >/dev/null 2>&1 || true
mountpoint -q /sys/fs/cgroup || mount -t cgroup2 none /sys/fs/cgroup 2>/dev/null || true
setsid dockerd --host=unix:///var/run/docker.sock --storage-driver=vfs </dev/null >/var/log/dockerd.log 2>&1 &
sleep 1
'
for i in $(seq 1 30); do
if wsl -u root -- docker info >/dev/null 2>&1; then echo "docker is up"; break; fi
echo "waiting for dockerd ($i)..."; sleep 2
done
if ! wsl -u root -- docker info >/dev/null 2>&1; then
echo "=== dockerd did not come up; diagnostics ==="
wsl -u root -- cat /var/log/dockerd.log 2>/dev/null || echo "(dockerd wrote no log)"
wsl -u root -- sh -c 'command -v dockerd; ls -la /var/run/docker.sock 2>&1; ls /sys/fs/cgroup 2>&1 | head; uname -r'
exit 1
fi
wsl -u root -- docker version
# Runs on all platforms (incl. darwin): `check` only reports, never installs,
# so it is safe on the macOS runner (which has no Docker). Informational —
# missing tools are expected pre-bootstrap, hence `|| echo`.
- name: 'CLI: prerequisites check'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: prerequisites check (report-only, informational)"
echo "==================================================================="
"$OF_BIN" prerequisites check || echo "some prerequisites missing (expected pre-bootstrap)"
# Cloud-cluster command SURFACE, with no cloud and no Docker: flag
# validation, the EKS/GKE dry-runs' graceful degradation, and `cluster
# use` error paths. Everything here is either exit-0 by design or an
# expected, asserted failure — no resource is ever created. Pure, so it
# runs on every OS (on Windows via the WSL forward, exercising the Linux
# binary's paths).
- name: 'Cluster: cloud command surface (no cloud, no Docker)'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: cloud cluster surface — validation, dry-run, use"
echo "==================================================================="
fail() { echo "::error::$1"; exit 1; }
# Portable hang guard: macOS runners ship coreutils' gtimeout, not
# timeout. Fall back to no guard rather than failing with exit 127.
run_to() {
local t="$1"; shift
if command -v timeout >/dev/null 2>&1; then timeout "$t" "$@"
elif command -v gtimeout >/dev/null 2>&1; then gtimeout "$t" "$@"
else "$@"; fi
}
echo "--- eks --skip-wizard demands --region"
if "$OF_BIN" cluster create eks-smoke --type eks --skip-wizard >/tmp/eks.out 2>&1; then
cat /tmp/eks.out; fail "eks without region must be rejected"
fi
grep -q '\-\-region is required' /tmp/eks.out || { cat /tmp/eks.out; fail "missing required-flag message"; }
echo "--- eks dry-run degrades gracefully without terraform/credentials"
# Two legitimate outcomes on an unauthenticated runner: terraform absent
# -> soft skip (exit 0); terraform present -> the AWS credential
# preflight fails fast (non-interactive, exit non-zero). Never a hang,
# never a created resource.
set +e
run_to 120 "$OF_BIN" cluster create eks-smoke --type eks \
--region us-east-1 --skip-wizard --dry-run </dev/null >/tmp/eks.out 2>&1
code=$?
set -e
[ "$code" -eq 124 ] && { cat /tmp/eks.out; fail "eks dry-run hung on a prompt"; }
grep -qiE "skipping the plan preview|cannot authenticate|not installed" /tmp/eks.out \
|| { cat /tmp/eks.out; fail "eks dry-run produced neither a skip nor an auth hint (exit $code)"; }
echo "OK (exit $code)"
echo "--- unknown --type is rejected"
if "$OF_BIN" cluster create x --type minikube --skip-wizard >/tmp/neg.out 2>&1; then
cat /tmp/neg.out; fail "unknown cluster type must be rejected"
fi
grep -q "unknown cluster type" /tmp/neg.out || { cat /tmp/neg.out; fail "missing unknown-type message"; }
echo "--- gke --skip-wizard demands --region and --project"
if "$OF_BIN" cluster create x --type gke --skip-wizard >/tmp/neg.out 2>&1; then
cat /tmp/neg.out; fail "gke without region/project must be rejected"
fi
grep -qE '\-\-(region|project) is required' /tmp/neg.out || { cat /tmp/neg.out; fail "missing required-flag message"; }
echo "--- gke dry-run degrades gracefully without terraform/credentials"
# Two legitimate outcomes on an unauthenticated runner: terraform absent
# -> soft skip (exit 0); terraform present -> the auth flow fails fast
# (non-interactive, exit non-zero). Both must be actionable, never a hang.
set +e
run_to 120 "$OF_BIN" cluster create gke-smoke --type gke \
--project no-such-project --region us-central1 --skip-wizard --dry-run </dev/null >/tmp/gke.out 2>&1
code=$?
set -e
[ "$code" -eq 124 ] && { cat /tmp/gke.out; fail "gke dry-run hung on a prompt"; }
grep -qiE "skipping the plan preview|not authenticated|not installed|application default|not accessible" /tmp/gke.out \
|| { cat /tmp/gke.out; fail "gke dry-run produced neither a skip nor an auth hint (exit $code)"; }
echo "OK (exit $code)"
echo "--- cluster use of an unknown cluster is an actionable error"
set +e
run_to 60 "$OF_BIN" cluster use no-such-cluster </dev/null >/tmp/use.out 2>&1
code=$?
set -e
[ "$code" -eq 124 ] && { cat /tmp/use.out; fail "cluster use hung on a prompt"; }
[ "$code" -eq 0 ] && { cat /tmp/use.out; fail "cluster use of an unknown cluster must fail"; }
grep -qiE "not (known|found) locally|gcloud" /tmp/use.out || { cat /tmp/use.out; fail "missing actionable use error"; }
echo "cloud surface OK"
# Everything below needs a Docker daemon + k3d. It runs on linux and on
# windows (the Windows binary forwards into WSL, where Docker was started
# above), exercising the real cluster lifecycle on both. darwin stays
# read-only (no nested Docker on the macOS runner).
- name: 'CLI: cluster queries (empty)'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: cluster list on an empty state (text/json/yaml)"
echo "==================================================================="
"$OF_BIN" cluster list
"$OF_BIN" cluster list -o json
"$OF_BIN" cluster list -o yaml
# Managed-cloud OWNERSHIP GUARDS, exercised on the real binary against a
# fake GKE workspace (an isolated OPENFRAME_CLUSTERS_DIR with a
# hand-written cluster.json) — no cloud, no terraform, fully
# deterministic. Locks: registry-backed list/status, the typed-confirm
# refusal on non-interactive cloud delete, and use's missing-context
# error. Linux only: one leg suffices and the Windows binary would need
# the fake workspace inside WSL's home.
- name: 'Cluster: managed-cloud guards (fake workspace, no cloud)'
if: matrix.os == 'linux'
shell: bash
env:
OPENFRAME_CLUSTERS_DIR: ${{ runner.temp }}/of-fake-clusters
run: |
echo "==================================================================="
echo "=== TEST: cloud ownership guards on a fake GKE workspace"
echo "==================================================================="
fail() { echo "::error::$1"; exit 1; }
export KUBECONFIG="$RUNNER_TEMP/of-fake-kubeconfig"
mkdir -p "$OPENFRAME_CLUSTERS_DIR/fake-gke"
cat > "$OPENFRAME_CLUSTERS_DIR/fake-gke/cluster.json" <<'EOF'
{"name":"fake-gke","type":"gke","status":"ready","region":"us-central1",
"project":"fake-project","node_count":3,"created_at":"2026-07-22T00:00:00Z",
"endpoint":"https://203.0.113.10","ca_cert":"ZmFrZS1jYQ=="}
EOF
echo "--- the registry cluster appears in list with source=openframe"
"$OF_BIN" cluster list -o json | jq -e \
'.[] | select(.name == "fake-gke")' >/dev/null \
|| fail "fake-gke missing from cluster list -o json"
echo "--- status resolves through the registry (no cloud calls)"
"$OF_BIN" cluster status fake-gke -o json | jq -e \
'.type == "gke"' >/dev/null || fail "status -o json lacks type gke"
echo "--- use without a kubeconfig context fails with the exact hint"
set +e
"$OF_BIN" cluster use fake-gke </dev/null >/tmp/use.out 2>&1
code=$?
set -e
[ "$code" -eq 0 ] && { cat /tmp/use.out; fail "use without a context must fail"; }
grep -q "no kubeconfig context" /tmp/use.out || { cat /tmp/use.out; fail "missing no-context hint"; }
echo "--- non-interactive cloud delete WITHOUT --force refuses"
# Two refusal gates guard this path, and either may fire first: the
# generic deletion confirmation ("confirmation required ... re-run
# with --force") refuses before the cloud-specific typed-confirm
# gate ("refusing to destroy cloud cluster") is reached. Both are
# correct refusals — assert the guarantee (non-zero exit, actionable
# message, workspace survives), not which gate fired.
set +e
timeout 60 "$OF_BIN" cluster delete fake-gke </dev/null >/tmp/del.out 2>&1
code=$?
set -e
[ "$code" -eq 124 ] && { cat /tmp/del.out; fail "cluster delete hung on a prompt"; }
[ "$code" -eq 0 ] && { cat /tmp/del.out; fail "non-interactive cloud delete without --force must refuse"; }
grep -qE "refusing to destroy cloud cluster|confirmation required.*--force" /tmp/del.out || { cat /tmp/del.out; fail "missing an actionable refusal message"; }
echo "--- the workspace (the only pointer to billed resources) survived"
[ -f "$OPENFRAME_CLUSTERS_DIR/fake-gke/cluster.json" ] || fail "refused delete must not remove the workspace"
echo "ownership guards OK"
# --- Medium: dry-run, still no real changes ----------------------------
- name: 'CLI: dry-run create & install'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: dry-run create & install (no real changes)"
echo "==================================================================="
"$OF_BIN" cluster create "$OF_CLUSTER" --type k3d --nodes 1 --skip-wizard --dry-run
# No cluster exists yet, so an install (even --dry-run) correctly exits
# non-zero ("no cluster selected"); this just smokes that it reaches the
# cluster check without crashing.
"$OF_BIN" app install --non-interactive --dry-run || echo "app install without a cluster exits non-zero (expected pre-create)"
# Guard: --silent must suppress non-error output. A dry-run create
# normally prints an INFO "Configuration Summary" box; with --silent it must
# be gone, while the same run without --silent must still show it.
- name: 'CLI: --silent suppresses output'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: --silent suppresses INFO output (with positive control)"
echo "==================================================================="
"$OF_BIN" --silent cluster create sil-$RANDOM --type k3d --nodes 1 --skip-wizard --dry-run >silent.out 2>&1 || true
"$OF_BIN" cluster create loud-$RANDOM --type k3d --nodes 1 --skip-wizard --dry-run >loud.out 2>&1 || true
# Strict contract: --silent means ZERO non-error output, not merely
# "no summary box" — the 0.4.7 verification report found blank lines
# leaking through raw fmt prints and graded them as silence.
if [ -s silent.out ]; then echo "::error::--silent produced output:"; cat silent.out; exit 1; fi
if ! grep -qi 'Configuration Summary' loud.out; then echo "::error::control run printed no summary — test is moot, check the assertion"; exit 1; fi
echo "silent output empty, control printed the summary — OK"
# --- Complex: real cluster lifecycle (heaviest, last) ------------------
- name: 'Cluster: create'
if: matrix.os != 'darwin'
shell: bash
timeout-minutes: 20
run: |
echo "==================================================================="
echo "=== TEST: real cluster create + list/status (text/json/yaml)"
echo "==================================================================="
"$OF_BIN" cluster create "$OF_CLUSTER" --type k3d --nodes 1 --skip-wizard
echo "--- cluster list & status after create"
"$OF_BIN" cluster list
"$OF_BIN" cluster status "$OF_CLUSTER"
"$OF_BIN" cluster status "$OF_CLUSTER" -o json
"$OF_BIN" cluster status "$OF_CLUSTER" -o yaml
# Guard: --non-interactive with a cluster present but NO cluster
# name must fail fast, never drop into the interactive picker (which hangs
# CI until the job timeout). Runs with a short timeout so a regression that
# reintroduces the prompt is caught as exit 124, not a 40-minute hang.
# Validation matrix: invalid parameters must exit non-zero with the
# documented message, and machine output must stay parseable. These need
# real tools on the runner (the text-mode prerequisite gate), so they
# live here rather than in the hermetic unit-level CLI matrix.
- name: 'CLI: validation matrix (negative + machine output)'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: invalid parameters exit non-zero; JSON stays parseable"
echo "==================================================================="
fail() { echo "::error::$1"; exit 1; }
expect_fail() {
desc="$1"; shift
if "$OF_BIN" "$@" </dev/null >/tmp/neg.out 2>&1; then
cat /tmp/neg.out; fail "expected failure: $desc"
fi
echo "OK (non-zero): $desc"
}
expect_fail "status with invalid --output" cluster status "$OF_CLUSTER" -o bogus
expect_fail "list with invalid --output" cluster list -o bogus
expect_fail "machine status without a name" cluster status -o json
expect_fail "status of a nonexistent cluster" cluster status no-such-cluster -o json
# Machine output contract: valid JSON on stdout, parseable by jq.
"$OF_BIN" cluster list -o json | jq -e 'type == "array"' >/dev/null \
|| fail "cluster list -o json is not a JSON array"
"$OF_BIN" cluster status "$OF_CLUSTER" -o json | jq -e '.name and (.ready_servers | type == "number") and (.total_servers | type == "number")' >/dev/null \
|| fail "cluster status -o json lacks name/ready_servers/total_servers"
echo "validation matrix OK"
# Guard (verification report N2): an explicit --context IS the install
# target — it must be accepted in non-interactive mode without a cluster
# name (0.4.7 failed this exact invocation with "requires a cluster name").
- name: 'App: --context works as the target in non-interactive mode'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: app install --context <ctx> --non-interactive --dry-run"
echo "==================================================================="
"$OF_BIN" app install --context "$OF_CONTEXT" --non-interactive --dry-run
- name: 'App: --non-interactive without a name fails fast'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: --non-interactive fails fast, never drops into a prompt"
echo "==================================================================="
set +e
timeout 60 "$OF_BIN" app install --non-interactive
code=$?
set -e
if [ "$code" -eq 124 ]; then echo "::error::app install --non-interactive hung on a prompt"; exit 1; fi
if [ "$code" -eq 0 ]; then echo "::error::app install --non-interactive without a cluster name must fail, got exit 0"; exit 1; fi
echo "fast-fail OK (exit $code)"
- name: 'App: install'
if: matrix.os != 'darwin'
shell: bash
timeout-minutes: 30
env:
# Cap the CLI's app-wait below this step's 30m limit so it fails with
# its OWN diagnostic (stuck apps + health messages) instead of being
# killed opaquely by the job timeout.
OPENFRAME_APP_WAIT_TIMEOUT: 20m
run: |
echo "==================================================================="
echo "=== TEST: real app install (non-interactive, full ArgoCD wait)"
echo "==================================================================="
"$OF_BIN" app install "$OF_CLUSTER" --non-interactive --verbose
- name: 'App: status & access'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: app status (text/json/yaml) & access after install"
echo "==================================================================="
"$OF_BIN" app status --context "$OF_CONTEXT"
"$OF_BIN" app status --context "$OF_CONTEXT" -o json
"$OF_BIN" app status --context "$OF_CONTEXT" -o yaml
"$OF_BIN" app access --context "$OF_CONTEXT"
# Guard (verification report N1): an upgrade with NO values file must
# fail fast naming the file — an empty values map would make helm replace
# the release values with chart defaults, wiping the configuration.
- name: 'App: upgrade without a values file fails fast'
if: matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEST: app upgrade --ref refuses to run without a values file"
echo "==================================================================="
BIN="$GITHUB_WORKSPACE/${OF_BIN#./}"
tmp="$(mktemp -d)"
set +e
out="$(cd "$tmp" && "$BIN" app upgrade --ref main "$OF_CLUSTER" 2>&1)"
code=$?
set -e
echo "$out" | tail -5
if [ "$code" -eq 0 ]; then echo "::error::upgrade without a values file must fail"; exit 1; fi
if ! echo "$out" | grep -q "openframe-helm-values.yaml"; then
echo "::error::the error must name the missing values file"; exit 1
fi
echo "fail-fast OK (exit $code)"
- name: 'App: upgrade (dry-run then force-sync)'
if: matrix.os != 'darwin'
shell: bash
timeout-minutes: 20
run: |
echo "==================================================================="
echo "=== TEST: app upgrade — dry-run, then force-sync"
echo "==================================================================="
"$OF_BIN" app upgrade --context "$OF_CONTEXT" --dry-run
echo "--- force-sync"
"$OF_BIN" app upgrade --context "$OF_CONTEXT" --sync --verbose
# cluster cleanup is a first-class non-interactive command with logic no
# other step exercises against a real cluster: kube-context-pinned Helm
# uninstalls, ArgoCD finalizer stripping, and node image pruning via crictl
# (k3d nodes run containerd, not docker). Runs on the live, populated cluster
# so all phases have something to do; --force skips the confirmation prompt.
# It leaves the cluster empty; the teardown then deletes it.
- name: 'Cluster: cleanup (releases + crictl image prune)'
if: matrix.os != 'darwin'
shell: bash
timeout-minutes: 10
run: |
echo "==================================================================="
echo "=== TEST: cluster cleanup --force (non-interactive)"
echo "==================================================================="
"$OF_BIN" cluster cleanup "$OF_CLUSTER" --force
# --- Teardown (runs even if a step above failed) -----------------------
- name: 'Teardown: uninstall & delete cluster'
if: always() && matrix.os != 'darwin'
shell: bash
run: |
echo "==================================================================="
echo "=== TEARDOWN: app uninstall & cluster delete --force"
echo "==================================================================="
"$OF_BIN" app uninstall --context "$OF_CONTEXT" --yes || true
"$OF_BIN" cluster delete "$OF_CLUSTER" --force || true