Compare commits

...
4 Commits
Author SHA1 Message Date
yukkop cc8a7cf80e fix: reconcile healthy terminal runners
runner nix smoke / nix label and flake smoke (push) Failing after 28s
2026-09-11 09:19:04 +00:00
yukkop 279df769db fix: retain healthy terminal runners 2026-09-11 09:19:03 +00:00
yukkop 069b18daa3 fix: check runner health before reuse 2026-09-11 09:19:03 +00:00
yukkop 37bd69e90e fix: avoid Cargo metadata IFD 2026-09-11 09:19:03 +00:00
11 changed files with 209 additions and 117 deletions
+1 -1
View File
@@ -126,7 +126,7 @@ in {
else throw (envErrorMessage varName);
# -- Cargo.toml --
cargoToml = src: (builtins.fromTOML (builtins.readFile "${src}/Cargo.toml"));
cargoToml = manifest: (builtins.fromTOML (builtins.readFile manifest));
# Consolidated SQL bundles for the `hectic` schema. Single source of truth
# for everything that creates objects in the `hectic` namespace, used by
+9 -51
View File
@@ -194,6 +194,7 @@ gcr_alloc_deferred() {
reused="$(gcr_record_get "$job_id" "$attempt")"
vm_id="$(gcr_record_field "$reused" vm_id)"
if gcr_vm_runner_service "$vm_id" start \
&& gcr_vm_runner_service "$vm_id" health \
&& gcr_gitea_runner_disabled "$repo" "$(gcr_record_field "$reused" vm_name)" false; then
reused="$(gcr_record_get "$job_id" "$attempt")"
reused="$(printf '%s' "$reused" | jq -c '.bootstrapped = true | del(.reused_vm)')"
@@ -410,6 +411,7 @@ gcr_bootstrap_pending() {
if [ "$(gcr_record_field "$rec" reused_vm)" = "true" ]; then
if gcr_vm_runner_service "$vm_id" start \
&& gcr_vm_runner_service "$vm_id" health \
&& gcr_gitea_runner_disabled "$repo" "$runner_name" false; then
rec="$(printf '%s' "$rec" | jq -c '.bootstrapped = true | del(.reused_vm)')"
gcr_record_put "$job_id" "$attempt" "$rec"
@@ -506,58 +508,14 @@ gcr_reap_finished_jobs() {
pending_vm|vm_active) ;;
*) gcr_lock_release "$key"; continue ;;
esac
vm_id="$(gcr_record_field "$rec" vm_id)"
if [ "$state" = "completed:success" ] \
&& idle_rec="$(gcr_record_idle_json "$rec")"; then
if ! gcr_lock_acquire idle-pool; then
gcr_lock_release "$key"
continue
fi
runner_name="$(gcr_record_field "$rec" vm_name)"
if ! gcr_gitea_runner_disabled "$repo" "$runner_name" true \
|| ! gcr_vm_runner_service "$vm_id" stop; then
gcr_lock_release idle-pool
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" \
idle-stop-failed false; then
gcr_event "vm-destroyed" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\",\"via\":\"reconcile\"}"
else
gcr_event "vm-cleanup-pending" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\",\"via\":\"reconcile\"}"
fi
gcr_lock_release "$key"
continue
fi
gcr_record_put "$job_id" "$attempt" "$idle_rec"
idle_expires="$(gcr_record_field "$idle_rec" idle_expires_at)"
gcr_lock_release idle-pool
gcr_lock_release "$key"
gcr_event "vm-idle" "$job_id" "{\"vm_id\":$vm_id,\"expires_at\":$idle_expires,\"via\":\"reconcile\"}"
gcr_log info --ns=sweep "job=$job_id succeeded, retaining vm=$vm_id until $idle_expires"
continue
fi
gcr_log info --ns=sweep "job=$job_id terminal ($state), destroying vm=$vm_id"
if [ -n "$vm_id" ] && [ "$vm_id" != "0" ] && [ "$vm_id" != "null" ]; then
case "$state" in
completed:success|completed:cancelled|completed:skipped) ;;
*)
ip="$(gcr_vm_public_ip "$vm_id" || true)"
gcr_vm_collect_diagnostics "$vm_id" "$ip" "$job_id" "$state" || true
;;
esac
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" \
job-completed false; then
gcr_event "vm-destroyed" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"job-completed\",\"state\":\"$state\"}"
else
gcr_event "vm-cleanup-pending" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"job-completed\",\"state\":\"$state\"}"
fi
else
gcr_record_del "$job_id" "$attempt"
fi
finish_status=0
gcr_vm_finish_terminal "$job_id" "$attempt" "$rec" "$state" reconcile \
|| finish_status="$?"
gcr_lock_release "$key"
case "$finish_status" in
0|2) ;;
*) return "$finish_status" ;;
esac
;;
esac
done
+76 -2
View File
@@ -409,7 +409,11 @@ gcr_vm_public_ip() {
# The controller starts the service only after the claim record is written.
gcr_vm_runner_service() {
vm_id="$1"; action="$2"
case "$action" in start|stop) ;; *) return 1 ;; esac
case "$action" in
start|stop) service_command="systemctl $action gitea-runner.service" ;;
health) service_command="systemctl is-active --quiet gitea-runner.service" ;;
*) return 1 ;;
esac
ip="$(gcr_vm_public_ip "$vm_id")" || return 1
[ -n "$ip" ] || return 1
test -n "${GCR_SSH_PRIVKEY_FILE:-}" && test -r "$GCR_SSH_PRIVKEY_FILE" || return 1
@@ -418,7 +422,7 @@ gcr_vm_runner_service() {
printf '\n' >> "$key_tmp"
chmod 0600 "$key_tmp"
ssh_opts="-i $key_tmp -o IdentitiesOnly=yes -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o ConnectTimeout=5 -o BatchMode=yes"
if timeout 30 ssh $ssh_opts "root@$ip" "systemctl $action gitea-runner.service"; then
if timeout 30 ssh $ssh_opts "root@$ip" "$service_command"; then
rm -f "$key_tmp"
return 0
fi
@@ -468,6 +472,76 @@ gcr_vm_collect_diagnostics() {
return 0
}
# Finish a terminal job under its allocation lock. Healthy bootstrapped VMs
# become idle until their existing billing boundary; every unsafe transition
# uses durable cleanup_pending teardown instead.
gcr_vm_finish_terminal() {
finish_job="$1"; finish_attempt="$2"; finish_rec="$3"; finish_state="$4"
finish_via="${5:-webhook}"
finish_vm_id="$(gcr_record_field "$finish_rec" vm_id)"
if [ -n "$finish_vm_id" ] && [ "$finish_vm_id" != "null" ] \
&& [ "$finish_vm_id" != "0" ]; then
case "$finish_state" in
completed:success|completed:cancelled|completed:skipped) ;;
completed:*)
finish_ip="$(gcr_vm_public_ip "$finish_vm_id" || true)"
gcr_vm_collect_diagnostics "$finish_vm_id" "$finish_ip" \
"$finish_job" "$finish_state" || true
;;
esac
fi
if finish_idle_rec="$(gcr_record_idle_json "$finish_rec")"; then
gcr_lock_acquire idle-pool || return 2
finish_repo="$(gcr_record_field "$finish_rec" repo)"
finish_runner="$(gcr_record_field "$finish_rec" vm_name)"
if ! gcr_vm_runner_service "$finish_vm_id" health; then
gcr_lock_release idle-pool
finish_cleanup_reason=idle-health-failed
elif ! gcr_gitea_runner_disabled "$finish_repo" "$finish_runner" true \
|| ! gcr_vm_runner_service "$finish_vm_id" stop; then
gcr_lock_release idle-pool
finish_cleanup_reason=idle-stop-failed
elif ! gcr_record_put "$finish_job" "$finish_attempt" "$finish_idle_rec"; then
gcr_lock_release idle-pool
finish_cleanup_reason=idle-state-write-failed
else
finish_expires="$(gcr_record_field "$finish_idle_rec" idle_expires_at)"
gcr_lock_release idle-pool
gcr_event "vm-idle" "$finish_job" \
"{\"vm_id\":$finish_vm_id,\"expires_at\":$finish_expires,\"via\":\"$finish_via\"}"
gcr_log info --ns=sweep \
"job=$finish_job terminal ($finish_state), retaining vm=$finish_vm_id until $finish_expires"
return 0
fi
if gcr_vm_cleanup_start "$finish_job" "$finish_attempt" "$finish_rec" \
"$finish_cleanup_reason" false; then
gcr_event "vm-destroyed" "$finish_job" \
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_cleanup_reason\",\"via\":\"$finish_via\"}"
else
gcr_event "vm-cleanup-pending" "$finish_job" \
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_cleanup_reason\",\"via\":\"$finish_via\"}"
fi
return 0
fi
if [ -n "$finish_vm_id" ] && [ "$finish_vm_id" != "null" ] \
&& [ "$finish_vm_id" != "0" ]; then
if gcr_vm_cleanup_start "$finish_job" "$finish_attempt" "$finish_rec" \
"$finish_state" false; then
gcr_event "vm-destroyed" "$finish_job" \
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_state\",\"via\":\"$finish_via\"}"
else
gcr_event "vm-cleanup-pending" "$finish_job" \
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_state\",\"via\":\"$finish_via\"}"
fi
else
gcr_record_del "$finish_job" "$finish_attempt"
fi
}
# Bootstrap delivery is SSH-push from the controller. The MicroOS snapshot's
# cloud-init cannot fetch user-data (Hetzner datasource DHCP failure), so the
# controller drives provisioning over SSH using GCR_SSH_PRIVKEY_FILE, whose
+2 -2
View File
@@ -98,8 +98,8 @@ gcr_now_epoch() {
date -u '+%s'
}
# Successful VMs remain reusable until next billing-hour boundary, but never
# beyond profile hard TTL. Prints updated idle record when retention is safe.
# Healthy bootstrapped VMs remain reusable until next billing-hour boundary,
# but never beyond profile hard TTL. Prints updated idle record when safe.
gcr_record_idle_json() {
gcr_idle_rec="$1"
[ "$(gcr_record_field "$gcr_idle_rec" bootstrapped)" = "true" ] || return 1
+8 -47
View File
@@ -155,6 +155,7 @@ gcr_alloc() {
vm_id="$(gcr_record_field "$reused" vm_id)"
vm_name="$(gcr_record_field "$reused" vm_name)"
if gcr_vm_runner_service "$vm_id" start \
&& gcr_vm_runner_service "$vm_id" health \
&& gcr_gitea_runner_disabled "$repo" "$vm_name" false; then
reused="$(gcr_record_get "$job_id" "$attempt")"
reused="$(printf '%s' "$reused" | jq -c '.bootstrapped = true | del(.reused_vm)')"
@@ -247,54 +248,14 @@ gcr_deallocate() {
;;
esac
vm_id="$(gcr_record_field "$rec" vm_id)"
if [ "$new_status" = "completed:success" ] \
&& idle_rec="$(gcr_record_idle_json "$rec")"; then
if ! gcr_lock_acquire idle-pool; then
gcr_lock_release "$key"
return 0
fi
runner_name="$(gcr_record_field "$rec" vm_name)"
if ! gcr_gitea_runner_disabled "$(gcr_record_field "$rec" repo)" "$runner_name" true \
|| ! gcr_vm_runner_service "$vm_id" stop; then
gcr_lock_release idle-pool
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" idle-stop-failed false; then
gcr_event "vm-destroyed" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\"}"
else
gcr_event "vm-cleanup-pending" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\"}"
fi
gcr_lock_release "$key"
return 0
fi
gcr_record_put "$job_id" "$attempt" "$idle_rec"
idle_expires="$(gcr_record_field "$idle_rec" idle_expires_at)"
gcr_lock_release idle-pool
gcr_lock_release "$key"
gcr_event "vm-idle" "$job_id" "{\"vm_id\":$vm_id,\"expires_at\":$idle_expires}"
return 0
fi
if [ -n "$vm_id" ] && [ "$vm_id" != "null" ] && [ "$vm_id" != "0" ]; then
case "$new_status" in
completed:success|completed:cancelled|completed:skipped) ;;
completed:*)
ip="$(gcr_vm_public_ip "$vm_id" || true)"
gcr_vm_collect_diagnostics "$vm_id" "$ip" "$job_id" "$new_status" || true
;;
esac
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" "$new_status" false; then
gcr_event "vm-destroyed" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"$new_status\"}"
else
gcr_event "vm-cleanup-pending" "$job_id" \
"{\"vm_id\":$vm_id,\"reason\":\"$new_status\"}"
fi
else
gcr_record_del "$job_id" "$attempt"
fi
finish_status=0
gcr_vm_finish_terminal "$job_id" "$attempt" "$rec" "$new_status" webhook \
|| finish_status="$?"
gcr_lock_release "$key"
case "$finish_status" in
0|2) return 0 ;;
*) return "$finish_status" ;;
esac
}
gcr_mark_in_progress() {
+1 -1
View File
@@ -5,7 +5,7 @@
...
}: let
src = ./.;
cargo = cargoToml src;
cargo = cargoToml ./Cargo.toml;
in
pkgs.rustPlatform.buildRustPackage {
pname = cargo.package.name;
+1 -1
View File
@@ -6,7 +6,7 @@
...
}: let
src = ./.;
cargo = cargoToml src;
cargo = cargoToml ./Cargo.toml;
in
pkgs.rustPlatform.buildRustPackage {
pname = cargo.package.name;
+1 -1
View File
@@ -5,7 +5,7 @@
...
}: let
src = ./.;
cargo = cargoToml src;
cargo = cargoToml ./Cargo.toml;
in
pkgs.rustPlatform.buildRustPackage {
pname = cargo.package.name;
@@ -67,21 +67,28 @@ jq -e 'select(.job_id == "302" and .vm_id == 71 and
"$(gcr_record_path 302 1)" >/dev/null
test "$(grep -Ec '^(budget|token|create)$' "$calls" || true)" = 0
# Start failure keeps Gitea runner disabled and record retryable.
# Failed post-start health check keeps runner disabled and record retryable.
retry_idle='{"job_id":"315","run_attempt":"1","repo":"hinterland/hearth","label":"gross-arm","created_at":1000,"ttl_min":180,"vm_id":79,"vm_name":"gcr-315-1","bootstrapped":true,"status":"idle_vm","idle_since":2000,"idle_expires_at":4600}'
gcr_record_put 315 1 "$retry_idle"
export GCR_PER_REPO_CAP=2
FAIL_START=1
FAIL_HEALTH=1
gcr_vm_runner_service() {
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
[ "$2" = start ] && [ "$FAIL_START" = 1 ] && return 1
[ "$2" = health ] && [ "$FAIL_HEALTH" = 1 ] && return 1
return 0
}
gcr_alloc 316 1 hinterland/hearth '["gross-arm"]'
retry_rec="$(gcr_record_get 316 1)"
test "$(gcr_record_field "$retry_rec" bootstrapped)" = false
test "$(gcr_record_field "$retry_rec" reused_vm)" = true
grep -q '^runner start vm=79$' "$calls"
grep -q '^runner health vm=79$' "$calls"
grep -q '^runner-disabled gcr-315-1 true$' "$calls"
FAIL_START=0
if grep -q '^runner-disabled gcr-315-1 false$' "$calls"; then
printf 'unhealthy reused runner must never become schedulable\n' >&2
exit 1
fi
FAIL_HEALTH=0
gcr_record_del 316 1
# Expired idle capacity is never claimed; normal allocation then charges once.
@@ -16,6 +16,7 @@ gcr_gitea_job_state() {
case "$2" in
101) printf 'completed:success' ;;
102) printf 'completed:failure' ;;
105) printf 'completed:skipped' ;;
*) return 1 ;;
esac
}
@@ -36,7 +37,9 @@ gcr_vm_runner_service() {
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
}
gcr_gitea_runner_disabled() { :; }
gcr_gitea_runner_disabled() {
printf 'disabled runner=%s value=%s\n' "$2" "$3" >> "$calls"
}
record_success='{"job_id":"101","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"0","ttl_min":480,"vm_id":41,"vm_name":"gcr-101-1","bootstrapped":true,"status":"vm_active"}'
record_failure='{"job_id":"102","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"1","ttl_min":480,"vm_id":42,"vm_name":"gcr-102-1","bootstrapped":true,"status":"vm_active"}'
@@ -45,8 +48,14 @@ gcr_record_put 101 1 "$record_success"
gcr_record_put 102 1 "$record_failure"
gcr_reap_finished_jobs
grep -q 'destroy vm=42' "$calls"
grep -q 'diag vm=42 ip=192.0.2.42 job=102 reason=completed:failure' "$calls"
grep -q 'runner health vm=42' "$calls"
grep -q 'disabled runner=gcr-102-1 value=true' "$calls"
grep -q 'runner stop vm=42' "$calls"
test "$(grep -E '^(diag vm=42|runner health vm=42|disabled runner=gcr-102-1|runner stop vm=42)' "$calls")" = 'diag vm=42 ip=192.0.2.42 job=102 reason=completed:failure
runner health vm=42
disabled runner=gcr-102-1 value=true
runner stop vm=42'
if grep -q 'diag vm=41' "$calls"; then
printf 'success job should not collect diagnostics\n' >&2
exit 1
@@ -56,12 +65,28 @@ if grep -q 'destroy vm=41' "$calls"; then
printf 'successful job VM should remain idle until billing boundary\n' >&2
exit 1
fi
if grep -q 'destroy vm=42' "$calls"; then
printf 'failed bootstrapped reconciled VM should remain idle until billing boundary\n' >&2
exit 1
fi
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3600)' \
"$(gcr_record_path 101 1)" >/dev/null
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3601)' \
"$(gcr_record_path 102 1)" >/dev/null
gcr_idle_record_usable "$(gcr_record_get 102 1)"
idle_once="$(gcr_record_get 101 1)"
gcr_reap_finished_jobs
test "$(gcr_record_get 101 1)" = "$idle_once"
test ! -e "$(gcr_record_path 102 1)"
# Skipped terminal jobs follow same healthy retention policy without diagnostics.
record_skipped='{"job_id":"105","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":45,"vm_name":"gcr-105-1","bootstrapped":true,"status":"vm_active"}'
gcr_record_put 105 1 "$record_skipped"
gcr_reap_finished_jobs
test "$(gcr_record_field "$(gcr_record_get 105 1)" status)" = idle_vm
if grep -q 'diag vm=45' "$calls" || grep -q 'destroy vm=45' "$calls"; then
printf 'healthy skipped VM must be retained without failure diagnostics\n' >&2
exit 1
fi
# Reaper keeps cleanup ownership after DELETE failure and retries next sweep.
record_delete_fail='{"job_id":"104","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":44,"vm_name":"gcr-104-1","bootstrapped":true,"status":"vm_active"}'
@@ -76,6 +101,10 @@ gcr_vm_destroy() {
printf 'destroy-failed vm=%s\n' "$1" >> "$calls"
return 1
}
gcr_vm_runner_service() {
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
[ "$1" != 44 ] || [ "$2" != stop ]
}
gcr_reap_finished_jobs
test "$(gcr_record_field "$(gcr_record_get 104 1)" status)" = cleanup_pending
test "$(gcr_count_active)" = 1
@@ -101,6 +130,10 @@ gcr_gitea_job_state() {
gcr_vm_public_ip() {
return 1
}
gcr_vm_runner_service() {
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
return 1
}
gcr_vm_destroy() {
printf 'destroy vm=%s\n' "$1" >> "$calls"
@@ -109,5 +142,6 @@ gcr_vm_destroy() {
gcr_reap_finished_jobs
grep -q 'diag vm=43 ip= job=103 reason=completed:failure' "$calls_ip_fail"
grep -q 'runner health vm=43' "$calls_ip_fail"
grep -q 'destroy vm=43' "$calls_ip_fail"
test ! -e "$(gcr_record_path 103 1)"
@@ -28,7 +28,9 @@ gcr_vm_runner_service() {
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
}
gcr_gitea_runner_disabled() { :; }
gcr_gitea_runner_disabled() {
printf 'disabled runner=%s value=%s\n' "$2" "$3" >> "$calls"
}
record_success='{"job_id":"201","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"0","ttl_min":480,"vm_id":51,"vm_name":"gcr-201-1","bootstrapped":true,"status":"vm_active"}'
record_failure='{"job_id":"202","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"1","ttl_min":480,"vm_id":52,"vm_name":"gcr-202-1","bootstrapped":true,"status":"vm_active"}'
@@ -38,8 +40,14 @@ gcr_record_put 202 1 "$record_failure"
gcr_deallocate 201 1 completed:success
gcr_deallocate 202 1 completed:failure
grep -q 'destroy vm=52' "$calls"
grep -q 'diag vm=52 ip=192.0.2.52 job=202 reason=completed:failure' "$calls"
grep -q 'runner health vm=52' "$calls"
grep -q 'disabled runner=gcr-202-1 value=true' "$calls"
grep -q 'runner stop vm=52' "$calls"
test "$(grep -E '^(diag vm=52|runner health vm=52|disabled runner=gcr-202-1|runner stop vm=52)' "$calls")" = 'diag vm=52 ip=192.0.2.52 job=202 reason=completed:failure
runner health vm=52
disabled runner=gcr-202-1 value=true
runner stop vm=52'
if grep -q 'diag vm=51' "$calls"; then
printf 'success webhook should not collect diagnostics\n' >&2
exit 1
@@ -49,12 +57,36 @@ if grep -q 'destroy vm=51' "$calls"; then
printf 'successful webhook VM should remain idle until billing boundary\n' >&2
exit 1
fi
if grep -q 'destroy vm=52' "$calls"; then
printf 'failed bootstrapped webhook VM should remain idle until billing boundary\n' >&2
exit 1
fi
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3600)' \
"$(gcr_record_path 201 1)" >/dev/null
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3601)' \
"$(gcr_record_path 202 1)" >/dev/null
gcr_idle_record_usable "$(gcr_record_get 202 1)"
idle_once="$(gcr_record_get 201 1)"
gcr_record_del 201 1
gcr_lock_acquire "$(gcr_alloc_key 205 1)"
gcr_claim_idle 205 1 hinterland/hearth gross-nix-x86-perf
gcr_lock_release "$(gcr_alloc_key 205 1)"
jq -e 'select(.job_id == "205" and .vm_id == 52 and .status == "pending_vm" and .reused_vm == true)' \
"$(gcr_record_path 205 1)" >/dev/null
gcr_record_del 205 1
gcr_record_put 201 1 "$idle_once"
gcr_deallocate 201 1 completed:success
test "$(gcr_record_get 201 1)" = "$idle_once"
test ! -e "$(gcr_record_path 202 1)"
# Cancelled terminal jobs use same retention policy without failure diagnostics.
record_cancelled='{"job_id":"207","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":57,"vm_name":"gcr-207-1","bootstrapped":true,"status":"vm_active"}'
gcr_record_put 207 1 "$record_cancelled"
gcr_deallocate 207 1 completed:cancelled
test "$(gcr_record_field "$(gcr_record_get 207 1)" status)" = idle_vm
if grep -q 'diag vm=57' "$calls" || grep -q 'destroy vm=57' "$calls"; then
printf 'healthy cancelled VM must be retained without failure diagnostics\n' >&2
exit 1
fi
calls_ip_fail="$GCR_STATE_DIR/calls-ip-fail"
calls="$calls_ip_fail"
@@ -64,20 +96,33 @@ gcr_record_put 203 1 "$record_ip_fail"
gcr_vm_public_ip() {
return 1
}
gcr_vm_runner_service() {
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
return 1
}
gcr_deallocate 203 1 completed:failure
grep -q 'diag vm=53 ip= job=203 reason=completed:failure' "$calls_ip_fail"
grep -q 'runner health vm=53' "$calls_ip_fail"
grep -q 'destroy vm=53' "$calls_ip_fail"
test ! -e "$(gcr_record_path 203 1)"
if grep -q 'disabled runner=gcr-203-1' "$calls_ip_fail"; then
printf 'unhealthy terminal runner must not enter idle shutdown path\n' >&2
exit 1
fi
# Terminal webhook retains ownership and capacity until DELETE succeeds.
# Runner teardown failure destroys and retains ownership until DELETE succeeds.
record_delete_fail='{"job_id":"204","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":54,"vm_name":"gcr-204-1","bootstrapped":true,"status":"vm_active"}'
gcr_record_put 204 1 "$record_delete_fail"
gcr_vm_destroy() {
printf 'destroy-failed vm=%s\n' "$1" >> "$calls_ip_fail"
return 1
}
gcr_vm_runner_service() {
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
[ "$2" != stop ]
}
gcr_deallocate 204 1 completed:failure
cleanup_rec="$(gcr_record_get 204 1)"
test "$(gcr_record_field "$cleanup_rec" status)" = cleanup_pending
@@ -94,6 +139,19 @@ test "$(gcr_count_active)" = 0
grep -q '^destroy-failed vm=54$' "$calls_ip_fail"
grep -q '^destroy-retry vm=54$' "$calls_ip_fail"
# Unbootstrapped terminal VMs are never retained.
gcr_vm_runner_service() { printf 'unexpected-runner %s\n' "$1" >> "$calls_ip_fail"; }
gcr_vm_destroy() { printf 'destroy-unbootstrapped vm=%s\n' "$1" >> "$calls_ip_fail"; }
record_unbootstrapped='{"job_id":"206","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":56,"vm_name":"gcr-206-1","bootstrapped":false,"status":"pending_vm"}'
gcr_record_put 206 1 "$record_unbootstrapped"
gcr_deallocate 206 1 completed:failure
grep -q '^destroy-unbootstrapped vm=56$' "$calls_ip_fail"
test ! -e "$(gcr_record_path 206 1)"
if grep -q '^unexpected-runner 56$' "$calls_ip_fail"; then
printf 'unbootstrapped VM must bypass idle teardown\n' >&2
exit 1
fi
# Gitea emits zero-based run_attempt values for initial workflow jobs.
gcr_read_request() {
gcr_hdr_event_type=workflow_job