Compare commits
4
Commits
572133a941
...
cc8a7cf80e
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
cc8a7cf80e | ||
|
|
279df769db | ||
|
|
069b18daa3 | ||
|
|
37bd69e90e |
+1
-1
@@ -126,7 +126,7 @@ in {
|
||||
else throw (envErrorMessage varName);
|
||||
|
||||
# -- Cargo.toml --
|
||||
cargoToml = src: (builtins.fromTOML (builtins.readFile "${src}/Cargo.toml"));
|
||||
cargoToml = manifest: (builtins.fromTOML (builtins.readFile manifest));
|
||||
|
||||
# Consolidated SQL bundles for the `hectic` schema. Single source of truth
|
||||
# for everything that creates objects in the `hectic` namespace, used by
|
||||
|
||||
@@ -194,6 +194,7 @@ gcr_alloc_deferred() {
|
||||
reused="$(gcr_record_get "$job_id" "$attempt")"
|
||||
vm_id="$(gcr_record_field "$reused" vm_id)"
|
||||
if gcr_vm_runner_service "$vm_id" start \
|
||||
&& gcr_vm_runner_service "$vm_id" health \
|
||||
&& gcr_gitea_runner_disabled "$repo" "$(gcr_record_field "$reused" vm_name)" false; then
|
||||
reused="$(gcr_record_get "$job_id" "$attempt")"
|
||||
reused="$(printf '%s' "$reused" | jq -c '.bootstrapped = true | del(.reused_vm)')"
|
||||
@@ -410,6 +411,7 @@ gcr_bootstrap_pending() {
|
||||
|
||||
if [ "$(gcr_record_field "$rec" reused_vm)" = "true" ]; then
|
||||
if gcr_vm_runner_service "$vm_id" start \
|
||||
&& gcr_vm_runner_service "$vm_id" health \
|
||||
&& gcr_gitea_runner_disabled "$repo" "$runner_name" false; then
|
||||
rec="$(printf '%s' "$rec" | jq -c '.bootstrapped = true | del(.reused_vm)')"
|
||||
gcr_record_put "$job_id" "$attempt" "$rec"
|
||||
@@ -506,58 +508,14 @@ gcr_reap_finished_jobs() {
|
||||
pending_vm|vm_active) ;;
|
||||
*) gcr_lock_release "$key"; continue ;;
|
||||
esac
|
||||
vm_id="$(gcr_record_field "$rec" vm_id)"
|
||||
if [ "$state" = "completed:success" ] \
|
||||
&& idle_rec="$(gcr_record_idle_json "$rec")"; then
|
||||
if ! gcr_lock_acquire idle-pool; then
|
||||
gcr_lock_release "$key"
|
||||
continue
|
||||
fi
|
||||
runner_name="$(gcr_record_field "$rec" vm_name)"
|
||||
if ! gcr_gitea_runner_disabled "$repo" "$runner_name" true \
|
||||
|| ! gcr_vm_runner_service "$vm_id" stop; then
|
||||
gcr_lock_release idle-pool
|
||||
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" \
|
||||
idle-stop-failed false; then
|
||||
gcr_event "vm-destroyed" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\",\"via\":\"reconcile\"}"
|
||||
else
|
||||
gcr_event "vm-cleanup-pending" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\",\"via\":\"reconcile\"}"
|
||||
fi
|
||||
gcr_lock_release "$key"
|
||||
continue
|
||||
fi
|
||||
gcr_record_put "$job_id" "$attempt" "$idle_rec"
|
||||
idle_expires="$(gcr_record_field "$idle_rec" idle_expires_at)"
|
||||
gcr_lock_release idle-pool
|
||||
gcr_lock_release "$key"
|
||||
gcr_event "vm-idle" "$job_id" "{\"vm_id\":$vm_id,\"expires_at\":$idle_expires,\"via\":\"reconcile\"}"
|
||||
gcr_log info --ns=sweep "job=$job_id succeeded, retaining vm=$vm_id until $idle_expires"
|
||||
continue
|
||||
fi
|
||||
|
||||
gcr_log info --ns=sweep "job=$job_id terminal ($state), destroying vm=$vm_id"
|
||||
if [ -n "$vm_id" ] && [ "$vm_id" != "0" ] && [ "$vm_id" != "null" ]; then
|
||||
case "$state" in
|
||||
completed:success|completed:cancelled|completed:skipped) ;;
|
||||
*)
|
||||
ip="$(gcr_vm_public_ip "$vm_id" || true)"
|
||||
gcr_vm_collect_diagnostics "$vm_id" "$ip" "$job_id" "$state" || true
|
||||
;;
|
||||
esac
|
||||
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" \
|
||||
job-completed false; then
|
||||
gcr_event "vm-destroyed" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"job-completed\",\"state\":\"$state\"}"
|
||||
else
|
||||
gcr_event "vm-cleanup-pending" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"job-completed\",\"state\":\"$state\"}"
|
||||
fi
|
||||
else
|
||||
gcr_record_del "$job_id" "$attempt"
|
||||
fi
|
||||
finish_status=0
|
||||
gcr_vm_finish_terminal "$job_id" "$attempt" "$rec" "$state" reconcile \
|
||||
|| finish_status="$?"
|
||||
gcr_lock_release "$key"
|
||||
case "$finish_status" in
|
||||
0|2) ;;
|
||||
*) return "$finish_status" ;;
|
||||
esac
|
||||
;;
|
||||
esac
|
||||
done
|
||||
|
||||
@@ -409,7 +409,11 @@ gcr_vm_public_ip() {
|
||||
# The controller starts the service only after the claim record is written.
|
||||
gcr_vm_runner_service() {
|
||||
vm_id="$1"; action="$2"
|
||||
case "$action" in start|stop) ;; *) return 1 ;; esac
|
||||
case "$action" in
|
||||
start|stop) service_command="systemctl $action gitea-runner.service" ;;
|
||||
health) service_command="systemctl is-active --quiet gitea-runner.service" ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
ip="$(gcr_vm_public_ip "$vm_id")" || return 1
|
||||
[ -n "$ip" ] || return 1
|
||||
test -n "${GCR_SSH_PRIVKEY_FILE:-}" && test -r "$GCR_SSH_PRIVKEY_FILE" || return 1
|
||||
@@ -418,7 +422,7 @@ gcr_vm_runner_service() {
|
||||
printf '\n' >> "$key_tmp"
|
||||
chmod 0600 "$key_tmp"
|
||||
ssh_opts="-i $key_tmp -o IdentitiesOnly=yes -o StrictHostKeyChecking=no -o UserKnownHostsFile=/dev/null -o ConnectTimeout=5 -o BatchMode=yes"
|
||||
if timeout 30 ssh $ssh_opts "root@$ip" "systemctl $action gitea-runner.service"; then
|
||||
if timeout 30 ssh $ssh_opts "root@$ip" "$service_command"; then
|
||||
rm -f "$key_tmp"
|
||||
return 0
|
||||
fi
|
||||
@@ -468,6 +472,76 @@ gcr_vm_collect_diagnostics() {
|
||||
return 0
|
||||
}
|
||||
|
||||
# Finish a terminal job under its allocation lock. Healthy bootstrapped VMs
|
||||
# become idle until their existing billing boundary; every unsafe transition
|
||||
# uses durable cleanup_pending teardown instead.
|
||||
gcr_vm_finish_terminal() {
|
||||
finish_job="$1"; finish_attempt="$2"; finish_rec="$3"; finish_state="$4"
|
||||
finish_via="${5:-webhook}"
|
||||
finish_vm_id="$(gcr_record_field "$finish_rec" vm_id)"
|
||||
|
||||
if [ -n "$finish_vm_id" ] && [ "$finish_vm_id" != "null" ] \
|
||||
&& [ "$finish_vm_id" != "0" ]; then
|
||||
case "$finish_state" in
|
||||
completed:success|completed:cancelled|completed:skipped) ;;
|
||||
completed:*)
|
||||
finish_ip="$(gcr_vm_public_ip "$finish_vm_id" || true)"
|
||||
gcr_vm_collect_diagnostics "$finish_vm_id" "$finish_ip" \
|
||||
"$finish_job" "$finish_state" || true
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
|
||||
if finish_idle_rec="$(gcr_record_idle_json "$finish_rec")"; then
|
||||
gcr_lock_acquire idle-pool || return 2
|
||||
finish_repo="$(gcr_record_field "$finish_rec" repo)"
|
||||
finish_runner="$(gcr_record_field "$finish_rec" vm_name)"
|
||||
if ! gcr_vm_runner_service "$finish_vm_id" health; then
|
||||
gcr_lock_release idle-pool
|
||||
finish_cleanup_reason=idle-health-failed
|
||||
elif ! gcr_gitea_runner_disabled "$finish_repo" "$finish_runner" true \
|
||||
|| ! gcr_vm_runner_service "$finish_vm_id" stop; then
|
||||
gcr_lock_release idle-pool
|
||||
finish_cleanup_reason=idle-stop-failed
|
||||
elif ! gcr_record_put "$finish_job" "$finish_attempt" "$finish_idle_rec"; then
|
||||
gcr_lock_release idle-pool
|
||||
finish_cleanup_reason=idle-state-write-failed
|
||||
else
|
||||
finish_expires="$(gcr_record_field "$finish_idle_rec" idle_expires_at)"
|
||||
gcr_lock_release idle-pool
|
||||
gcr_event "vm-idle" "$finish_job" \
|
||||
"{\"vm_id\":$finish_vm_id,\"expires_at\":$finish_expires,\"via\":\"$finish_via\"}"
|
||||
gcr_log info --ns=sweep \
|
||||
"job=$finish_job terminal ($finish_state), retaining vm=$finish_vm_id until $finish_expires"
|
||||
return 0
|
||||
fi
|
||||
|
||||
if gcr_vm_cleanup_start "$finish_job" "$finish_attempt" "$finish_rec" \
|
||||
"$finish_cleanup_reason" false; then
|
||||
gcr_event "vm-destroyed" "$finish_job" \
|
||||
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_cleanup_reason\",\"via\":\"$finish_via\"}"
|
||||
else
|
||||
gcr_event "vm-cleanup-pending" "$finish_job" \
|
||||
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_cleanup_reason\",\"via\":\"$finish_via\"}"
|
||||
fi
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ -n "$finish_vm_id" ] && [ "$finish_vm_id" != "null" ] \
|
||||
&& [ "$finish_vm_id" != "0" ]; then
|
||||
if gcr_vm_cleanup_start "$finish_job" "$finish_attempt" "$finish_rec" \
|
||||
"$finish_state" false; then
|
||||
gcr_event "vm-destroyed" "$finish_job" \
|
||||
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_state\",\"via\":\"$finish_via\"}"
|
||||
else
|
||||
gcr_event "vm-cleanup-pending" "$finish_job" \
|
||||
"{\"vm_id\":$finish_vm_id,\"reason\":\"$finish_state\",\"via\":\"$finish_via\"}"
|
||||
fi
|
||||
else
|
||||
gcr_record_del "$finish_job" "$finish_attempt"
|
||||
fi
|
||||
}
|
||||
|
||||
# Bootstrap delivery is SSH-push from the controller. The MicroOS snapshot's
|
||||
# cloud-init cannot fetch user-data (Hetzner datasource DHCP failure), so the
|
||||
# controller drives provisioning over SSH using GCR_SSH_PRIVKEY_FILE, whose
|
||||
|
||||
@@ -98,8 +98,8 @@ gcr_now_epoch() {
|
||||
date -u '+%s'
|
||||
}
|
||||
|
||||
# Successful VMs remain reusable until next billing-hour boundary, but never
|
||||
# beyond profile hard TTL. Prints updated idle record when retention is safe.
|
||||
# Healthy bootstrapped VMs remain reusable until next billing-hour boundary,
|
||||
# but never beyond profile hard TTL. Prints updated idle record when safe.
|
||||
gcr_record_idle_json() {
|
||||
gcr_idle_rec="$1"
|
||||
[ "$(gcr_record_field "$gcr_idle_rec" bootstrapped)" = "true" ] || return 1
|
||||
|
||||
@@ -155,6 +155,7 @@ gcr_alloc() {
|
||||
vm_id="$(gcr_record_field "$reused" vm_id)"
|
||||
vm_name="$(gcr_record_field "$reused" vm_name)"
|
||||
if gcr_vm_runner_service "$vm_id" start \
|
||||
&& gcr_vm_runner_service "$vm_id" health \
|
||||
&& gcr_gitea_runner_disabled "$repo" "$vm_name" false; then
|
||||
reused="$(gcr_record_get "$job_id" "$attempt")"
|
||||
reused="$(printf '%s' "$reused" | jq -c '.bootstrapped = true | del(.reused_vm)')"
|
||||
@@ -247,54 +248,14 @@ gcr_deallocate() {
|
||||
;;
|
||||
esac
|
||||
|
||||
vm_id="$(gcr_record_field "$rec" vm_id)"
|
||||
if [ "$new_status" = "completed:success" ] \
|
||||
&& idle_rec="$(gcr_record_idle_json "$rec")"; then
|
||||
if ! gcr_lock_acquire idle-pool; then
|
||||
gcr_lock_release "$key"
|
||||
return 0
|
||||
fi
|
||||
runner_name="$(gcr_record_field "$rec" vm_name)"
|
||||
if ! gcr_gitea_runner_disabled "$(gcr_record_field "$rec" repo)" "$runner_name" true \
|
||||
|| ! gcr_vm_runner_service "$vm_id" stop; then
|
||||
gcr_lock_release idle-pool
|
||||
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" idle-stop-failed false; then
|
||||
gcr_event "vm-destroyed" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\"}"
|
||||
else
|
||||
gcr_event "vm-cleanup-pending" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"idle-stop-failed\"}"
|
||||
fi
|
||||
gcr_lock_release "$key"
|
||||
return 0
|
||||
fi
|
||||
gcr_record_put "$job_id" "$attempt" "$idle_rec"
|
||||
idle_expires="$(gcr_record_field "$idle_rec" idle_expires_at)"
|
||||
gcr_lock_release idle-pool
|
||||
gcr_lock_release "$key"
|
||||
gcr_event "vm-idle" "$job_id" "{\"vm_id\":$vm_id,\"expires_at\":$idle_expires}"
|
||||
return 0
|
||||
fi
|
||||
|
||||
if [ -n "$vm_id" ] && [ "$vm_id" != "null" ] && [ "$vm_id" != "0" ]; then
|
||||
case "$new_status" in
|
||||
completed:success|completed:cancelled|completed:skipped) ;;
|
||||
completed:*)
|
||||
ip="$(gcr_vm_public_ip "$vm_id" || true)"
|
||||
gcr_vm_collect_diagnostics "$vm_id" "$ip" "$job_id" "$new_status" || true
|
||||
;;
|
||||
esac
|
||||
if gcr_vm_cleanup_start "$job_id" "$attempt" "$rec" "$new_status" false; then
|
||||
gcr_event "vm-destroyed" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"$new_status\"}"
|
||||
else
|
||||
gcr_event "vm-cleanup-pending" "$job_id" \
|
||||
"{\"vm_id\":$vm_id,\"reason\":\"$new_status\"}"
|
||||
fi
|
||||
else
|
||||
gcr_record_del "$job_id" "$attempt"
|
||||
fi
|
||||
finish_status=0
|
||||
gcr_vm_finish_terminal "$job_id" "$attempt" "$rec" "$new_status" webhook \
|
||||
|| finish_status="$?"
|
||||
gcr_lock_release "$key"
|
||||
case "$finish_status" in
|
||||
0|2) return 0 ;;
|
||||
*) return "$finish_status" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
gcr_mark_in_progress() {
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
...
|
||||
}: let
|
||||
src = ./.;
|
||||
cargo = cargoToml src;
|
||||
cargo = cargoToml ./Cargo.toml;
|
||||
in
|
||||
pkgs.rustPlatform.buildRustPackage {
|
||||
pname = cargo.package.name;
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
...
|
||||
}: let
|
||||
src = ./.;
|
||||
cargo = cargoToml src;
|
||||
cargo = cargoToml ./Cargo.toml;
|
||||
in
|
||||
pkgs.rustPlatform.buildRustPackage {
|
||||
pname = cargo.package.name;
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
...
|
||||
}: let
|
||||
src = ./.;
|
||||
cargo = cargoToml src;
|
||||
cargo = cargoToml ./Cargo.toml;
|
||||
in
|
||||
pkgs.rustPlatform.buildRustPackage {
|
||||
pname = cargo.package.name;
|
||||
|
||||
@@ -67,21 +67,28 @@ jq -e 'select(.job_id == "302" and .vm_id == 71 and
|
||||
"$(gcr_record_path 302 1)" >/dev/null
|
||||
test "$(grep -Ec '^(budget|token|create)$' "$calls" || true)" = 0
|
||||
|
||||
# Start failure keeps Gitea runner disabled and record retryable.
|
||||
# Failed post-start health check keeps runner disabled and record retryable.
|
||||
retry_idle='{"job_id":"315","run_attempt":"1","repo":"hinterland/hearth","label":"gross-arm","created_at":1000,"ttl_min":180,"vm_id":79,"vm_name":"gcr-315-1","bootstrapped":true,"status":"idle_vm","idle_since":2000,"idle_expires_at":4600}'
|
||||
gcr_record_put 315 1 "$retry_idle"
|
||||
export GCR_PER_REPO_CAP=2
|
||||
FAIL_START=1
|
||||
FAIL_HEALTH=1
|
||||
gcr_vm_runner_service() {
|
||||
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
|
||||
[ "$2" = start ] && [ "$FAIL_START" = 1 ] && return 1
|
||||
[ "$2" = health ] && [ "$FAIL_HEALTH" = 1 ] && return 1
|
||||
return 0
|
||||
}
|
||||
gcr_alloc 316 1 hinterland/hearth '["gross-arm"]'
|
||||
retry_rec="$(gcr_record_get 316 1)"
|
||||
test "$(gcr_record_field "$retry_rec" bootstrapped)" = false
|
||||
test "$(gcr_record_field "$retry_rec" reused_vm)" = true
|
||||
grep -q '^runner start vm=79$' "$calls"
|
||||
grep -q '^runner health vm=79$' "$calls"
|
||||
grep -q '^runner-disabled gcr-315-1 true$' "$calls"
|
||||
FAIL_START=0
|
||||
if grep -q '^runner-disabled gcr-315-1 false$' "$calls"; then
|
||||
printf 'unhealthy reused runner must never become schedulable\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
FAIL_HEALTH=0
|
||||
gcr_record_del 316 1
|
||||
|
||||
# Expired idle capacity is never claimed; normal allocation then charges once.
|
||||
|
||||
@@ -16,6 +16,7 @@ gcr_gitea_job_state() {
|
||||
case "$2" in
|
||||
101) printf 'completed:success' ;;
|
||||
102) printf 'completed:failure' ;;
|
||||
105) printf 'completed:skipped' ;;
|
||||
*) return 1 ;;
|
||||
esac
|
||||
}
|
||||
@@ -36,7 +37,9 @@ gcr_vm_runner_service() {
|
||||
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
|
||||
}
|
||||
|
||||
gcr_gitea_runner_disabled() { :; }
|
||||
gcr_gitea_runner_disabled() {
|
||||
printf 'disabled runner=%s value=%s\n' "$2" "$3" >> "$calls"
|
||||
}
|
||||
|
||||
record_success='{"job_id":"101","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"0","ttl_min":480,"vm_id":41,"vm_name":"gcr-101-1","bootstrapped":true,"status":"vm_active"}'
|
||||
record_failure='{"job_id":"102","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"1","ttl_min":480,"vm_id":42,"vm_name":"gcr-102-1","bootstrapped":true,"status":"vm_active"}'
|
||||
@@ -45,8 +48,14 @@ gcr_record_put 101 1 "$record_success"
|
||||
gcr_record_put 102 1 "$record_failure"
|
||||
gcr_reap_finished_jobs
|
||||
|
||||
grep -q 'destroy vm=42' "$calls"
|
||||
grep -q 'diag vm=42 ip=192.0.2.42 job=102 reason=completed:failure' "$calls"
|
||||
grep -q 'runner health vm=42' "$calls"
|
||||
grep -q 'disabled runner=gcr-102-1 value=true' "$calls"
|
||||
grep -q 'runner stop vm=42' "$calls"
|
||||
test "$(grep -E '^(diag vm=42|runner health vm=42|disabled runner=gcr-102-1|runner stop vm=42)' "$calls")" = 'diag vm=42 ip=192.0.2.42 job=102 reason=completed:failure
|
||||
runner health vm=42
|
||||
disabled runner=gcr-102-1 value=true
|
||||
runner stop vm=42'
|
||||
if grep -q 'diag vm=41' "$calls"; then
|
||||
printf 'success job should not collect diagnostics\n' >&2
|
||||
exit 1
|
||||
@@ -56,12 +65,28 @@ if grep -q 'destroy vm=41' "$calls"; then
|
||||
printf 'successful job VM should remain idle until billing boundary\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
if grep -q 'destroy vm=42' "$calls"; then
|
||||
printf 'failed bootstrapped reconciled VM should remain idle until billing boundary\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3600)' \
|
||||
"$(gcr_record_path 101 1)" >/dev/null
|
||||
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3601)' \
|
||||
"$(gcr_record_path 102 1)" >/dev/null
|
||||
gcr_idle_record_usable "$(gcr_record_get 102 1)"
|
||||
idle_once="$(gcr_record_get 101 1)"
|
||||
gcr_reap_finished_jobs
|
||||
test "$(gcr_record_get 101 1)" = "$idle_once"
|
||||
test ! -e "$(gcr_record_path 102 1)"
|
||||
|
||||
# Skipped terminal jobs follow same healthy retention policy without diagnostics.
|
||||
record_skipped='{"job_id":"105","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":45,"vm_name":"gcr-105-1","bootstrapped":true,"status":"vm_active"}'
|
||||
gcr_record_put 105 1 "$record_skipped"
|
||||
gcr_reap_finished_jobs
|
||||
test "$(gcr_record_field "$(gcr_record_get 105 1)" status)" = idle_vm
|
||||
if grep -q 'diag vm=45' "$calls" || grep -q 'destroy vm=45' "$calls"; then
|
||||
printf 'healthy skipped VM must be retained without failure diagnostics\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Reaper keeps cleanup ownership after DELETE failure and retries next sweep.
|
||||
record_delete_fail='{"job_id":"104","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":44,"vm_name":"gcr-104-1","bootstrapped":true,"status":"vm_active"}'
|
||||
@@ -76,6 +101,10 @@ gcr_vm_destroy() {
|
||||
printf 'destroy-failed vm=%s\n' "$1" >> "$calls"
|
||||
return 1
|
||||
}
|
||||
gcr_vm_runner_service() {
|
||||
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
|
||||
[ "$1" != 44 ] || [ "$2" != stop ]
|
||||
}
|
||||
gcr_reap_finished_jobs
|
||||
test "$(gcr_record_field "$(gcr_record_get 104 1)" status)" = cleanup_pending
|
||||
test "$(gcr_count_active)" = 1
|
||||
@@ -101,6 +130,10 @@ gcr_gitea_job_state() {
|
||||
gcr_vm_public_ip() {
|
||||
return 1
|
||||
}
|
||||
gcr_vm_runner_service() {
|
||||
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
|
||||
return 1
|
||||
}
|
||||
|
||||
gcr_vm_destroy() {
|
||||
printf 'destroy vm=%s\n' "$1" >> "$calls"
|
||||
@@ -109,5 +142,6 @@ gcr_vm_destroy() {
|
||||
gcr_reap_finished_jobs
|
||||
|
||||
grep -q 'diag vm=43 ip= job=103 reason=completed:failure' "$calls_ip_fail"
|
||||
grep -q 'runner health vm=43' "$calls_ip_fail"
|
||||
grep -q 'destroy vm=43' "$calls_ip_fail"
|
||||
test ! -e "$(gcr_record_path 103 1)"
|
||||
|
||||
@@ -28,7 +28,9 @@ gcr_vm_runner_service() {
|
||||
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
|
||||
}
|
||||
|
||||
gcr_gitea_runner_disabled() { :; }
|
||||
gcr_gitea_runner_disabled() {
|
||||
printf 'disabled runner=%s value=%s\n' "$2" "$3" >> "$calls"
|
||||
}
|
||||
|
||||
record_success='{"job_id":"201","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"0","ttl_min":480,"vm_id":51,"vm_name":"gcr-201-1","bootstrapped":true,"status":"vm_active"}'
|
||||
record_failure='{"job_id":"202","run_attempt":"1","repo":"hinterland/hearth","label":"gross-nix-x86-perf","created_at":"1","ttl_min":480,"vm_id":52,"vm_name":"gcr-202-1","bootstrapped":true,"status":"vm_active"}'
|
||||
@@ -38,8 +40,14 @@ gcr_record_put 202 1 "$record_failure"
|
||||
gcr_deallocate 201 1 completed:success
|
||||
gcr_deallocate 202 1 completed:failure
|
||||
|
||||
grep -q 'destroy vm=52' "$calls"
|
||||
grep -q 'diag vm=52 ip=192.0.2.52 job=202 reason=completed:failure' "$calls"
|
||||
grep -q 'runner health vm=52' "$calls"
|
||||
grep -q 'disabled runner=gcr-202-1 value=true' "$calls"
|
||||
grep -q 'runner stop vm=52' "$calls"
|
||||
test "$(grep -E '^(diag vm=52|runner health vm=52|disabled runner=gcr-202-1|runner stop vm=52)' "$calls")" = 'diag vm=52 ip=192.0.2.52 job=202 reason=completed:failure
|
||||
runner health vm=52
|
||||
disabled runner=gcr-202-1 value=true
|
||||
runner stop vm=52'
|
||||
if grep -q 'diag vm=51' "$calls"; then
|
||||
printf 'success webhook should not collect diagnostics\n' >&2
|
||||
exit 1
|
||||
@@ -49,12 +57,36 @@ if grep -q 'destroy vm=51' "$calls"; then
|
||||
printf 'successful webhook VM should remain idle until billing boundary\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
if grep -q 'destroy vm=52' "$calls"; then
|
||||
printf 'failed bootstrapped webhook VM should remain idle until billing boundary\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3600)' \
|
||||
"$(gcr_record_path 201 1)" >/dev/null
|
||||
jq -e 'select(.status == "idle_vm" and .idle_expires_at == 3601)' \
|
||||
"$(gcr_record_path 202 1)" >/dev/null
|
||||
gcr_idle_record_usable "$(gcr_record_get 202 1)"
|
||||
idle_once="$(gcr_record_get 201 1)"
|
||||
gcr_record_del 201 1
|
||||
gcr_lock_acquire "$(gcr_alloc_key 205 1)"
|
||||
gcr_claim_idle 205 1 hinterland/hearth gross-nix-x86-perf
|
||||
gcr_lock_release "$(gcr_alloc_key 205 1)"
|
||||
jq -e 'select(.job_id == "205" and .vm_id == 52 and .status == "pending_vm" and .reused_vm == true)' \
|
||||
"$(gcr_record_path 205 1)" >/dev/null
|
||||
gcr_record_del 205 1
|
||||
gcr_record_put 201 1 "$idle_once"
|
||||
gcr_deallocate 201 1 completed:success
|
||||
test "$(gcr_record_get 201 1)" = "$idle_once"
|
||||
test ! -e "$(gcr_record_path 202 1)"
|
||||
|
||||
# Cancelled terminal jobs use same retention policy without failure diagnostics.
|
||||
record_cancelled='{"job_id":"207","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":57,"vm_name":"gcr-207-1","bootstrapped":true,"status":"vm_active"}'
|
||||
gcr_record_put 207 1 "$record_cancelled"
|
||||
gcr_deallocate 207 1 completed:cancelled
|
||||
test "$(gcr_record_field "$(gcr_record_get 207 1)" status)" = idle_vm
|
||||
if grep -q 'diag vm=57' "$calls" || grep -q 'destroy vm=57' "$calls"; then
|
||||
printf 'healthy cancelled VM must be retained without failure diagnostics\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
calls_ip_fail="$GCR_STATE_DIR/calls-ip-fail"
|
||||
calls="$calls_ip_fail"
|
||||
@@ -64,20 +96,33 @@ gcr_record_put 203 1 "$record_ip_fail"
|
||||
gcr_vm_public_ip() {
|
||||
return 1
|
||||
}
|
||||
gcr_vm_runner_service() {
|
||||
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
|
||||
return 1
|
||||
}
|
||||
|
||||
gcr_deallocate 203 1 completed:failure
|
||||
|
||||
grep -q 'diag vm=53 ip= job=203 reason=completed:failure' "$calls_ip_fail"
|
||||
grep -q 'runner health vm=53' "$calls_ip_fail"
|
||||
grep -q 'destroy vm=53' "$calls_ip_fail"
|
||||
test ! -e "$(gcr_record_path 203 1)"
|
||||
if grep -q 'disabled runner=gcr-203-1' "$calls_ip_fail"; then
|
||||
printf 'unhealthy terminal runner must not enter idle shutdown path\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Terminal webhook retains ownership and capacity until DELETE succeeds.
|
||||
# Runner teardown failure destroys and retains ownership until DELETE succeeds.
|
||||
record_delete_fail='{"job_id":"204","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":54,"vm_name":"gcr-204-1","bootstrapped":true,"status":"vm_active"}'
|
||||
gcr_record_put 204 1 "$record_delete_fail"
|
||||
gcr_vm_destroy() {
|
||||
printf 'destroy-failed vm=%s\n' "$1" >> "$calls_ip_fail"
|
||||
return 1
|
||||
}
|
||||
gcr_vm_runner_service() {
|
||||
printf 'runner %s vm=%s\n' "$2" "$1" >> "$calls"
|
||||
[ "$2" != stop ]
|
||||
}
|
||||
gcr_deallocate 204 1 completed:failure
|
||||
cleanup_rec="$(gcr_record_get 204 1)"
|
||||
test "$(gcr_record_field "$cleanup_rec" status)" = cleanup_pending
|
||||
@@ -94,6 +139,19 @@ test "$(gcr_count_active)" = 0
|
||||
grep -q '^destroy-failed vm=54$' "$calls_ip_fail"
|
||||
grep -q '^destroy-retry vm=54$' "$calls_ip_fail"
|
||||
|
||||
# Unbootstrapped terminal VMs are never retained.
|
||||
gcr_vm_runner_service() { printf 'unexpected-runner %s\n' "$1" >> "$calls_ip_fail"; }
|
||||
gcr_vm_destroy() { printf 'destroy-unbootstrapped vm=%s\n' "$1" >> "$calls_ip_fail"; }
|
||||
record_unbootstrapped='{"job_id":"206","run_attempt":"1","repo":"hinterland/hearth","label":"nix","created_at":"1","ttl_min":480,"vm_id":56,"vm_name":"gcr-206-1","bootstrapped":false,"status":"pending_vm"}'
|
||||
gcr_record_put 206 1 "$record_unbootstrapped"
|
||||
gcr_deallocate 206 1 completed:failure
|
||||
grep -q '^destroy-unbootstrapped vm=56$' "$calls_ip_fail"
|
||||
test ! -e "$(gcr_record_path 206 1)"
|
||||
if grep -q '^unexpected-runner 56$' "$calls_ip_fail"; then
|
||||
printf 'unbootstrapped VM must bypass idle teardown\n' >&2
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Gitea emits zero-based run_attempt values for initial workflow jobs.
|
||||
gcr_read_request() {
|
||||
gcr_hdr_event_type=workflow_job
|
||||
|
||||
Reference in New Issue
Block a user