All checks were successful
CICD / Build and Publish CICD Base Image (push) Successful in 6m8s
CICD / Build and Push CICD Image (push) Successful in 23m14s
CICD / Build CICD Image Failure Postmortem (push) Has been skipped
CICD / Backend Tests (push) Successful in 7m10s
CICD / Frontend Tests (push) Successful in 45s
CICD / Backend Doctests (push) Successful in 18s
CICD / Pre-commit Checks (push) Successful in 14m57s
CICD / Source Lanes Failure Postmortem (push) Has been skipped
CICD / CICD Tests Complete (push) Successful in 3s
CICD / Build Backend Base Image (push) Successful in 18s
CICD / Build Integration Tester Image (push) Successful in 1m5s
CICD / Build Backend Main Image (push) Successful in 1m52s
CICD / Build Frontend Base Image (push) Successful in 10m42s
CICD / Build Frontend Main Image (push) Successful in 33s
CICD / Build E2E Tester Image (push) Successful in 32m17s
CICD / Production Images Complete (push) Successful in 5s
CICD / Production Image Failures Postmortem (push) Has been skipped
CICD / Runtime Black-Box Integration Tests (push) Successful in 1m13s
CICD / Integration Tests Failure Postmortem (push) Has been skipped
CICD / End-to-End Tests (push) Successful in 11m23s
CICD / E2E Tests Failure Postmortem (push) Has been skipped
## Summary Hardens CI workflows for self-hosted Gitea runners by stabilizing E2E execution and Renovate behavior across internal/external network paths. Closes #62 ## What Changed ### E2E workflow reliability - Fixed E2E workspace handoff to ensure expected repository contents are present during test execution. - Added stricter preflight checks for required frontend files before running E2E. - Reduced mount/path fragility while preserving runtime image pull and compose flow. ### Renovate workflow hardening - Added internal-first endpoint reachability selection with fallback handling. - Added token preflight checks for repository access. - Added explicit host-rule auth handling for API/git paths. - Added container-level connectivity preflight diagnostics. - Added git URL override aligned with selected endpoint context. - Removed incorrect forced Dogar host-IP pinning that broke HTTPS clone routing. ## Why CI behavior was sensitive to runner networking and Renovate clone/auth interactions. These changes make the workflow deterministic in our runner topology and address recurring CI failures. ## Scope - Workflow logic only (`cicd.yaml`, `renovate.yml`) - No app feature or API behavior changes ## Validation - Workflow YAML validation passed during updates. - Changes were applied and verified iteratively from real failing run diagnostics. Co-authored-by: copilotcoder <copilotcoder@darkhelm.org> Reviewed-on: #73
122 lines
3.9 KiB
Plaintext
Executable File
122 lines
3.9 KiB
Plaintext
Executable File
#!/usr/bin/env xonsh
|
|
|
|
"""Roll out one-runner stabilization on a target host (runner1 active, runner2 removed)."""
|
|
|
|
import sys
|
|
from datetime import datetime
|
|
|
|
|
|
HOST_PATHS = {
|
|
"kankali.darkhelm.lan": "/home/darkhelm/Projects/DarkHelm.org/gitea",
|
|
"zhokq.darkhelm.lan": "/home/darkhelm/Projects/DarkHelm.org/gitea-runner",
|
|
"urtzul.darkhelm.lan": "/home/darkhelm/Projects/DarkHelm.org/gitea-runner",
|
|
"pi-desktop.darkhelm.lan": "/home/darkhelm/Projects/DarkHelm.org/gitea-runner",
|
|
}
|
|
|
|
WARMUP_IMAGES = [
|
|
"ubuntu:22.04",
|
|
"python:3.14-slim",
|
|
"node:20-bookworm-slim",
|
|
"ghcr.io/catthehacker/ubuntu:act-latest",
|
|
"kankali.darkhelm.lan:3001/darkhelm.org/act-ubuntu:act-latest",
|
|
"ghcr.io/renovatebot/renovate:41",
|
|
]
|
|
|
|
|
|
def usage(exit_code=1):
|
|
print("Usage: xonsh rollout-one-runner.xsh <host> [--batch]")
|
|
print("Hosts:")
|
|
for host in HOST_PATHS:
|
|
print(f" - {host}")
|
|
raise SystemExit(exit_code)
|
|
|
|
|
|
if len(sys.argv) < 2:
|
|
usage()
|
|
|
|
host = sys.argv[1]
|
|
if host not in HOST_PATHS:
|
|
print(f"Unknown host: {host}")
|
|
usage()
|
|
|
|
batch_mode = "--batch" in sys.argv[2:]
|
|
root = HOST_PATHS[host]
|
|
ts = datetime.now().strftime("%Y%m%d_%H%M%S")
|
|
|
|
|
|
def ssh(cmd):
|
|
if batch_mode:
|
|
out = !(ssh -o BatchMode=yes -o ConnectTimeout=20 -o ServerAliveInterval=30 -o ServerAliveCountMax=3 -o ControlMaster=auto -o ControlPersist=2h -o ControlPath=~/.ssh/cm-%r@%h:%p @(host) @(cmd))
|
|
else:
|
|
out = !(ssh -o ConnectTimeout=60 -o ServerAliveInterval=30 -o ServerAliveCountMax=6 -o ControlMaster=auto -o ControlPersist=2h -o ControlPath=~/.ssh/cm-%r@%h:%p @(host) @(cmd))
|
|
return out.returncode, str(out.out).rstrip("\n")
|
|
|
|
|
|
print(f"Host: {host}")
|
|
print(f"Path: {root}")
|
|
print(f"SSH mode: {'batch' if batch_mode else 'interactive'}")
|
|
print("Phase 1: pre-change snapshot")
|
|
|
|
rc, out = ssh(f"docker compose -f {root}/compose.yml config --services")
|
|
print(out)
|
|
|
|
rc, out = ssh("docker inspect gitea-act-runner-1 --format 'runner1 restart={{.RestartCount}} oom={{.State.OOMKilled}} status={{.State.Status}}'")
|
|
print(out if rc == 0 else "runner1-missing")
|
|
|
|
rc, out = ssh("docker inspect gitea-act-runner-2 --format 'runner2 restart={{.RestartCount}} oom={{.State.OOMKilled}} status={{.State.Status}}'")
|
|
print(out if rc == 0 else "runner2-missing")
|
|
|
|
print("Phase 2: backup compose and env")
|
|
for backup_step in [
|
|
f"mkdir -p {root}/backups",
|
|
f"cp -f {root}/compose.yml {root}/backups/compose.yml.{ts}",
|
|
f"cp -f {root}/.env {root}/backups/.env.{ts}",
|
|
]:
|
|
rc, out = ssh(backup_step)
|
|
if out:
|
|
print(out)
|
|
if rc != 0:
|
|
print("Backup failed. Aborting.")
|
|
raise SystemExit(2)
|
|
print("backups-created")
|
|
|
|
print("Phase 3: switch to one runner")
|
|
for may_fail in [
|
|
f"docker compose -f {root}/compose.yml stop act_runner_2",
|
|
f"docker compose -f {root}/compose.yml rm -f act_runner_2",
|
|
]:
|
|
rc, out = ssh(may_fail)
|
|
if out:
|
|
print(out)
|
|
|
|
rc, out = ssh(f"docker compose -f {root}/compose.yml up -d act_runner_1")
|
|
if out:
|
|
print(out)
|
|
if rc != 0:
|
|
print("Runner switch failed.")
|
|
raise SystemExit(3)
|
|
|
|
print("Phase 4: verify")
|
|
for verify_cmd in [
|
|
"docker ps -a --format 'table {{.Names}}\t{{.Status}}' | grep -E 'NAMES|gitea-act-runner'",
|
|
"docker inspect gitea-act-runner-1 --format 'runner1 restart={{.RestartCount}} oom={{.State.OOMKilled}} status={{.State.Status}} started={{.State.StartedAt}}'",
|
|
"docker logs --tail=30 gitea-act-runner-1 | tail -n 30",
|
|
]:
|
|
rc, out = ssh(verify_cmd)
|
|
print(out)
|
|
|
|
rc, out = ssh("docker inspect gitea-act-runner-2 --format 'runner2 status={{.State.Status}}'")
|
|
print(out if rc == 0 else "runner2-removed")
|
|
|
|
print("Phase 5: warm images")
|
|
pulls = "\n".join([f"docker pull {image}" for image in WARMUP_IMAGES])
|
|
rc, out = ssh(f"set -e\n{pulls}")
|
|
if rc != 0:
|
|
print("Warmup failed")
|
|
print(out)
|
|
raise SystemExit(4)
|
|
print(out)
|
|
|
|
print("One-runner rollout complete for host")
|
|
print(f"Backup files: {root}/backups/compose.yml.{ts} and {root}/backups/.env.{ts}")
|