Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
61 changes: 36 additions & 25 deletions .github/skills/ctf-testing/deploy_and_test.sh
Original file line number Diff line number Diff line change
Expand Up @@ -15,7 +15,7 @@
# reboot and progress persists
#
# Prerequisites:
# - terraform (>= 1.0)
# - terraform (>= 1.0; Azure requires >= 1.14.0)
# - jq (for AWS terraform config)
# - sshpass (macOS: brew install hudochenkov/sshpass/sshpass)
# - aws CLI (for AWS, must be logged in)
Expand Down Expand Up @@ -398,7 +398,8 @@ _reboot_vm() {
local provider="$1"
local ip="$2"

_log INFO "Rebooting VM (${provider})..."
# This function returns the VM IP via stdout, so logs must go to stderr.
_log INFO "Rebooting VM (${provider})..." >&2

case "${provider}" in
aws)
Expand All @@ -416,33 +417,27 @@ _reboot_vm() {
return 1
fi

echo " Stopping instance ${instance_id}..."
echo " Stopping instance ${instance_id}..." >&2
aws ec2 stop-instances --instance-ids "${instance_id}" > /dev/null
aws ec2 wait instance-stopped --instance-ids "${instance_id}"
echo " Starting instance ${instance_id}..."
echo " Starting instance ${instance_id}..." >&2
aws ec2 start-instances --instance-ids "${instance_id}" > /dev/null
aws ec2 wait instance-running --instance-ids "${instance_id}"
# IP may change, get new one
sleep 10
ip=$(_get_public_ip "${provider}")
;;
azure)
echo " Restarting Azure VM..."
az vm restart --resource-group ctf-resources --name ctf-vm
# az vm restart waits by default, but add explicit wait for running state
az vm wait \
--resource-group ctf-resources \
--name ctf-vm \
--created \
--timeout 120 2>/dev/null || true
echo " Restarting Azure VM..." >&2
az vm restart --resource-group ctf-resources --name ctf-vm >&2 || return 1
;;
Comment on lines 430 to 433
Comment on lines +431 to 433
gcp)
echo " Restarting GCP VM..."
echo " Restarting GCP VM..." >&2
local zone
zone=$(cd "${REPO_ROOT}/${provider}" \
&& terraform output -raw zone 2>/dev/null \
|| echo "us-central1-a")
gcloud compute instances reset ctf-instance --zone="${zone}" --quiet
gcloud compute instances reset ctf-instance --zone="${zone}" --quiet >&2 || return 1
# Wait for VM to be running
local attempts=0
while [[ ${attempts} -lt 30 ]]; do
Expand All @@ -467,6 +462,19 @@ _reboot_vm() {
# TEST EXECUTION
# =============================================================================

# Copy test script to VM
# Arguments:
# $1 - Cloud provider name
# $2 - IP address of the VM
_copy_test_script() {
local provider="$1"
local ip="$2"

_log INFO "Copying test script to ${provider} VM..."
# shellcheck disable=SC2086
_sshpass_cmd scp ${SSH_OPTS} "${TEST_SCRIPT}" "${SSH_USER}@${ip}:/tmp/test_ctf_challenges.sh"
}
Comment on lines +469 to +476

# Copy test script to VM and execute it
# Arguments:
# $1 - Cloud provider name
Expand All @@ -482,9 +490,7 @@ _run_tests() {
test_flags="${test_flags} --with-reboot"
fi

_log INFO "Copying test script to VM..."
# shellcheck disable=SC2086
_sshpass_cmd scp ${SSH_OPTS} "${TEST_SCRIPT}" "${SSH_USER}@${ip}:/tmp/test_ctf_challenges.sh"
_copy_test_script "${provider}" "${ip}"

_log INFO "Running tests on ${provider} VM (${ip})..."
echo ""
Expand All @@ -508,11 +514,14 @@ _run_post_reboot_tests() {
local provider="$1"
local ip="$2"

_copy_test_script "${provider}" "${ip}"

_log INFO "Running post-reboot verification on ${provider}..."

local exit_code=0
# shellcheck disable=SC2086
_sshpass_cmd ssh ${SSH_OPTS} "${SSH_USER}@${ip}" "/tmp/test_ctf_challenges.sh" \
_sshpass_cmd ssh ${SSH_OPTS} "${SSH_USER}@${ip}" \
"chmod +x /tmp/test_ctf_challenges.sh && /tmp/test_ctf_challenges.sh --post-reboot" \
|| exit_code=$?

return "${exit_code}"
Expand Down Expand Up @@ -592,13 +601,15 @@ _test_provider() {
_log WARN "Reboot requested - performing VM reboot..."

local new_ip
new_ip=$(_reboot_vm "${provider}" "${ip}")

# Wait for SSH after reboot
_wait_for_ssh "${new_ip}"

# Run post-reboot tests
_run_post_reboot_tests "${provider}" "${new_ip}" || test_exit_code=$?
if ! new_ip=$(_reboot_vm "${provider}" "${ip}"); then
_log ERROR "VM reboot failed for ${provider}"
result=1
elif ! _wait_for_ssh "${new_ip}"; then
_log ERROR "SSH connection failed after reboot for ${provider}"
result=1
elif ! _run_post_reboot_tests "${provider}" "${new_ip}"; then
result=1
fi
elif [[ ${test_exit_code} -ne 0 ]]; then
result=1
fi
Expand Down
60 changes: 45 additions & 15 deletions .github/skills/ctf-testing/test_ctf_challenges.sh
Original file line number Diff line number Diff line change
Expand Up @@ -9,11 +9,12 @@
# can complete the CTF.
#
# Usage:
# ./test_ctf_challenges.sh [--with-reboot]
# ./test_ctf_challenges.sh [--with-reboot|--post-reboot]
# DEBUG=true ./test_ctf_challenges.sh # Enable debug tracing
#
# Flags:
# --with-reboot After tests pass, signal reboot to verify services persist
# --post-reboot Run only the post-reboot verification phase
#
# Exit codes:
# 0 - All tests passed
Expand Down Expand Up @@ -44,9 +45,10 @@ readonly GREEN='\033[0;32m'
readonly YELLOW='\033[1;33m'
readonly NC='\033[0m' # No Color

# File paths for reboot test coordination
readonly REBOOT_MARKER="/tmp/.ctf_reboot_test_marker"
readonly PROGRESS_SNAPSHOT="/tmp/.ctf_progress_snapshot"
# File paths for reboot test coordination. These must survive a VM reboot.
readonly TEST_STATE_DIR="${HOME}/.linux-ctfs-test"
readonly REBOOT_MARKER="${TEST_STATE_DIR}/.ctf_reboot_test_marker"
readonly PROGRESS_SNAPSHOT="${TEST_STATE_DIR}/.ctf_progress_snapshot"

# =============================================================================
# GLOBAL STATE
Expand All @@ -58,15 +60,34 @@ FAILED=0

# Parse arguments
WITH_REBOOT=false
for arg in "$@"; do
case $arg in
POST_REBOOT=false
usage() {
echo "Usage: $0 [--with-reboot|--post-reboot]"
}

while [[ $# -gt 0 ]]; do
case "$1" in
--with-reboot)
WITH_REBOOT=true
shift
;;
--post-reboot)
POST_REBOOT=true
;;
Comment on lines 71 to +75
*)
echo "Unknown argument: $1"
usage
exit 1
;;
esac
shift
done
Comment on lines 61 to 83

if [[ "${WITH_REBOOT}" == true && "${POST_REBOOT}" == true ]]; then
echo "--with-reboot and --post-reboot cannot be used together."
usage
exit 1
fi

# =============================================================================
# HELPER FUNCTIONS
# =============================================================================
Expand Down Expand Up @@ -130,31 +151,39 @@ _verify_flag() {
# ============================================================================
# POST-REBOOT VERIFICATION
# ============================================================================
if [[ -f "${REBOOT_MARKER}" ]]; then
if [[ "${POST_REBOOT}" == true ]]; then
_section "POST-REBOOT VERIFICATION"


if [[ ! -f "${REBOOT_MARKER}" ]]; then
_fail "Reboot marker not found - reboot verification was not prepared"
echo ""
echo "Passed: ${PASSED} | Failed: ${FAILED}"
exit 1
fi

echo "Verifying services survived reboot..."

for service in ctf-secret-service ctf-monitor-directory ctf-ping-message ctf-secret-process nginx; do
if systemctl is-active "${service}" &>/dev/null; then
_pass "${service} is running after reboot"
else
_fail "${service} failed to start after reboot - SETUP BUG"
fi
done

if [ -f "$PROGRESS_SNAPSHOT" ]; then
EXPECTED=$(cat "$PROGRESS_SNAPSHOT")
ACTUAL=$(sort -u /var/ctf/completed_challenges 2>/dev/null | wc -l)
ACTUAL=$( { sort -u /var/ctf/completed_challenges 2>/dev/null || true; } | wc -l )
if [ "$ACTUAL" -ge "$EXPECTED" ]; then
_pass "Progress persisted after reboot ($ACTUAL checks)"
else
_fail "Progress lost after reboot (expected ${EXPECTED}, got ${ACTUAL})"
fi
fi

rm -f "${REBOOT_MARKER}" "${PROGRESS_SNAPSHOT}"

rmdir "${TEST_STATE_DIR}" 2>/dev/null || true

echo ""
echo "Passed: ${PASSED} | Failed: ${FAILED}"
[[ ${FAILED} -eq 0 ]] && exit 0 || exit 1
Expand Down Expand Up @@ -644,9 +673,10 @@ echo "Flags captured: ${#FLAGS[@]}"
echo ""

if [ "$WITH_REBOOT" = true ] && [ $FAILED -eq 0 ]; then
mkdir -p "${TEST_STATE_DIR}"
sort -u /var/ctf/completed_challenges 2>/dev/null | wc -l > "$PROGRESS_SNAPSHOT"
touch "$REBOOT_MARKER"
echo "Reboot marker created. Re-run after reboot to verify services."
echo "Reboot marker created. After reboot, re-run with --post-reboot to verify services."
exit 100
fi

Expand Down
8 changes: 7 additions & 1 deletion CONTRIBUTING.md
Original file line number Diff line number Diff line change
Expand Up @@ -71,7 +71,7 @@ All PRs that change setup, challenges, verify behavior, or Terraform should be t

Install:

1. `terraform` 1.0 or newer
1. `terraform` 1.0 or newer; Azure requires Terraform 1.14.0 or newer
2. `jq`
3. `sshpass`
4. The cloud CLI for the provider you want to test
Expand Down Expand Up @@ -120,6 +120,12 @@ Most contributors only need to know this:
- Contributor testing uses local files.
- `deploy_and_test.sh` handles contributor mode for you.

Setup readiness differs by provider:

- Azure release mode uses VM Custom Script Extension (Terraform 1.14.0 or newer), so Terraform waits for extension success or failure.
- AWS and GCP release mode still use the shared SSH marker wait.
- Contributor mode stays on `use_local_setup=true`, uploading local files over SSH for test runs.

If you manually run Terraform to test local setup changes, pass:

```bash
Expand Down
10 changes: 9 additions & 1 deletion azure/README.md
Original file line number Diff line number Diff line change
Expand Up @@ -93,9 +93,17 @@ Type `yes` when prompted.
## Troubleshooting

1. Ensure your Azure CLI is logged in with valid credentials
2. Check that you're using Terraform v1.9.0 or later
2. Check that you're using Terraform v1.14.0 or later
3. Verify you have permissions to create VMs, VNets, and Network Security Groups

If release setup fails during `terraform apply`, Azure reports the failure through the VM Custom Script Extension. Useful VM-side logs are:

```text
/var/log/ctf_setup.log
/var/log/waagent.log
/var/log/azure/custom-script/handler.log
```

If problems persist, please open an issue:

https://github.com/learntocloud/linux-ctfs/issues
Expand Down
Loading
Loading