#!/usr/bin/env bash
# Copyright (c) 2023-2026 Tigera, Inc. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
#     http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.

set -e

# Enable job control so that background processes get their own process groups.
set -m

my_dir="$(dirname $0)"
repo_dir="$my_dir/../.."
vm_name_prefix=$1
artifacts_dir="$repo_dir/artifacts"
# We bulk-create the VMs across all the zones in a region (based on available
# capacity) rather than pinning a single zone, so a single zone running out of
# capacity no longer blocks the job.  REGION selects the region; the zone each
# VM actually lands in is discovered after creation (see below).
region=${REGION:-europe-west3}
: "${VM_MACHINE_TYPE:=n4-standard-4}"
: "${IMAGE_FAMILY:=ubuntu-minimal-2404-lts-amd64}"
: "${MAX_RUN_DURATION:=4h}"
disk_size=${VM_DISK_SIZE:-20GB}

source "$my_dir/ssh-options"

# shellcheck source=../../felix/.semaphore/batches.sh
source "$repo_dir/$COMPONENT/.semaphore/batches.sh"

###### Create the test VMs ######

names=""
for batch in "${batches[@]}"; do
  vm_name="$vm_name_prefix$batch"
  if [ -n "$names" ]; then
    names="$names,$vm_name"
  else
    names="$vm_name"
  fi
done

# Labels for the GCP instances.  These can only contain alphanumerics,
# dashes, underscores (so we can't just default to the semaphore job name).
# Sanitize the branch name to be compatible with GCP label requirements.
branch_label="${SEMAPHORE_GIT_BRANCH:-unknown}"
branch_label="$(echo "$branch_label" | tr '[:upper:]' '[:lower:]')"
branch_label="${branch_label//[^a-z0-9_-]/-}"
labels="ci-runner=true"
labels+=",ci-job-type=${CI_JOB_TYPE_LABEL:-${SEMAPHORE_GIT_REF_TYPE:-unknown}}"
labels+=",ci-group=${CI_GROUP_LABEL:-unknown}"
labels+=",ci-job=${CI_JOB_LABEL:-unknown}"
labels+=",ci-is-rerun=${SEMAPHORE_PIPELINE_RERUN:-false}"
labels+=",ci-branch=${branch_label}"
labels+=",ci-project=${CALICO_DIR_NAME}"
if [ -n "${SEMAPHORE_GIT_PR_NUMBER}" ]; then
  labels+=",ci-pr-number=${SEMAPHORE_GIT_PR_NUMBER}"
fi
if [ -n "${SEMAPHORE_WORKFLOW_ID}" ]; then
  labels+=",ci-workflow-id=${SEMAPHORE_WORKFLOW_ID}"
fi
if [ -n "${SEMAPHORE_JOB_ID}" ]; then
  labels+=",ci-job-id=${SEMAPHORE_JOB_ID}"
fi
if [ "${SEMAPHORE_WORKFLOW_TRIGGERED_BY_SCHEDULE}" = "true" ]; then
  labels+=",ci-scheduled=true"
else
  labels+=",ci-scheduled=false"
fi

# Do a bulk create; this is faster and it saves API quota.
#
# We pass --region (not --zone) with --target-distribution-shape=any so that
# GCP spreads the VMs across the zones of the region according to available
# capacity.  This avoids a single zone filling up and failing the whole job.
echo "Creating test VMs in bulk across region ${region}..."
gcloud --quiet compute instances bulk create \
       --service-account="semaphore-v2-gcr@unique-caldron-775.iam.gserviceaccount.com" \
       --scopes="https://www.googleapis.com/auth/cloud-platform" \
       --predefined-names="$names" \
       --region=${region} \
       --target-distribution-shape=any \
       --machine-type=${VM_MACHINE_TYPE} \
       --image-family=${IMAGE_FAMILY} \
       --image-project=ubuntu-os-cloud \
       --boot-disk-size=$disk_size \
       --boot-disk-type=hyperdisk-balanced \
       --max-run-duration="${MAX_RUN_DURATION}" \
       --instance-termination-action=DELETE \
       --labels="${labels}" \
       --metadata-from-file startup-script="$my_dir/vm-bootstrap.sh" \
       --metadata block-project-ssh-keys=TRUE,ssh-keys="ubuntu:$(ssh-keygen -y -f $HOME/.ssh/id_rsa)",enable-guest-attributes=TRUE

# Since the VMs are spread across the region's zones, we can no longer assume a
# single zone for the downstream describe/ssh/delete commands.  Build a
# name->zone map by listing the instances we just created, filtered by our
# workflow-unique name prefix.  bulk create is synchronous, so the instances
# already exist by the time it returns.
echo "Discovering which zone each VM landed in..."
declare -A vm_zone
while IFS=, read -r name vm_z; do
  [ -n "$name" ] || continue
  vm_zone["$name"]="$vm_z"
  echo "  $name -> $vm_z"
done < <(gcloud --quiet compute instances list \
              --filter="name~'^${vm_name_prefix}'" \
              --format='csv[no-heading](name,zone.basename())')

# Fail fast if any expected VM is missing from the map; the rest of the script
# relies on knowing every VM's zone.
for batch in "${batches[@]}"; do
  vm_name="$vm_name_prefix$batch"
  if [ -z "${vm_zone[$vm_name]:-}" ]; then
    echo "ERROR: could not determine the zone of VM $vm_name after bulk create." 1>&2
    exit 1
  fi
done

###### Configure VMs, run tests and shut them down ######

log_monitor_regexps=(
  "(?<!Decode)Failure"
  "SUCCESS"
  "PASSED"
  "Parallel test node"
  "Test batch"
  "FV-TEST-START"
  "^test.*\.\.\. ok"
  "\.\.\. ERROR$"
  "Failure output:"
  "^ERROR:"
  "^Traceback"
  "^FAILED"
  "^OK$"
  "^XML:"
  "^\[success\]"
  "^\[error\]"
  "RUNNER:"
)

# Combine the regexps; in Perl mode, grep only supports one
# pattern so we combine them with '|'.
monitor_pattern=""
for r in "${log_monitor_regexps[@]}"; do
  monitor_pattern="${monitor_pattern}|$r"
done
monitor_pattern="${monitor_pattern:1}" # Strip leading '|'

test_pid=()
watchdog_pids=()
monitor_pids=()
log_files=()

# Calculate watchdog timeout before starting batches.
# JOB_TIMEOUT_MINUTES can be set to match the Semaphore execution_time_limit;
# the watchdog fires 5 minutes before that to leave time for cleanup.
: "${JOB_TIMEOUT_MINUTES:=60}"
if [ "$JOB_TIMEOUT_MINUTES" -le 5 ] 2>/dev/null; then
  echo "WARNING: JOB_TIMEOUT_MINUTES ($JOB_TIMEOUT_MINUTES) is too low for watchdog, disabling."
  watchdog_seconds=0
else
  watchdog_seconds=$(( (JOB_TIMEOUT_MINUTES - 5) * 60 ))
fi

# Format batch name for log files: zero-pad numeric values to 3 digits
format_batch_for_log() {
  local batch="$1"
  # Check if batch is a number
  if [[ "$batch" =~ ^[0-9]+$ ]]; then
    printf "%03d" "$batch"
  else
    echo "$batch"
  fi
}

for batch in "${batches[@]}"; do
  vm_name="$vm_name_prefix$batch"
  # Each VM may be in a different zone; look up the one it landed in.  Every
  # downstream command in this iteration (vm-ip, on-test-vm, configure-test-vm,
  # the delete, and the watchdog's serial/snapshot capture) uses $zone.  We also
  # export ZONE: run_batch (sourced from batches.sh) shells out to
  # 'gcloud get-serial-port-output --zone="${ZONE}"' to grab the serial console
  # when a VM wedges, and it reads ZONE from the environment.  The backgrounded
  # per-batch subshell below captures this value at fork time, so each batch
  # sees its own VM's zone.
  zone="${vm_zone[$vm_name]}"
  export ZONE="$zone"
  vm_ip="$(env VM_NAME="$vm_name" ZONE="$zone" "$my_dir/vm-ip")"
  batch_formatted="$(format_batch_for_log "$batch")"
  log_file="$artifacts_dir/test-$batch_formatted.log"
  failed_log_file="$artifacts_dir/test-$batch_formatted-FAILED.log"
  ssh_cmd=( env "VM_NAME=$vm_name" "ZONE=$zone" "$my_dir/on-test-vm" )
  prefix="[batch=${batch}]"
  touch "$log_file"

  # Run the configuration, test, and, teardown in a subshell so we can
  # background it.
  (
    set +e

    # Tell the watchdog (below) when this batch has fully finished, so it can
    # distinguish "done, just waiting to be reaped" from "genuinely hung".
    done_marker="$artifacts_dir/.batch-${batch_formatted}.done"
    rm -f "$done_marker"
    trap 'touch "$done_marker"' EXIT

    conf_log_file="$artifacts_dir/configure-vm-$batch_formatted.log"
    echo "$prefix Configuring test VM $vm_name. Redirecting log to $conf_log_file."
    if env ZONE="$zone" "${my_dir}/configure-test-vm" "$vm_name" >& "$conf_log_file"; then
      echo "$prefix Configuration of VM $vm_name SUCCEEDED."
    else
      echo "$prefix Configuration of VM $vm_name FAILED.  Log file will be uploaded as artifact $conf_log_file. "
      exit 1
    fi

    echo "$prefix Test batch $batch STARTING (sending logs to $log_file)..."
    run_batch "$my_dir/on-test-vm" "$batch" "$vm_name" "$log_file"
    rc=$?
    if [ $rc = 0 ]; then
      echo "$prefix Test batch $batch SUCCEEDED."
    else
      mv "$log_file" "$failed_log_file"
      echo "$prefix Test batch $batch FAILED.  Log file will be uploaded as artifact $failed_log_file."
      if [ -n "${GCS_WORKFLOW_DIR:-}" ] && [ -n "${COMPONENT:-}" ]; then
        failed_log_upload_dir="${GCS_WORKFLOW_DIR}/${COMPONENT}/failed-fv-logs"
        if [ -n "${JOB_TAG:-}" ]; then
          failed_log_upload_dir="${failed_log_upload_dir}/${JOB_TAG}"
        fi
        gcloud storage cp "$failed_log_file" "${failed_log_upload_dir}/" || true
      fi
    fi

    # The SSH-based teardown steps below are wrapped in 'timeout': if the VM
    # wedged (e.g. a kernel BPF/vmap teardown deadlock that leaves it
    # SSH-unreachable), these would otherwise hang until the watchdog kills the
    # whole batch.  Bounding them lets a batch whose tests already finished
    # proceed straight to the (control-plane) VM delete.  gcloud delete does not
    # need SSH, so it reclaims a wedged VM regardless.
    collect_log_file="$artifacts_dir/collect-artifacts-$batch_formatted.log"
    if timeout 180 "${ssh_cmd[@]}" COMPONENT="${COMPONENT}" "${CALICO_DIR_NAME}/.semaphore/collect-artifacts" >& "$collect_log_file"; then
      echo "$prefix Remote artifact collection SUCCEEDED"
    else
      echo "$prefix Remote artifact collection FAILED (rc=$?)"
    fi
    if timeout 180 scp "${SSH_OPTIONS[@]}" -r -C "ubuntu@${vm_ip}:${CALICO_DIR_NAME}/artifacts" "${repo_dir}/artifacts/${batch}" >> "$collect_log_file" 2>&1; then
      echo "$prefix Artifact retrieval SUCCEEDED"
    else
      echo "$prefix Artifact retrieval FAILED (rc=$?)"
    fi

    echo "$prefix Deleting test VM $vm_name"
    if timeout 300 gcloud --quiet beta compute instances delete "$vm_name" --zone="${zone}" --no-graceful-shutdown; then
      echo "$prefix Deletion of test VM $vm_name SUCCEEDED"
    else
      echo "$prefix Deletion of test VM $vm_name FAILED, will retry at end of job. "
    fi

    # Mark the batch fully complete so the watchdog skips snapshot/kill for it.
    touch "$done_marker"
    exit $rc
  ) &
  pid=$!

  log_files+=( "$log_file" )
  test_pid+=( "$pid" )

  # Start a per-batch watchdog that kills this batch before the Semaphore job
  # timeout.  The watchdog only acts if the batch is genuinely still running
  # (checked via a marker file, see below), so a batch that finished early but
  # hasn't been reaped yet by the sequential wait loop does not get a spurious
  # timeout report.  Each batch also kills its own watchdog when it is reaped
  # (in the wait loop below), so there is no PID-reuse race.
  if [ "$watchdog_seconds" -gt 0 ]; then
  (
    sleep "$watchdog_seconds"

    # Distinguish a genuinely hung batch from one that has merely finished but
    # not yet been reaped by the sequential loop below.  We use a marker file
    # rather than 'kill -0 $pid' because a finished-but-unreaped batch is a
    # zombie whose pid still exists, so kill -0 would wrongly report it alive
    # and we would emit a spurious timeout report for a batch that passed.
    batch_formatted="$(format_batch_for_log "$batch")"
    done_marker="$artifacts_dir/.batch-${batch_formatted}.done"
    if [ -f "$done_marker" ]; then
      exit 0
    fi

    echo
    echo "===== WATCHDOG: batch $batch approaching job timeout (${JOB_TIMEOUT_MINUTES}m) ====="

    # 1. Serial console via the GCP control plane.  This does not use the VM's
    # network/SSH path, so it still works when the VM is wedged and SSH would
    # return rc=255 -- exactly the case that matters most (kernel soft-lockup /
    # OOM / oops).  The kernel logs to ttyS0 (console=ttyS0), so a hang shows up
    # here even when the in-VM snapshot below cannot connect.
    serial_log="$artifacts_dir/vm-serial-$batch_formatted.log"
    echo "  Capturing VM serial console to $serial_log"
    timeout 60 gcloud --quiet compute instances get-serial-port-output "$vm_name" \
        --zone="$zone" --port=1 \
        > "$serial_log" 2>&1 \
        || echo "  Serial console capture failed/timed out (rc=$?)"

    # 2. SSH-based in-VM snapshot (ps/bpftool/bpffs/dmesg/kernel stacks).  Useful
    # when the VM is merely slow rather than wedged.  Wrapped in 'timeout' so a
    # hung VM/SSH cannot extend the kill.
    snapshot_log="$artifacts_dir/vm-snapshot-$batch_formatted.log"
    echo "  Capturing pre-kill VM snapshot to $snapshot_log"
    timeout 60 env VM_NAME="$vm_name" ZONE="$zone" "$my_dir/on-test-vm" \
        bash "$CALICO_DIR_NAME/.semaphore/vms/vm-snapshot" \
        > "$snapshot_log" 2>&1 \
        || echo "  VM snapshot capture failed/timed out (rc=$?)"

    # Push the diagnostics immediately.  A job killed at its timeout may never
    # run publish-artifacts (or have it truncated mid-upload), and these files
    # sort after test-*.log alphabetically, so they would be the first dropped.
    # Pushing by basename from $artifacts_dir keeps the remote path tidy.
    # The watchdog is always the first pusher of these files (publish-artifacts
    # runs later in the epilogue), so no --force/overwrite is needed here.
    for diag in "$serial_log" "$snapshot_log"; do
      [ -s "$diag" ] || continue
      ( cd "$artifacts_dir" && artifact push job "$(basename "$diag")" ) \
          >/dev/null 2>&1 || true
    done

    echo "  Killing hung batch $batch (pid $pid)"
    kill -TERM "-$pid" 2>/dev/null || true

    # Generate a JUnit XML report for the timed-out batch so it shows up
    # in Semaphore test results as a clear failure.
    if [ "$batch" = "ut" ]; then
      report_type="ut"
    else
      report_type="fv"
    fi
    report_dir="$artifacts_dir/$batch/report"
    mkdir -p "$report_dir"
    cat > "$report_dir/felix_${report_type}_timeout_${batch_formatted}.xml" <<JUNIT_EOF
<?xml version="1.0" encoding="UTF-8"?>
<testsuites>
  <testsuite name="batch-${batch}" tests="1" failures="1" errors="0" time="$watchdog_seconds">
    <testcase name="batch ${batch} completion" classname="ci.watchdog" time="$watchdog_seconds">
      <failure message="Batch ${batch} timed out after ${watchdog_seconds}s (job timeout ${JOB_TIMEOUT_MINUTES}m)">
The watchdog killed batch ${batch} because it was still running when the job
approached its ${JOB_TIMEOUT_MINUTES}-minute timeout. Check the batch log file
for details on which test was hung.
      </failure>
    </testcase>
  </testsuite>
</testsuites>
JUNIT_EOF
    echo "==============================================================="
  ) &
  watchdog_pids+=( "$!" )
  else
    watchdog_pids+=( "" )
  fi

  (
    # Redirect tail's stdin from /dev/null to prevent waiting
    # forever on reads in semaphore CI
    tail -f --retry "$log_file" < /dev/null | \
      grep --line-buffered --perl "${monitor_pattern}" -B 2 -A 15 | \
      sed 's/.*/'"${prefix}"' &/' | \
      grep --perl --line-buffered -v '^\[batch=[^\]]+\]\s+$';
  ) &
  mon_pid=$!
  monitor_pids+=( "$mon_pid" )
done

final_result=0

# Give the batches time to emit their start-up logs.
sleep 5
echo
echo "===== Waiting for background test runners to finish ===="
echo

summary=()
for i in "${!batches[@]}"; do
  batch=${batches[$i]}
  pid=${test_pid[$i]}
  if wait "$pid"; then
    summary+=( "Test batch $batch SUCCEEDED" )
  else
    batch_formatted="$(format_batch_for_log "$batch")"
    failed_summary_log="$artifacts_dir/test-$batch_formatted-FAILED.log"
    if [ ! -f "$failed_summary_log" ] && [ -f "$artifacts_dir/test-$batch_formatted.log" ]; then
      # The batch failed without renaming its own log.  A batch the watchdog
      # kills never reaches that rename, because the kill takes out the process
      # group that would have done it.  Rename it here instead: the -FAILED
      # suffix is how analyze-test-failures discovers which batches to analyse,
      # so a log left under its running name gets no fv-tests-guru diagnosis and
      # no DIAGS artifact, and the failure reaches CI reporting as nothing but
      # the summary line below.  Safe at this point: we are past `wait`, so the
      # batch is finished and nothing is still writing to the log.
      mv "$artifacts_dir/test-$batch_formatted.log" "$failed_summary_log" ||
        failed_summary_log="$artifacts_dir/test-$batch_formatted.log"
      # Reaching here means the batch never ran its own failure handler, so the
      # log stops wherever the batch was killed.  Say so in the log itself: it
      # is now an input to fv-tests-guru, and a truncated transcript otherwise
      # reads as a test that simply stopped emitting.  Best-effort under the
      # script's `set -e`: a full disk is one of the things that gets a batch
      # killed in the first place, and losing the marker must not cost us the
      # summary, the failure exit code, or the watchdog cleanup below.
      printf '\n===== Batch %s did not complete: killed before it could finish, so this log is truncated mid-run =====\n' \
        "$batch" >> "$failed_summary_log" ||
        echo "  Could not mark $failed_summary_log as truncated (rc=$?)"
    fi
    summary+=( "Test batch $batch FAILED; Log file will be uploaded as artifact $failed_summary_log" )
    final_result=1
  fi
  # Cancel this batch's watchdog now that it has finished, preventing
  # the PID-reuse race that a single global watchdog would have.
  wd_pid=${watchdog_pids[$i]}
  if [ -n "$wd_pid" ]; then
    kill "$wd_pid" 2>/dev/null || true
    wait "$wd_pid" 2>/dev/null || true
  fi
done

echo
echo "===== Shutting down test monitors ====="
for pid in "${monitor_pids[@]}"; do
  # Note: negative PID to kill the entire process group.
  kill -TERM "-$pid" || true
done

echo "===== Results summary ====="
for s in "${summary[@]}"; do
  echo "  $s"
done
echo

echo "===== Done, exiting with RC=$final_result ====="

exit $final_result
