#!/usr/bin/env bash
# =============================================================================
#  lib/common.sh - small helpers shared by start-chrony.sh and test-chrony.sh
# =============================================================================
#  This file is "sourced" (loaded) by the other scripts; it is not run alone.
# =============================================================================

KIT_DIR="$(cd -- "$(dirname -- "${BASH_SOURCE[0]}")/.." && pwd)"

# If any command fails unexpectedly, say where, instead of exiting silently.
set -E
trap 'printf "\033[1;31m[ERROR ]\033[0m unexpected failure in %s line %s: %s\n" \
      "${BASH_SOURCE[0]##*/}" "$LINENO" "$BASH_COMMAND" >&2' ERR

# Load all tunable values (port, test sizes, targets ...).
# shellcheck source=../settings.conf
source "$KIT_DIR/settings.conf"

NTP_LOAD="$KIT_DIR/lib/ntpload.py"      # the NTP load generator (Python 3, standard library only)


# ----------------------------------------------------------------------------
#  Printing messages
# ----------------------------------------------------------------------------
log()  { printf '\033[1;34m[ INFO ]\033[0m %s\n' "$*"; }
ok()   { printf '\033[1;32m[  OK  ]\033[0m %s\n' "$*"; }
warn() { printf '\033[1;33m[ WARN ]\033[0m %s\n' "$*" >&2; }
die()  { printf '\033[1;31m[ERROR ]\033[0m %s\n' "$*" >&2; exit 1; }


# ----------------------------------------------------------------------------
#  Checks
# ----------------------------------------------------------------------------
require_root() {
    if [[ $EUID -ne 0 ]]; then
        die "Please run as root, for example:  sudo $0 $*"
    fi
}

# True when systemd manages this machine (normal RHEL host).
# False inside a plain container, where chronyd is started directly.
has_systemd() {
    [[ -d /run/systemd/system ]]
}

# True when SELinux is switched on (enforcing or permissive).
selinux_is_on() {
    command -v selinuxenabled >/dev/null && selinuxenabled
}

chrony_version() {
    rpm -q chrony 2>/dev/null || echo "chrony"
}


# ----------------------------------------------------------------------------
#  Talking to the test server
# ----------------------------------------------------------------------------

# Send one NTP request to the test server. Prints the answer as
# "key = value" lines (stratum, leap, offset_us ...). Fails if no answer.
ntp_probe() {
    python3 "$NTP_LOAD" probe --server "$LISTEN_ADDRESS" --port "$NTP_PORT"
}

ntp_answers() {
    ntp_probe >/dev/null 2>&1
}

# Wait (up to $1 seconds, default 30) until the test server answers.
wait_until_ntp_answers() {
    local timeout_seconds="${1:-30}"
    python3 "$NTP_LOAD" wait --server "$LISTEN_ADDRESS" --port "$NTP_PORT" \
        --timeout "$timeout_seconds" >/dev/null
}

# chronyc, connected to the TEST chronyd (not the normal one), through its
# own command socket. Example:  test_chronyc serverstats
# "-n" = show client addresses as numbers. Without it chronyc looks up the
# name of every client in DNS, which takes minutes on an offline network.
test_chronyc() {
    chronyc -n -h "$TEST_SOCKET" "$@"
}


# ----------------------------------------------------------------------------
#  The test configuration file
# ----------------------------------------------------------------------------
#  write_test_config [extra line]
#  Writes /etc/chrony-perf-test.conf. The "ratelimit" test calls it with
#  RATELIMIT_LINE as the extra line, and afterwards without it again.
write_test_config() {
    local extra_line="${1:-}"
    cat > "$TEST_CONF" <<EOF
# Created by $KIT_DIR - "./start-chrony.sh cleanup" removes it.
# Test NTP server of the Chrony performance kit.      See: man 5 chrony.conf
# It answers on $LISTEN_ADDRESS port $NTP_PORT only, and it never changes the
# system clock (chronyd is started with -x). /etc/chrony.conf is not used.

# --- Where it answers, and whom ---
port $NTP_PORT
bindaddress $LISTEN_ADDRESS
allow 127.0.0.0/8

# --- Time source ---
# An offline network has no internet time servers: serve this machine's
# own clock, and tell clients it is stratum $LOCAL_STRATUM.
local stratum $LOCAL_STRATUM

# --- Memory for client records (rate limiting, "chronyc clients") ---
clientloglimit $CLIENT_LOG_LIMIT

# --- Own files, so it never meets the normal chronyd ---
cmdport 0
bindcmdaddress $TEST_SOCKET
pidfile $TEST_PIDFILE
EOF
    if [[ -n $extra_line ]]; then
        printf '\n# --- Added for one test ---\n%s\n' "$extra_line" >> "$TEST_CONF"
    fi
    command -v restorecon >/dev/null && restorecon "$TEST_CONF"
    return 0
}


# ----------------------------------------------------------------------------
#  Starting and stopping the test chronyd
# ----------------------------------------------------------------------------
#  With systemd:    systemctl start/stop chronyd-perf-test
#  Without systemd: chronyd is started directly and stopped via its PID file.

# Process ID of the running test chronyd, or nothing.
test_chronyd_pid() {
    local pid
    pid="$(cat "$TEST_PIDFILE" 2>/dev/null || true)"
    if [[ -n $pid && -d /proc/$pid && $(cat "/proc/$pid/comm" 2>/dev/null) == chronyd ]]; then
        echo "$pid"
    fi
}

test_chronyd_is_running() {
    [[ -n $(test_chronyd_pid) ]]
}

# The full chronyd command line. It is used by the systemd service and,
# without systemd, run directly.
#   -x   never change the system clock (the normal chronyd keeps doing that)
#   -f   read the test configuration instead of /etc/chrony.conf
chronyd_command() {
    echo "/usr/sbin/chronyd $CHRONYD_OPTIONS -x -f $TEST_CONF"
}

chronyd_start() {
    if has_systemd; then
        systemctl start "$TEST_SERVICE"
    else
        # chronyd puts itself in the background and writes its PID file.
        # shellcheck disable=SC2046   # the command line is split on purpose
        $(chronyd_command) || die "chronyd did not start. Try it by hand:  $(chronyd_command) -d"
    fi
}

chronyd_stop() {
    if has_systemd; then
        systemctl stop "$TEST_SERVICE"
    else
        local pid waited=0
        pid="$(test_chronyd_pid)"
        [[ -n $pid ]] || return 0
        kill "$pid"
        # Wait (up to 10 s) until the process is really gone and the port is free.
        while [[ -d /proc/$pid ]] && (( waited < 100 )); do
            sleep 0.1
            waited=$(( waited + 1 ))
        done
    fi
}

chronyd_restart() {
    chronyd_stop
    chronyd_start
}


# ----------------------------------------------------------------------------
#  CPU and memory of the test chronyd
# ----------------------------------------------------------------------------

# CPU seconds (user + system) the test chronyd has used so far.
# Fields 14 and 15 of /proc/<pid>/stat, counted in "clock ticks".
chronyd_cpu_seconds() {
    local pid
    pid="$(test_chronyd_pid)"
    [[ -n $pid ]] || { echo "n/a"; return; }
    awk -v hz="$(getconf CLK_TCK)" '{ printf "%.3f", ($14 + $15) / hz }' "/proc/$pid/stat"
}

# Resident memory (RAM really used) of the test chronyd, in kB.
chronyd_rss_kb() {
    local pid
    pid="$(test_chronyd_pid)"
    [[ -n $pid ]] || { echo 0; return; }
    awk '$1 == "VmRSS:" { print $2 }' "/proc/$pid/status"
}
