#!/usr/bin/env bash
# =============================================================================
#  test-kerberos.sh - measure Kerberos server (KDC + kadmind) performance, one
#                     metric at a time, and save an easy-to-read report
# =============================================================================
#
#  USAGE (as root, after ./start-kerberos.sh)
#      ./test-kerberos.sh                  run every test (about 4 minutes)
#      ./test-kerberos.sh as               run one test
#      ./test-kerberos.sh latency cpu      run several tests
#      ./test-kerberos.sh --list           show the test names
#
#      DURATION=30 CLIENTS=16 ./test-kerberos.sh as
#                                          override settings.conf for one run
#
#  THE TESTS  (explained in detail in PERFORMANCE-METRICS.md)
#      startup      time from "start KDC" until the first ticket is handed out
#      as           logins per second (AS requests: TGTs), at full speed
#      tgs          service tickets per second (TGS requests), at full speed
#      latency      time a client waits for a ticket, at a steady normal load
#      concurrency  how throughput and waiting time change with more clients
#      tcp          logins per second when clients use TCP instead of UDP
#      kadmin       administration operations per second (add, read,
#                   change key, delete a principal) through kadmind
#      dump         time to dump the whole database (backup / replication)
#      cpu          CPU used by the KDC at the steady load
#      memory       memory used by the KDC, idle and under load
#
#  OUTPUT
#      results/<date>-<time>/report.md    the human-readable report
#      results/<date>-<time>/summary.csv  one line per result (for spreadsheets)
#      results/<date>-<time>/raw/         unmodified measurement output
#      results/latest                     link to the newest result folder
#
#  EXIT CODE
#      0 = every test passed,  1 = at least one FAIL,  2 = could not run
#
#  Load is generated by lib/krb5-load.py, which uses the system's own
#  Kerberos library (like kinit). All traffic stays on 127.0.0.1.
# =============================================================================

set -euo pipefail
source "$(dirname -- "${BASH_SOURCE[0]}")/lib/common.sh"

ALL_TESTS=(startup as tgs latency concurrency tcp kadmin dump cpu memory)

SAMPLE_INTERVAL_SECONDS=1       # how often CPU / memory are sampled
CLOCK_TICKS_PER_SECOND="$(getconf CLK_TCK)"
GENERATOR_BUSY_PCT=90           # client processes above this = they may be the limit

# Filled in by the functions below.
RESULT_DIR=""
RAW_DIR=""
DETAILS_FILE=""
SUMMARY_ROWS=()     # "Test|Measured|Target|Verdict"
FAIL_COUNT=0
GENERATOR_LIMITED_TESTS=()


# =============================================================================
#  PART 1 - small helpers
# =============================================================================

# Floating-point math and comparisons (bash itself only knows whole numbers).
calc()          { awk "BEGIN { printf \"%.1f\", $* }"; }
calc2()         { awk "BEGIN { printf \"%.2f\", $* }"; }
is_at_most()    { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a <= b) }'; }
is_at_least()   { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a >= b) }'; }
now_seconds()   { date +%s.%N; }

# Turn a comparison into the words PASS / FAIL.
verdict_at_most()  { if is_at_most  "$1" "$2"; then echo PASS; else echo FAIL; fi; }
verdict_at_least() { if is_at_least "$1" "$2"; then echo PASS; else echo FAIL; fi; }

# Remember one line for the summary table at the top of the report.
#   record_result "Test name" "measured value" "target" PASS|FAIL|INFO
record_result() {
    local name="$1" measured="$2" target="$3" verdict="$4"
    SUMMARY_ROWS+=("$name|$measured|$target|$verdict")
    [[ $verdict == FAIL ]] && FAIL_COUNT=$(( FAIL_COUNT + 1 ))

    local colour='1;32'                       # green  = PASS
    [[ $verdict == FAIL ]] && colour='1;31'   # red    = FAIL
    [[ $verdict == INFO ]] && colour='1;36'   # cyan   = INFO
    printf "    => \033[${colour}m%-4s\033[0m  %s   (target: %s)\n" "$verdict" "$measured" "$target"
}

# Append Markdown text (read from stdin) to the "details" part of the report.
add_details() {
    cat >> "$DETAILS_FILE"
}

section_title() {
    echo
    printf '\033[1m==> %s\033[0m\n' "$*"
}


# =============================================================================
#  PART 2 - running the load generator and reading its output
# =============================================================================
#
#  run_load <name> <mode> [generator options ...]
#
#  Runs lib/krb5-load.py once and saves:
#      raw/<name>.txt              all results, as "key = value" lines
#      raw/<name>-percentiles.csv  latency for every percentile 1..100
#
#  Read single values afterwards with:  value <key>
#  Example:  run_load as-full-speed as --clients 8 --duration 10
#            value as_per_s          ->  1765.687
# -----------------------------------------------------------------------------
LOAD_OUTPUT=""

run_load() {
    local name="$1"
    shift
    LOAD_OUTPUT="$RAW_DIR/$name.txt"

    printf '    %-30s ' "$name"
    if ! run_load_generator "$@" \
            --output "$LOAD_OUTPUT" \
            --percentiles "$RAW_DIR/$name-percentiles.csv" \
            2> "$RAW_DIR/$name-errors.txt"; then
        echo "failed"
        die "The load generator failed. See $RAW_DIR/$name-errors.txt"
    fi
    [[ -s $RAW_DIR/$name-errors.txt ]] || rm -f "$RAW_DIR/$name-errors.txt"

    printf '%10.0f requests/s, %s%% failed\n' "$(value rate_per_s)" "$(calc2 "$(value failed_pct)")"

    # Note it when the client processes were almost never waiting for the
    # KDC: then the result shows the limit of the load generator, not of the KDC.
    if is_at_least "$(value generator_busy_pct)" "$GENERATOR_BUSY_PCT"; then
        GENERATOR_LIMITED_TESTS+=("$name")
        warn "The client processes were ${GENERATOR_BUSY_PCT}%+ busy in '$name'; the KDC may be faster than measured."
    fi
}

# One value from the last run_load output.  value <key>  (missing -> "0")
value() {
    awk -v key="$1" '$1 == key { print $3; found = 1 } END { if (!found) print 0 }' "$LOAD_OUTPUT"
}

# A latency value, rounded to 2 decimals.
ms() {
    calc2 "$(value "$1")"
}

# The error messages of the last run, as Markdown list lines (or "none").
error_list() {
    local kinds
    kinds="$(value error_kinds)"
    if (( kinds == 0 )); then
        echo "- none"
        return
    fi
    awk -F' = ' '
        $1 ~ /^error_count_/ { count = $2 }
        $1 ~ /^error_text_/  { printf "- %s x \"%s\"\n", count, $2 }
    ' "$LOAD_OUTPUT"
}


# =============================================================================
#  PART 3 - measuring the KDC process itself (CPU, memory)
# =============================================================================
#  With worker processes (KDC_WORKERS > 0) the KDC is several processes;
#  their values are added up.

# CPU time (seconds) that all KDC processes have used so far.
# Fields 14 and 15 of /proc/<pid>/stat are user and system CPU ticks.
kdc_cpu_seconds() {
    local pid ticks=0
    for pid in $(daemon_pids krb5kdc); do
        ticks=$(( ticks + $(awk '{ print $14 + $15 }' "/proc/$pid/stat" 2>/dev/null || echo 0) ))
    done
    calc2 "$ticks / $CLOCK_TICKS_PER_SECOND"
}

# Memory in RAM ("resident set size") of all KDC processes, in MB.
kdc_memory_mb() {
    local pid kb=0
    for pid in $(daemon_pids krb5kdc); do
        kb=$(( kb + $(awk '/^VmRSS:/ { print $2 }' "/proc/$pid/status" 2>/dev/null || echo 0) ))
    done
    calc "$kb / 1024"
}

kdc_log_bytes() {
    stat -c %s "$KDC_LOG" 2>/dev/null || echo 0
}

# Runs in the background: writes one CSV line per second until killed.
sampler_loop() {
    local csv_file="$1"
    local start previous_time previous_cpu now cpu

    echo "elapsed_s,kdc_cpu_percent_of_one_core,kdc_memory_mb" > "$csv_file"
    start="$(now_seconds)"
    previous_time="$start"
    previous_cpu="$(kdc_cpu_seconds)"

    while true; do
        sleep "$SAMPLE_INTERVAL_SECONDS"
        now="$(now_seconds)"
        cpu="$(kdc_cpu_seconds)"
        printf '%s,%s,%s\n' \
            "$(calc "$now - $start")" \
            "$(calc "($cpu - $previous_cpu) / ($now - $previous_time) * 100")" \
            "$(kdc_memory_mb)" >> "$csv_file"
        previous_time="$now"
        previous_cpu="$cpu"
    done
}

# -----------------------------------------------------------------------------
#  run_steady_load
#
#  One run of the "normal day" load: LOGIN_RATE users log in per second and
#  each asks for TGS_PER_LOGIN service tickets, while the KDC's CPU and memory
#  are sampled every second. The latency, cpu and memory tests all read from
#  this same run, so when you run more than one of them the load is generated
#  only once.
#
#  Sets: STEADY_OUTPUT (generator results file)
#        STEADY_CPU_AVG_PCT  STEADY_CPU_PEAK_PCT  STEADY_CPU_SECONDS
#        STEADY_MEM_IDLE_MB  STEADY_MEM_PEAK_MB  STEADY_LOG_BYTES  STEADY_SECONDS
# -----------------------------------------------------------------------------
STEADY_LOAD_DONE=no

run_steady_load() {
    if [[ $STEADY_LOAD_DONE == yes ]]; then
        LOAD_OUTPUT="$STEADY_OUTPUT"
        return
    fi
    local samples="$RAW_DIR/resource-samples.csv"

    # Values before the load starts.
    STEADY_MEM_IDLE_MB="$(kdc_memory_mb)"
    local cpu_before time_before log_before
    cpu_before="$(kdc_cpu_seconds)"
    log_before="$(kdc_log_bytes)"
    time_before="$(now_seconds)"

    sampler_loop "$samples" &
    local sampler_pid=$!

    run_load "steady-load-${LOGIN_RATE}-logins-per-s" mix \
        --rate "$LOGIN_RATE" --tgs-per-login "$TGS_PER_LOGIN" \
        --clients "$CLIENTS" --duration "$DURATION"
    STEADY_OUTPUT="$LOAD_OUTPUT"

    local cpu_after time_after
    cpu_after="$(kdc_cpu_seconds)"
    time_after="$(now_seconds)"
    kill "$sampler_pid" 2>/dev/null || true
    wait "$sampler_pid" 2>/dev/null || true

    STEADY_SECONDS="$(calc2 "$time_after - $time_before")"
    STEADY_CPU_SECONDS="$(calc2 "$cpu_after - $cpu_before")"
    STEADY_CPU_AVG_PCT="$(calc "$STEADY_CPU_SECONDS / $STEADY_SECONDS * 100")"
    STEADY_CPU_PEAK_PCT="$(awk -F, 'NR > 1 && $2 > m { m = $2 } END { printf "%.1f", m }' "$samples")"
    STEADY_MEM_PEAK_MB="$(awk -F, -v m="$STEADY_MEM_IDLE_MB" 'NR > 1 && $3 > m { m = $3 } END { printf "%.1f", m }' "$samples")"
    STEADY_LOG_BYTES=$(( $(kdc_log_bytes) - log_before ))

    STEADY_LOAD_DONE=yes
}


# =============================================================================
#  PART 4 - the tests, one function per metric
# =============================================================================

# -----------------------------------------------------------------------------
test_startup() {
    section_title "Startup time  ($STARTUP_ROUNDS restarts)"
    local round start_time success_time elapsed_ms times=()

    for (( round = 1; round <= STARTUP_ROUNDS; round++ )); do
        kdc_stop

        # Start the probe BEFORE the daemon: it tries to log in every 10 ms
        # and notes the exact moment the first login works.
        local probe_output="$RAW_DIR/startup-round-$round.txt"
        local ready_file="$RAW_DIR/.probe-ready"
        rm -f "$ready_file"
        run_load_generator probe --timeout 60 --ready-file "$ready_file" \
            --output "$probe_output" 2>/dev/null &
        local probe_pid=$!
        while [[ ! -e $ready_file ]] && kill -0 "$probe_pid" 2>/dev/null; do
            sleep 0.01
        done
        rm -f "$ready_file"

        start_time="$(now_seconds)"
        kdc_start
        if ! wait "$probe_pid"; then
            die "The KDC did not hand out a ticket within 60 s after restart."
        fi

        success_time="$(awk '$1 == "first_success_epoch" { print $3 }' "$probe_output")"
        elapsed_ms="$(calc "($success_time - $start_time) * 1000")"
        times+=("$elapsed_ms")
        printf '    restart %d: %s ms\n' "$round" "$elapsed_ms"
    done

    local average fastest slowest
    average="$(printf '%s\n' "${times[@]}" | awk '{ s += $1 } END { printf "%.1f", s / NR }')"
    fastest="$(printf '%s\n' "${times[@]}" | sort -n | head -1)"
    slowest="$(printf '%s\n' "${times[@]}" | sort -n | tail -1)"

    record_result "Startup time" "$average ms average" "<= $TARGET_STARTUP_MAX_MS ms" \
        "$(verdict_at_most "$average" "$TARGET_STARTUP_MAX_MS")"

    add_details <<EOF
## Startup time

The KDC was stopped and started $STARTUP_ROUNDS times. Each time was measured
from the start command until a client had its first ticket (a full login).

| Round | Time (ms) |
|------:|----------:|
$(for i in "${!times[@]}"; do printf '| %5d | %9s |\n' $(( i + 1 )) "${times[$i]}"; done)

Average **$average ms**, fastest $fastest ms, slowest $slowest ms.

EOF
}

# -----------------------------------------------------------------------------
test_as() {
    section_title "Logins per second (AS)  ($CLIENTS clients, full speed)"
    run_load "as-full-speed" as --clients "$CLIENTS" --duration "$DURATION"
    AS_UDP_RATE="$(calc "$(value as_per_s)")"      # the tcp test compares with this

    record_result "Logins (AS requests)" "$AS_UDP_RATE logins/s" \
        ">= $TARGET_AS_MIN_RPS logins/s" \
        "$(verdict_at_least "$AS_UDP_RATE" "$TARGET_AS_MIN_RPS")"

    add_details <<EOF
## Logins per second (AS requests)

$CLIENTS clients logged in again and again for $DURATION seconds, each one as
soon as its previous login was done. One login is what "kinit" does: the user
proves its identity (pre-authentication) and receives a ticket-granting
ticket (TGT). It takes two request/answer round trips with the KDC.

| Item | Value |
|---|---:|
| Logins per second | **$AS_UDP_RATE** |
| Logins done | $(value as_ok) |
| Failed logins | $(value requests_failed) |
| Average time per login | $(ms as_avg_ms) ms |
| p95 / p99 | $(ms as_p95_ms) ms / $(ms as_p99_ms) ms |
| Client processes busy | $(value generator_busy_pct)% |

Errors:

$(error_list)

Raw output: \`raw/as-full-speed.txt\`

EOF
}

# -----------------------------------------------------------------------------
test_tgs() {
    section_title "Service tickets per second (TGS)  ($CLIENTS clients, full speed)"
    run_load "tgs-full-speed" tgs --clients "$CLIENTS" --duration "$DURATION"

    local rate
    rate="$(calc "$(value tgs_per_s)")"

    record_result "Service tickets (TGS)" "$rate tickets/s" \
        ">= $TARGET_TGS_MIN_RPS tickets/s" \
        "$(verdict_at_least "$rate" "$TARGET_TGS_MIN_RPS")"

    add_details <<EOF
## Service tickets per second (TGS requests)

Each of the $CLIENTS clients logged in once (not measured) and then asked for
service tickets (for HTTP/web0001.perf.test ... HTTP/web$(printf '%04d' "$TEST_SERVICES").perf.test)
again and again for $DURATION seconds. This is what happens each time a user
opens a Kerberos-protected web site, file share or SSH server. One service
ticket is one request/answer round trip.

| Item | Value |
|---|---:|
| Service tickets per second | **$rate** |
| Tickets handed out | $(value tgs_ok) |
| Failed requests | $(value requests_failed) |
| Average time per ticket | $(ms tgs_avg_ms) ms |
| p95 / p99 | $(ms tgs_p95_ms) ms / $(ms tgs_p99_ms) ms |
| Client processes busy | $(value generator_busy_pct)% |

Errors:

$(error_list)

EOF
}

# -----------------------------------------------------------------------------
test_latency() {
    section_title "Latency  ($LOGIN_RATE logins/s + $TGS_PER_LOGIN service tickets each)"
    run_steady_load

    # Logins and service tickets each get their own verdict (p95 and p99 must both be met).
    local login_verdict=PASS ticket_verdict=PASS
    is_at_most "$(value as_p95_ms)"  "$TARGET_AS_P95_MAX_MS"  || login_verdict=FAIL
    is_at_most "$(value as_p99_ms)"  "$TARGET_AS_P99_MAX_MS"  || login_verdict=FAIL
    is_at_most "$(value tgs_p95_ms)" "$TARGET_TGS_P95_MAX_MS" || ticket_verdict=FAIL
    is_at_most "$(value tgs_p99_ms)" "$TARGET_TGS_P99_MAX_MS" || ticket_verdict=FAIL

    record_result "Latency, login p95 / p99" "$(ms as_p95_ms) ms / $(ms as_p99_ms) ms" \
        "<= $TARGET_AS_P95_MAX_MS ms / <= $TARGET_AS_P99_MAX_MS ms" "$login_verdict"
    record_result "Latency, ticket p95 / p99" "$(ms tgs_p95_ms) ms / $(ms tgs_p99_ms) ms" \
        "<= $TARGET_TGS_P95_MAX_MS ms / <= $TARGET_TGS_P99_MAX_MS ms" "$ticket_verdict"

    local failed_pct
    failed_pct="$(awk -v x="$(value failed_pct)" 'BEGIN { printf "%.3f", x }')"
    record_result "Errors at steady load" "$failed_pct% ($(value requests_failed) of $(value requests_total))" \
        "<= $TARGET_ERRORS_MAX_PCT%" \
        "$(verdict_at_most "$failed_pct" "$TARGET_ERRORS_MAX_PCT")"

    local total_rate
    total_rate="$(calc "$LOGIN_RATE * (1 + $TGS_PER_LOGIN)")"

    add_details <<EOF
## Latency (time to get a ticket) and errors

A steady "normal day" load for $DURATION seconds: $LOGIN_RATE users logged in
per second, and each then asked for $TGS_PER_LOGIN service tickets. That is
$total_rate requests per second in total. "p95 = 3 ms" means 95 of every 100
requests were answered within 3 ms.

| Statistic | Login (AS), ms | Service ticket (TGS), ms |
|---|---:|---:|
| Average      | $(ms as_avg_ms) | $(ms tgs_avg_ms) |
| p50 (median) | $(ms as_p50_ms) | $(ms tgs_p50_ms) |
| p90          | $(ms as_p90_ms) | $(ms tgs_p90_ms) |
| p95          | **$(ms as_p95_ms)** | **$(ms tgs_p95_ms)** |
| p99          | **$(ms as_p99_ms)** | **$(ms tgs_p99_ms)** |
| p99.9        | $(ms as_p99_9_ms) | $(ms tgs_p99_9_ms) |
| Slowest      | $(ms as_max_ms) | $(ms tgs_max_ms) |

A login needs two round trips with the KDC (the first one only tells the client
how to prove its identity), a service ticket needs one. So a login normally
takes about twice as long.

| Item | Value |
|---|---:|
| Requests sent | $(value requests_total) |
| Logins / service tickets done | $(value as_ok) / $(value tgs_ok) |
| **Failed requests** | **$failed_pct%** |
| Logins started more than 100 ms late (the load could not keep its rate) | $(value units_late_pct)% |

Errors:

$(error_list)

Every percentile from 1 to 100: \`raw/steady-load-${LOGIN_RATE}-logins-per-s-percentiles.csv\`

EOF
}

# -----------------------------------------------------------------------------
test_concurrency() {
    section_title "Concurrency  (clients: $CLIENT_LEVELS)"
    local clients table="" peak_rate=0 peak_clients=0 last_rate=0 top_clients=0

    for clients in $CLIENT_LEVELS; do
        run_load "concurrency-${clients}-clients" as --clients "$clients" --duration "$DURATION"

        local rate
        rate="$(calc "$(value as_per_s)")"
        if is_at_least "$rate" "$peak_rate"; then
            peak_rate="$rate"
            peak_clients="$clients"
        fi
        last_rate="$rate"
        top_clients="$clients"

        table+="$(printf '| %7s | %9s | %8s | %8s | %8s | %8s | %9s |' \
            "$clients" "$rate" "$(ms as_avg_ms)" "$(ms as_p95_ms)" "$(ms as_p99_ms)" \
            "$(calc2 "$(value failed_pct)")" "$(value generator_busy_pct)")"$'\n'
    done

    local kept_pct
    kept_pct="$(awk -v a="$last_rate" -v p="$peak_rate" 'BEGIN { printf "%.1f", (p > 0 ? a / p * 100 : 0) }')"

    record_result "Concurrency (scaling)" \
        "$kept_pct% of peak at $top_clients clients (peak $peak_rate/s at $peak_clients)" \
        ">= $TARGET_SCALING_MIN_PCT% of peak" \
        "$(verdict_at_least "$kept_pct" "$TARGET_SCALING_MIN_PCT")"

    add_details <<EOF
## Concurrency (scaling)

The number of clients logging in at the same time was raised step by step,
$DURATION s per step. Each client logs in as fast as it can. When the KDC is
fully busy, more clients cannot raise the number of logins per second; they
only wait longer. A good server keeps its peak rate (it does not slow down
under pressure) and does not start to fail requests.

| Clients | Logins/s | Avg (ms) | p95 (ms) | p99 (ms) | Failed % | Clients busy % |
|--------:|---------:|---------:|---------:|---------:|---------:|---------------:|
${table}
Peak: **$peak_rate logins/s** with $peak_clients clients. With the most
clients ($top_clients) the KDC still delivered **$kept_pct%** of that peak.

"Clients busy %" is how busy the client processes were. Near 100% means the
load generator, not the KDC, set the limit at that step.

$(if (( KDC_WORKERS == 0 )); then
    echo "The KDC runs as one single process here (KDC_WORKERS=0), so it can use at"
    echo "most one CPU core. More cores are used with KDC_WORKERS in settings.conf."
  fi)

EOF
}

# -----------------------------------------------------------------------------
test_tcp() {
    section_title "Logins over TCP  ($CLIENTS clients, full speed)"
    KRB5_CONFIG="$TCP_CLIENT_CONF:$CLIENT_CONF" \
        run_load "as-over-tcp" as --clients "$CLIENTS" --duration "$DURATION"

    local rate compared=""
    rate="$(calc "$(value as_per_s)")"
    if [[ -n ${AS_UDP_RATE:-} ]] && is_at_least "$AS_UDP_RATE" 1; then
        compared="$(calc "$rate / $AS_UDP_RATE * 100")"
    fi

    record_result "Logins over TCP" "$rate logins/s${compared:+ ($compared% of UDP)}" \
        ">= $TARGET_TCP_MIN_RPS logins/s" \
        "$(verdict_at_least "$rate" "$TARGET_TCP_MIN_RPS")"

    add_details <<EOF
## Logins over TCP

The same test as "Logins per second", but every request was sent over TCP
(client setting udp_preference_limit = 1). Each round trip then needs its own
TCP connection (connect, send, receive, close), so a login opens two TCP
connections. Clients use TCP when a request is too big for UDP (for example
large tickets with many group memberships), or when a firewall blocks UDP.

| Item | Value |
|---|---:|
| Logins per second over TCP | **$rate** |
| Logins per second over UDP (test "as") | ${AS_UDP_RATE:-not run} |
| TCP compared to UDP | ${compared:-n/a}${compared:+%} |
| Average time per login | $(ms as_avg_ms) ms |
| p95 / p99 | $(ms as_p95_ms) ms / $(ms as_p99_ms) ms |
| Failed logins | $(value requests_failed) |

Errors:

$(error_list)

EOF
}

# -----------------------------------------------------------------------------
#  The kadmin test runs four batches through kadmind, each in ONE kadmin
#  session (like a script that creates many accounts):
#      add      addprinc -randkey   create a principal (new random keys)
#      read     getprinc            read a principal
#      change   cpw -randkey        give a principal new random keys
#      delete   delprinc -force     delete a principal
# -----------------------------------------------------------------------------
test_kadmin() {
    section_title "Administration through kadmind  ($KADMIN_OPERATIONS principals per operation)"
    local prefix="perfadmtest"

    # Remove leftovers of an earlier, interrupted run.
    kadmin.local -r "$REALM" -q "listprincs $prefix*" 2>/dev/null | grep "^$prefix" |
        sed 's/^/delprinc -force /' | kadmin.local -r "$REALM" >/dev/null 2>&1 || true

    local operation command success_pattern table="" slowest_rate="" slowest_name=""
    for operation in add read change delete; do
        case "$operation" in
            add)    command="addprinc -randkey -clearpolicy"; success_pattern='^Principal ".*" created\.' ;;
            read)   command="getprinc";                       success_pattern='^Principal: ' ;;
            change) command="cpw -randkey";                   success_pattern='^Key for ".*" randomized\.' ;;
            delete) command="delprinc -force";                success_pattern='^Principal ".*" deleted\.' ;;
        esac

        local output="$RAW_DIR/kadmin-$operation.txt"
        local start end seconds done_count rate
        start="$(now_seconds)"
        for (( i = 1; i <= KADMIN_OPERATIONS; i++ )); do
            printf '%s %s%05d\n' "$command" "$prefix" "$i"
        done | run_kadmin > "$output" 2>&1 || true
        end="$(now_seconds)"

        seconds="$(calc2 "$end - $start")"
        done_count="$(grep -cE "$success_pattern" "$output" || true)"
        rate="$(calc "$done_count / ($end - $start)")"
        printf '    %-8s %6s operations/s  (%s of %s done)\n' "$operation" "$rate" "$done_count" "$KADMIN_OPERATIONS"

        # An operation that did not finish every principal counts as rate 0.
        (( done_count < KADMIN_OPERATIONS )) && rate=0
        if [[ -z $slowest_rate ]] || is_at_most "$rate" "$slowest_rate"; then
            slowest_rate="$rate"
            slowest_name="$operation"
        fi
        table+="| $operation | \`$command\` | $done_count / $KADMIN_OPERATIONS | $seconds | **$rate** | $(calc2 "($end - $start) * 1000 / $KADMIN_OPERATIONS") |"$'\n'
    done

    record_result "Admin operations (kadmind)" "$slowest_rate ops/s (slowest: $slowest_name)" \
        ">= $TARGET_KADMIN_MIN_OPS ops/s" \
        "$(verdict_at_least "$slowest_rate" "$TARGET_KADMIN_MIN_OPS")"

    add_details <<EOF
## Administration operations (kadmind)

$KADMIN_OPERATIONS test principals were created, read, given new keys and
deleted through the admin server kadmind, as a script with "kadmin" would do
it. Each operation type ran as one batch in one kadmin session (one login).
Writing operations change the database on disk, so they depend on disk speed.

| Operation | kadmin command | Done | Seconds | Operations/s | ms per operation |
|---|---|---:|---:|---:|---:|
${table}
The verdict uses the slowest operation type. An operation type that did not
complete every principal counts as 0 operations/s. The kadmin output is in
\`raw/kadmin-<operation>.txt\`.

EOF
}

# -----------------------------------------------------------------------------
test_dump() {
    section_title "Database dump  ($DUMP_ROUNDS rounds)"
    local round start end times=() dump_file="$RAW_DIR/.database-dump"

    for (( round = 1; round <= DUMP_ROUNDS; round++ )); do
        start="$(now_seconds)"
        kdb5_util -r "$REALM" dump "$dump_file"
        end="$(now_seconds)"
        times+=("$(calc2 "$end - $start")")
        printf '    dump %d: %s s\n' "$round" "${times[-1]}"
    done

    local principals size_mb average rate
    principals="$(grep -c '^princ' "$dump_file" || true)"
    size_mb="$(calc2 "$(stat -c %s "$dump_file") / 1048576")"
    average="$(printf '%s\n' "${times[@]}" | awk '{ s += $1 } END { printf "%.2f", s / NR }')"
    rate="$(awk -v n="$principals" -v t="$average" 'BEGIN { printf "%.0f", (t > 0 ? n / t : 0) }')"
    rm -f "$dump_file" "$dump_file.dump_ok"     # the dump contains keys; do not keep it

    record_result "Database dump" "$average s for $principals principals" \
        "<= $TARGET_DUMP_MAX_S s" \
        "$(verdict_at_most "$average" "$TARGET_DUMP_MAX_S")"

    add_details <<EOF
## Database dump (backup and replication)

"kdb5_util dump" writes the whole principal database to a text file. It is
the first step of every backup, and of replication to replica KDCs with
"kprop" (dump, send, load on the replica). The dump was repeated
$DUMP_ROUNDS times while the KDC was running.

| Item | Value |
|---|---:|
| Principals in the database | $principals |
| Dump size | $size_mb MB |
| Time per dump (average of $DUMP_ROUNDS) | **$average s** |
| Principals per second | $rate |
| Rounds (s) | ${times[*]} |

The dump file itself is deleted after the test, because it contains all keys.

EOF
}

# -----------------------------------------------------------------------------
test_cpu() {
    section_title "CPU usage  ($LOGIN_RATE logins/s + $TGS_PER_LOGIN service tickets each)"
    run_steady_load

    local requests ms_per_1000 log_per_request
    requests="$(value requests_ok)"
    ms_per_1000="$(awk -v cpu="$STEADY_CPU_SECONDS" -v n="$requests" \
        'BEGIN { printf "%.1f", (n > 0 ? cpu * 1000 / n * 1000 : 0) }')"
    log_per_request="$(awk -v b="$STEADY_LOG_BYTES" -v n="$requests" \
        'BEGIN { printf "%.0f", (n > 0 ? b / n : 0) }')"

    record_result "CPU usage (KDC)" "$STEADY_CPU_AVG_PCT% of one core" \
        "<= $TARGET_CPU_MAX_PCT%" \
        "$(verdict_at_most "$STEADY_CPU_AVG_PCT" "$TARGET_CPU_MAX_PCT")"

    add_details <<EOF
## CPU usage

CPU used by the KDC while it handled the steady load ($LOGIN_RATE logins per
second, each with $TGS_PER_LOGIN service tickets). $(if (( KDC_WORKERS == 0 )); then
    echo "The KDC runs as one single process, so it can never use more than one"
    echo "CPU core: **100% here means the KDC is at its limit**, no matter how many"
    echo "cores the machine has."
else
    echo "The KDC runs with $KDC_WORKERS worker processes; their CPU is added up,"
    echo "so values above 100% are possible (up to ${KDC_WORKERS}00%)."
fi)

| Item | Value |
|---|---:|
| Average CPU, % of one core | **$STEADY_CPU_AVG_PCT%** |
| Busiest second, % of one core | $STEADY_CPU_PEAK_PCT% |
| Requests handled | $requests |
| CPU time per 1000 requests | $ms_per_1000 ms |
| KDC log written per request | $log_per_request bytes |

Per-second samples: \`raw/resource-samples.csv\`

EOF
}

# -----------------------------------------------------------------------------
test_memory() {
    section_title "Memory usage  ($LOGIN_RATE logins/s + $TGS_PER_LOGIN service tickets each)"
    run_steady_load

    record_result "Memory usage (KDC)" "$STEADY_MEM_PEAK_MB MB peak (idle $STEADY_MEM_IDLE_MB MB)" \
        "<= $TARGET_MEMORY_MAX_MB MB" \
        "$(verdict_at_most "$STEADY_MEM_PEAK_MB" "$TARGET_MEMORY_MAX_MB")"

    add_details <<EOF
## Memory usage

Memory in RAM (resident set size) of the KDC$( (( KDC_WORKERS > 0 )) && echo ", all processes added up").

| Item | Idle | Under load (peak) |
|---|---:|---:|
| Memory (MB) | $STEADY_MEM_IDLE_MB | **$STEADY_MEM_PEAK_MB** |

The KDC keeps no per-user state between requests, so its memory hardly grows
with load. The database is read from disk (through the page cache of the
operating system), not kept inside the KDC process.

EOF
}


# =============================================================================
#  PART 5 - preparing, and writing the final report
# =============================================================================

# Stop with exit code 2 ("could not run") and a clear message.
cannot_run() {
    printf '\033[1;31m[ERROR ]\033[0m %s\n' "$*" >&2
    exit 2
}

check_ready_to_test() {
    [[ $EUID -eq 0 ]]             || cannot_run "Please run as root, for example:  sudo $0"
    command -v python3 >/dev/null || cannot_run "'python3' not found."
    [[ -f $CLIENT_CONF ]]         || cannot_run "The test realm is not set up. Run ./start-kerberos.sh first."
    kdc_is_running                || cannot_run "The test KDC is not running. Run ./start-kerberos.sh first."
    kdc_is_answering              || cannot_run "The test KDC hands out no tickets. Run ./start-kerberos.sh first."
    kadmind_is_answering          || cannot_run "The test kadmind does not answer. Run ./start-kerberos.sh first."
}

# The server may have been started with other settings than settings.conf
# has now (for example "KDC_WORKERS=4 ./start-kerberos.sh"). Describe the
# server that really runs: read its configuration file and count its processes.
read_running_server_settings() {
    local processes library
    processes="$(daemon_pids krb5kdc | wc -l)"
    if (( processes > 1 )); then
        KDC_WORKERS=$(( processes - 1 ))     # one supervisor + N workers
    else
        KDC_WORKERS=0
    fi

    library="$(awk -F'= *' '/^ *db_library/ { print $2 }' "$KDC_CONF")"
    DB_BACKEND=db2
    [[ $library == klmdb ]] && DB_BACKEND=lmdb

    DISABLE_LAST_SUCCESS="$(awk -F'= *' '/^ *disable_last_success/ { print $2 }' "$KDC_CONF")"

    SPAKE_PREAUTH=no
    grep -q '^ *spake_preauth_kdc_challenge' "$KDC_CONF" && SPAKE_PREAUTH=yes
    return 0
}

prepare_result_folder() {
    RESULT_DIR="$KIT_DIR/results/$(date +%Y%m%d-%H%M%S)"
    RAW_DIR="$RESULT_DIR/raw"
    DETAILS_FILE="$RAW_DIR/.details.md"
    mkdir -p "$RAW_DIR"
    : > "$DETAILS_FILE"
    ln -sfn "$(basename "$RESULT_DIR")" "$KIT_DIR/results/latest"
}

# A short unmeasured run, so the first real test does not start "cold".
warm_up() {
    log "Warm-up: 2 seconds of light load (not measured)"
    run_load_generator mix --rate 50 --clients 2 --duration 2 --output "$RAW_DIR/warm-up.txt" >/dev/null 2>&1 || true
}

write_report() {
    local tests_run="$1" started="$2" finished="$3"
    local report="$RESULT_DIR/report.md"
    local os_name selinux cpu_model memory_total workers preauth

    os_name="$(. /etc/os-release && echo "$PRETTY_NAME")"
    selinux="$(getenforce 2>/dev/null || echo 'not available')"
    cpu_model="$(awk -F': ' '/^model name/ { print $2; exit }' /proc/cpuinfo)"
    memory_total="$(awk '/^MemTotal/ { printf "%.1f GB", $2 / 1048576 }' /proc/meminfo)"
    workers="single process"
    (( KDC_WORKERS > 0 )) && workers="$KDC_WORKERS worker processes"
    preauth="encrypted timestamp"
    [[ $SPAKE_PREAUTH == yes ]] && preauth="SPAKE"

    local overall="ALL PASSED"
    (( FAIL_COUNT > 0 )) && overall="$FAIL_COUNT RESULT(S) FAILED"

    {
        echo "# Kerberos Server Performance Test Report"
        echo
        echo "**Result: $overall**"
        echo
        echo "| | |"
        echo "|---|---|"
        echo "| Test started  | $started |"
        echo "| Test finished | $finished |"
        echo "| Host | $(hostname) |"
        echo "| Operating system | $os_name (kernel $(uname -r)) |"
        echo "| CPU | $(nproc) x $cpu_model |"
        echo "| Memory | $memory_total |"
        echo "| Kerberos server | MIT Kerberos $(krb5_version) (krb5kdc, kadmind) |"
        echo "| SELinux | $selinux |"
        echo "| Realm | $REALM, KDC on $KDC_ADDRESS:$KDC_PORT ($workers) |"
        echo "| Database | $DB_BACKEND, $(( TEST_USERS + TEST_SERVICES + FILLER_PRINCIPALS )) test principals; last-login writes $([[ $DISABLE_LAST_SUCCESS == true ]] && echo off || echo on) |"
        echo "| Pre-authentication | $preauth; keys: ${SUPPORTED_ENCTYPES%%:*} first |"
        echo "| Tests run | $tests_run |"
        echo "| Load per run | $DURATION s; full speed: $CLIENTS clients; steady: $LOGIN_RATE logins/s x (1 + $TGS_PER_LOGIN tickets) |"
        echo
        echo "## Summary"
        echo
        echo "PASS = target met, FAIL = target missed, INFO = measured only (no target)."
        echo "Targets are set in \`settings.conf\`. What each metric means: \`PERFORMANCE-METRICS.md\`."
        echo
        printf '| %-27s | %-50s | %-26s | %-8s |\n' "Test" "Measured" "Target" "Verdict"
        printf '|%s|%s|%s|%s|\n' "$(printf -- '-%.0s' {1..29})" "$(printf -- '-%.0s' {1..52})" \
                                 "$(printf -- '-%.0s' {1..28})" "$(printf -- '-%.0s' {1..10})"
        local row name measured target verdict
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            [[ $verdict == FAIL ]] && verdict="**FAIL**"
            printf '| %-27s | %-50s | %-26s | %-8s |\n' "$name" "$measured" "$target" "$verdict"
        done
        echo
        if (( ${#GENERATOR_LIMITED_TESTS[@]} > 0 )); then
            echo "> **Note:** the client processes were at least ${GENERATOR_BUSY_PCT}% busy in:"
            echo "> ${GENERATOR_LIMITED_TESTS[*]}."
            echo "> In those runs the KDC may be faster than the numbers show."
            echo
        fi
        echo "# Details"
        echo
        cat "$DETAILS_FILE"
        echo "## Files in this folder"
        echo
        echo "- \`report.md\` - this report"
        echo "- \`summary.csv\` - the summary table, for spreadsheets"
        echo "- \`raw/\` - full output of every load run (\"key = value\" lines),"
        echo "  latency percentile tables, kadmin output and per-second resource samples"
    } > "$report"

    # The same summary as CSV.
    {
        echo "test,measured,target,verdict"
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            printf '"%s","%s","%s","%s"\n' "$name" "$measured" "$target" "$verdict"
        done
    } > "$RESULT_DIR/summary.csv"

    rm -f "$DETAILS_FILE"
}

print_usage() {
    # Print the comment block at the top of this file.
    sed -n '3,/^set -euo pipefail/{/^#/p}' "$0" | sed 's/^# \{0,1\}//'
}


# =============================================================================
#  MAIN
# =============================================================================
main() {
    case "${1:-}" in
        -h|--help) print_usage; exit 0 ;;
        --list)    printf '%s\n' "${ALL_TESTS[@]}"; exit 0 ;;
    esac

    # Which tests to run: the names given, or all of them.
    local tests=("$@")
    if (( ${#tests[@]} == 0 )) || [[ ${tests[0]} == all ]]; then
        tests=("${ALL_TESTS[@]}")
    fi
    local test_name
    for test_name in "${tests[@]}"; do
        if [[ " ${ALL_TESTS[*]} " != *" $test_name "* ]]; then
            echo "Unknown test: '$test_name'. Valid tests: ${ALL_TESTS[*]}" >&2
            exit 2
        fi
    done

    check_ready_to_test
    read_running_server_settings
    prepare_result_folder

    # Always stop background helpers if the script is interrupted.
    trap 'kill $(jobs -p) 2>/dev/null || true' EXIT

    local started finished
    started="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    log "Testing the Kerberos server of realm $REALM on $KDC_ADDRESS - tests: ${tests[*]}"
    log "Results folder: $RESULT_DIR"
    if [[ $RESET_KDC_LOG_BEFORE_TEST == yes ]]; then
        # The KDC keeps writing to the same (now empty) file.
        log "Emptying the test KDC log first (RESET_KDC_LOG_BEFORE_TEST=yes)"
        : > "$KDC_LOG"
    fi
    warm_up

    for test_name in "${tests[@]}"; do
        "test_$test_name"
    done

    finished="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    write_report "${tests[*]}" "$started" "$finished"

    echo
    sed -n '/^## Summary/,/^# Details/p' "$RESULT_DIR/report.md" | sed '$d'
    echo
    ok "Report saved: $RESULT_DIR/report.md"

    # Exit code: 0 when nothing failed, otherwise 1.
    if (( FAIL_COUNT > 0 )); then
        exit 1
    fi
}

main "$@"
