#!/usr/bin/env bash
# =============================================================================
#  test-dns.sh - measure DNS server (BIND named) performance, one metric at
#                a time, and save an easy-to-read report
# =============================================================================
#
#  USAGE (as root, after ./start-dns.sh)
#      ./test-dns.sh                    run every test (about 3 minutes)
#      ./test-dns.sh throughput         run one test
#      ./test-dns.sh latency cache      run several tests
#      ./test-dns.sh --list             show the test names
#
#      DURATION=30 OUTSTANDING=200 ./test-dns.sh throughput
#                                       override settings.conf for one run
#
#  THE TESTS  (explained in detail in PERFORMANCE-METRICS.md)
#      startup      time from "start daemon" until the first answer
#      throughput   queries answered per second (QPS), at full speed
#      latency      response time per query at a steady, normal load
#      concurrency  how QPS and latency change as queries in flight increase
#      loss         queries lost or failed at a fixed high query rate
#      tcp          queries over TCP compared with UDP
#      cache        answers from the cache compared with forwarded lookups
#      axfr         time to transfer the whole zone (zone transfer)
#      cpu          CPU used by named under load
#      memory       memory used by named, idle and under load
#
#  OUTPUT
#      results/<date>-<time>/report.md    the human-readable report
#      results/<date>-<time>/summary.csv  one line per test (for spreadsheets)
#      results/<date>-<time>/raw/         unmodified tool output
#      results/latest                     link to the newest result folder
#
#  EXIT CODE
#      0 = every test passed,  1 = at least one FAIL,  2 = could not run
#
#  Load is generated by lib/dnsload.py (needs only Python 3 from RHEL).
#  All traffic stays on this machine; no internet access is needed.
# =============================================================================

set -euo pipefail
source "$(dirname -- "${BASH_SOURCE[0]}")/lib/common.sh"

ALL_TESTS=(startup throughput latency concurrency loss tcp cache axfr cpu memory)

DNSLOAD="$KIT_DIR/lib/dnsload.py"
SAMPLE_INTERVAL_SECONDS=1       # how often CPU / memory are sampled
CPU_COUNT="$(nproc)"

# Filled in by the functions below.
RESULT_DIR=""
RAW_DIR=""
DETAILS_FILE=""
MIXED_QUERIES=""                # query list used by most tests
SUMMARY_ROWS=()                 # "Test|Measured|Target|Verdict"
FAIL_COUNT=0
declare -A R                    # results of the latest load run, e.g. ${R[QPS]}


# =============================================================================
#  PART 1 - small helpers
# =============================================================================

# Floating-point math and comparisons (bash itself only knows whole numbers).
calc()          { awk "BEGIN { printf \"%.1f\", $* }"; }
calc3()         { awk "BEGIN { printf \"%.3f\", $* }"; }
is_at_most()    { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a <= b) }'; }
is_at_least()   { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a >= b) }'; }
now_seconds()   { date +%s.%N; }
# 1234567.8 -> 1,234,568 (rounded, with thousands separators, easier to read)
thousands()     { printf '%.0f' "$1" | sed -e ':again' -e 's/\B[0-9]\{3\}\>/,&/' -e 't again'; }

# Turn a comparison into the words PASS / FAIL.
verdict_at_most()  { if is_at_most  "$1" "$2"; then echo PASS; else echo FAIL; fi; }
verdict_at_least() { if is_at_least "$1" "$2"; then echo PASS; else echo FAIL; fi; }

# Remember one line for the summary table at the top of the report.
#   record_result "Test name" "measured value" "target" PASS|FAIL|INFO
record_result() {
    local name="$1" measured="$2" target="$3" verdict="$4"
    SUMMARY_ROWS+=("$name|$measured|$target|$verdict")
    [[ $verdict == FAIL ]] && FAIL_COUNT=$(( FAIL_COUNT + 1 ))

    local colour='1;32'                       # green  = PASS
    [[ $verdict == FAIL ]] && colour='1;31'   # red    = FAIL
    [[ $verdict == INFO ]] && colour='1;36'   # cyan   = INFO
    printf "    => \033[${colour}m%-4s\033[0m  %s   (target: %s)\n" "$verdict" "$measured" "$target"
}

# Append Markdown text (read from stdin) to the "details" part of the report.
add_details() {
    cat >> "$DETAILS_FILE"
}

section_title() {
    echo
    printf '\033[1m==> %s\033[0m\n' "$*"
}

# A warning line for the report when the load generator, not named, was busy.
client_limit_note() {
    if is_at_least "${R[CLIENT_CPU_PCT]}" 90; then
        echo "> **Note:** the load generator used ${R[CLIENT_CPU_PCT]}% of its CPU, so it may"
        echo "> have been the limit, not named. Try a higher LOAD_PROCESSES in settings.conf."
    fi
}


# =============================================================================
#  PART 2 - running the load generator (lib/dnsload.py) and reading its output
# =============================================================================
#
#  run_load <name> <query file> [extra dnsload.py options ...]
#
#  Default load: $DURATION seconds, $OUTSTANDING queries in flight, UDP.
#  Extra options replace the defaults, for example:
#      run_load latency "$MIXED_QUERIES" --rate 10000
#      run_load tcp-new "$MIXED_QUERIES" --tcp --new-connection
#
#  Saves   raw/<name>.txt             all numbers, one KEY=VALUE per line
#          raw/<name>-per-second.csv  answered queries in each second
#  Sets    R[...] for the caller, for example:
#          R[QPS] R[SENT] R[ANSWERED] R[LOST] R[LOST_PCT] R[LATENCY_P95_MS]
#          R[RCODE_NOERROR] R[RCODE_SERVFAIL] R[CLIENT_CPU_PCT] ...
#          (the full list is at the end of lib/dnsload.py)
# -----------------------------------------------------------------------------
run_load() {
    local name="$1" query_file="$2"
    shift 2
    local output="$RAW_DIR/$name.txt"

    local options=(
        --server "$DNS_SERVER" --port "$DNS_PORT"
        --queries "$query_file"
        --processes "$LOAD_PROCESSES"
        --timeout "$QUERY_TIMEOUT"
        --duration "$DURATION"
        --outstanding "$OUTSTANDING"
        --per-second-csv "$RAW_DIR/$name-per-second.csv"
        "$@"                              # the caller's options come last and win
    )

    printf '    %-28s %s ... ' "$name" "${*:-"$OUTSTANDING in flight, ${DURATION}s"}"

    local errors="$RAW_DIR/$name.errors.txt"
    if ! python3 "$DNSLOAD" "${options[@]}" > "$output" 2> "$errors"; then
        echo "failed"
        die "The load generator failed. See $errors"
    fi
    [[ -s $errors ]] || rm -f "$errors"      # keep the file only if it has messages

    R=()
    local key value
    while IFS='=' read -r key value; do
        R[$key]="$value"
    done < "$output"

    if (( R[ANSWERED] == 0 )); then
        echo "no answers"
        die "named did not answer any query. Is it running? (./start-dns.sh status)"
    fi
    printf '%10s queries/s\n' "$(thousands "${R[QPS]}")"
}

# Queries that did not get a proper answer: lost (no answer at all) plus
# server failures and refusals. NXDOMAIN ("name does not exist") is a correct
# answer, so it is not counted.
failed_queries() {
    echo $(( R[LOST] + R[RCODE_SERVFAIL] + R[RCODE_REFUSED] + R[RCODE_FORMERR] \
             + R[RCODE_NOTIMP] + R[RCODE_OTHER] ))
}


# =============================================================================
#  PART 3 - sampling named's CPU and memory while load runs
# =============================================================================

# Total CPU time (seconds) that named has used so far.
# With systemd the service's cgroup counter is used; otherwise /proc.
named_cpu_seconds() {
    local cgroup cpu_stat
    if has_systemd; then
        cgroup="$(systemctl show --property ControlGroup --value named 2>/dev/null)"
        cpu_stat="/sys/fs/cgroup${cgroup}/cpu.stat"
        if [[ -n $cgroup && -r $cpu_stat ]]; then
            awk '/^usage_usec/ {printf "%.3f", $2 / 1000000}' "$cpu_stat"
            return
        fi
    fi
    # Fields 14 and 15 of /proc/<pid>/stat = user and system CPU ticks.
    awk -v ticks_per_second="$(getconf CLK_TCK)" \
        '{printf "%.3f", ($14 + $15) / ticks_per_second}' "/proc/$(named_pid)/stat"
}

# Memory used by named, in MB. PSS ("proportional set size") counts memory
# shared with other programs (like system libraries) only partly, fairly.
named_memory_mb() {
    awk '/^Pss:/ {printf "%.1f", $2 / 1024}' "/proc/$(named_pid)/smaps_rollup"
}

named_thread_count() {
    awk '/^Threads:/ {print $2}' "/proc/$(named_pid)/status"
}

# Runs in the background: writes one CSV line per second until killed.
sampler_loop() {
    local csv_file="$1"
    local start previous_time previous_cpu now cpu cpu_percent

    # This loop is stopped with "kill"; that is expected, not an error to report.
    trap - ERR

    echo "elapsed_s,memory_mb,cpu_percent_of_all_cpus,threads" > "$csv_file"
    start="$(now_seconds)"
    previous_time="$start"
    previous_cpu="$(named_cpu_seconds)"

    while true; do
        sleep "$SAMPLE_INTERVAL_SECONDS"
        now="$(now_seconds)"
        cpu="$(named_cpu_seconds)"
        cpu_percent="$(calc "($cpu - $previous_cpu) / ($now - $previous_time) / $CPU_COUNT * 100")"

        printf '%s,%s,%s,%s\n' "$(calc "$now - $start")" "$(named_memory_mb)" \
            "$cpu_percent" "$(named_thread_count)" >> "$csv_file"

        previous_time="$now"
        previous_cpu="$cpu"
    done
}

# -----------------------------------------------------------------------------
#  run_sampled_load
#
#  One full-speed load run while CPU and memory are sampled every second.
#  The cpu and memory tests both read from this same run, so when you run
#  both, the load is generated only once.
#
#  Sets: LOAD_CPU_AVG_PCT  LOAD_CPU_PEAK_PCT  LOAD_CPU_SECONDS  LOAD_QPS
#        LOAD_ANSWERED     LOAD_CLIENT_CPU_PCT
#        LOAD_MEM_IDLE_MB  LOAD_MEM_PEAK_MB   LOAD_THREADS
# -----------------------------------------------------------------------------
SAMPLED_LOAD_DONE=no

run_sampled_load() {
    [[ $SAMPLED_LOAD_DONE == yes ]] && return
    local samples="$RAW_DIR/resource-samples.csv"

    LOAD_MEM_IDLE_MB="$(named_memory_mb)"      # before the load starts
    LOAD_THREADS="$(named_thread_count)"

    sampler_loop "$samples" &
    local sampler_pid=$!

    local cpu_before time_before cpu_after time_after
    cpu_before="$(named_cpu_seconds)"
    time_before="$(now_seconds)"

    run_load "resource-load" "$MIXED_QUERIES"

    cpu_after="$(named_cpu_seconds)"
    time_after="$(now_seconds)"
    kill "$sampler_pid" 2>/dev/null || true
    wait "$sampler_pid" 2>/dev/null || true

    LOAD_QPS="${R[QPS]}"
    LOAD_ANSWERED="${R[ANSWERED]}"
    LOAD_CLIENT_CPU_PCT="${R[CLIENT_CPU_PCT]}"
    LOAD_CPU_SECONDS="$(calc3 "$cpu_after - $cpu_before")"
    LOAD_CPU_AVG_PCT="$(calc "$LOAD_CPU_SECONDS / ($time_after - $time_before) / $CPU_COUNT * 100")"

    # Column numbers in the CSV: 2 = memory_mb, 3 = cpu%
    LOAD_CPU_PEAK_PCT="$(awk -F, 'NR > 1 && $3 > m {m = $3} END {printf "%.1f", m}' "$samples")"
    LOAD_MEM_PEAK_MB="$(awk  -F, -v m="$LOAD_MEM_IDLE_MB" 'NR > 1 && $2 > m {m = $2} END {printf "%.1f", m}' "$samples")"

    SAMPLED_LOAD_DONE=yes
}


# =============================================================================
#  PART 4 - the tests, one function per metric
# =============================================================================

# -----------------------------------------------------------------------------
test_startup() {
    section_title "Startup time  ($STARTUP_ROUNDS restarts, zone of $ZONE_RECORDS hosts)"
    local round start_time end_time elapsed_ms times=()

    for (( round = 1; round <= STARTUP_ROUNDS; round++ )); do
        named_stop
        start_time="$(now_seconds)"
        named_start
        if ! wait_until_answering 120; then
            die "named did not come back after restart. See: journalctl -u named"
        fi
        end_time="$(now_seconds)"
        elapsed_ms="$(calc "($end_time - $start_time) * 1000")"
        times+=("$elapsed_ms")
        printf '    restart %d: %s ms\n' "$round" "$elapsed_ms"
    done

    # Reload: named reads its configuration and changed zones again while
    # it keeps answering. "touch" marks the zone file as changed.
    touch "$ZONE_DIR/$TEST_ZONE.zone"
    start_time="$(now_seconds)"
    rndc reload > "$RAW_DIR/startup-reload.txt" 2>&1
    end_time="$(now_seconds)"
    local reload_ms
    reload_ms="$(calc "($end_time - $start_time) * 1000")"
    wait_until_answering 60 || die "named stopped answering after 'rndc reload'."
    printf '    reload (rndc reload): %s ms\n' "$reload_ms"

    local average fastest slowest
    average="$(printf '%s\n' "${times[@]}" | awk '{s += $1} END {printf "%.1f", s / NR}')"
    fastest="$(printf '%s\n' "${times[@]}" | sort -n | head -1)"
    slowest="$(printf '%s\n' "${times[@]}" | sort -n | tail -1)"

    record_result "Startup time" "$average ms average" "<= $TARGET_STARTUP_MAX_MS ms" \
        "$(verdict_at_most "$average" "$TARGET_STARTUP_MAX_MS")"

    add_details <<EOF
## Startup time

named was stopped and started $STARTUP_ROUNDS times. Each time was measured from
the start command until the first DNS answer. Startup includes loading the
test zone ($ZONE_RECORDS hosts = $(( ZONE_RECORDS * 2 + 7 )) records).

| Round | Time (ms) |
|------:|----------:|
$(for i in "${!times[@]}"; do printf '| %5d | %9s |\n' $(( i + 1 )) "${times[$i]}"; done)

Average **$average ms**, fastest $fastest ms, slowest $slowest ms.

A reload (\`rndc reload\`, after marking the zone file as changed) took
**$reload_ms ms**. During a reload named keeps answering with the old data,
so a reload does not interrupt the service; a restart does.

EOF
}

# -----------------------------------------------------------------------------
test_throughput() {
    section_title "Throughput  (mixed queries, $OUTSTANDING in flight, full speed)"
    run_load "throughput" "$MIXED_QUERIES"

    record_result "Throughput" "$(thousands "${R[QPS]}") queries/s" \
        ">= $(thousands "$TARGET_THROUGHPUT_MIN_QPS") queries/s" \
        "$(verdict_at_least "${R[QPS]}" "$TARGET_THROUGHPUT_MIN_QPS")"

    add_details <<EOF
## Throughput (queries per second)

The load generator kept $OUTSTANDING queries in flight for $DURATION seconds:
as soon as an answer came back, the next query was sent. The query mix is
described in PERFORMANCE-METRICS.md (A, AAAA, MX, TXT and non-existent names).

| Item | Value |
|---|---:|
| Queries answered per second | **$(thousands "${R[QPS]}")** |
| Queries sent | $(thousands "${R[SENT]}") |
| Answered | $(thousands "${R[ANSWERED]}") |
| Lost (no answer within ${QUERY_TIMEOUT}s) | ${R[LOST]} |
| Answers: NOERROR / NXDOMAIN / SERVFAIL | $(thousands "${R[RCODE_NOERROR]}") / $(thousands "${R[RCODE_NXDOMAIN]}") / ${R[RCODE_SERVFAIL]} |
| Average response time | ${R[LATENCY_AVG_MS]} ms |
| Load generator CPU (of $LOAD_PROCESSES processes) | ${R[CLIENT_CPU_PCT]}% |

$(client_limit_note)

Answers per second, second by second: \`raw/throughput-per-second.csv\`

EOF
}

# -----------------------------------------------------------------------------
test_latency() {
    section_title "Latency  (steady load of $(thousands "$LATENCY_TEST_QPS") queries/s)"
    # Up to 1000 in flight, so the rate can always be reached.
    run_load "latency" "$MIXED_QUERIES" --rate "$LATENCY_TEST_QPS" --outstanding 1000

    local verdict=PASS
    is_at_most "${R[LATENCY_P95_MS]}" "$TARGET_LATENCY_P95_MAX_MS" || verdict=FAIL
    is_at_most "${R[LATENCY_P99_MS]}" "$TARGET_LATENCY_P99_MAX_MS" || verdict=FAIL

    record_result "Latency (p95 / p99)" "${R[LATENCY_P95_MS]} ms / ${R[LATENCY_P99_MS]} ms" \
        "<= $TARGET_LATENCY_P95_MAX_MS ms / <= $TARGET_LATENCY_P99_MAX_MS ms" "$verdict"

    local rate_note=""
    if ! is_at_least "${R[QPS]}" "$(calc "$LATENCY_TEST_QPS * 0.95")"; then
        rate_note="> **Note:** only $(thousands "${R[QPS]}") of the requested $(thousands "$LATENCY_TEST_QPS") queries/s were reached."
    fi

    add_details <<EOF
## Latency (response time)

How long one query took, from sending it to receiving the answer, while a
steady $(thousands "$LATENCY_TEST_QPS") queries per second were sent (a normal, not
maximum, load). "p95 = 0.5 ms" means 95 of every 100 answers came within 0.5 ms.

| Statistic | Time (ms) |
|---|---:|
| Fastest | ${R[LATENCY_MIN_MS]} |
| Average | ${R[LATENCY_AVG_MS]} |
| p50 (median) | ${R[LATENCY_P50_MS]} |
| p90 | ${R[LATENCY_P90_MS]} |
| p95 | **${R[LATENCY_P95_MS]}** |
| p99 | **${R[LATENCY_P99_MS]}** |
| p99.9 | ${R[LATENCY_P999_MS]} |
| Slowest | ${R[LATENCY_MAX_MS]} |

Queries answered: $(thousands "${R[ANSWERED]}") at $(thousands "${R[QPS]}") per second, lost: ${R[LOST]}.

$rate_note

EOF
}

# -----------------------------------------------------------------------------
test_concurrency() {
    section_title "Concurrency scaling  (queries in flight: $OUTSTANDING_LEVELS)"
    local level table="" peak_qps=0 peak_level=0 last_qps=0 last_level=0

    for level in $OUTSTANDING_LEVELS; do
        run_load "concurrency-${level}-in-flight" "$MIXED_QUERIES" --outstanding "$level"
        table+="$(printf '| %9s | %9s | %8s | %8s | %8s | %6s |' \
            "$level" "$(thousands "${R[QPS]}")" "${R[LATENCY_AVG_MS]}" \
            "${R[LATENCY_P95_MS]}" "${R[LATENCY_P99_MS]}" "$(failed_queries)")"$'\n'

        if is_at_least "${R[QPS]}" "$peak_qps"; then
            peak_qps="${R[QPS]}"
            peak_level="$level"
        fi
        last_qps="${R[QPS]}"
        last_level="$level"
    done

    local kept_pct
    kept_pct="$(calc "$last_qps / $peak_qps * 100")"

    record_result "Concurrency scaling" \
        "$kept_pct% of peak kept at $last_level in flight" \
        ">= $TARGET_SCALING_MIN_PCT% of peak" \
        "$(verdict_at_least "$kept_pct" "$TARGET_SCALING_MIN_PCT")"

    add_details <<EOF
## Concurrency scaling

The full-speed test repeated with more and more queries in flight at the same
time. Healthy behaviour: queries/second rises and then levels off, and it does
not collapse at the highest level. Response time grows once named is busy.

| In flight | Queries/s | Avg (ms) | p95 (ms) | p99 (ms) | Failed |
|----------:|----------:|---------:|---------:|---------:|-------:|
${table}
Peak: **$(thousands "$peak_qps") queries/s with $peak_level in flight**.
With $last_level in flight named still delivered **$kept_pct%** of that peak.

EOF
}

# -----------------------------------------------------------------------------
# One value from /proc/net/snmp, the kernel's network counters.
#   udp_counter RcvbufErrors  -> UDP packets dropped because a receive buffer was full
udp_counter() {
    awk -v name="$1" '/^Udp:/ {
        if (!header_seen) { for (i = 2; i <= NF; i++) column[$i] = i; header_seen = 1 }
        else              { print $column[name]; exit }
    }' /proc/net/snmp
}

test_loss() {
    section_title "Query loss  (fixed rate of $(thousands "$LOSS_TEST_QPS") queries/s)"

    local received_before kernel_drops_before
    received_before="$(named_counter nsstats Requestv4)"
    kernel_drops_before="$(udp_counter RcvbufErrors)"

    # Up to 5000 in flight, so a slow moment on the server does not slow the sender.
    run_load "loss" "$MIXED_QUERIES" --rate "$LOSS_TEST_QPS" --outstanding 5000

    local received kernel_drops failed failure_pct
    received=$(( $(named_counter nsstats Requestv4) - received_before ))
    kernel_drops=$(( $(udp_counter RcvbufErrors) - kernel_drops_before ))
    failed="$(failed_queries)"
    failure_pct="$(awk -v bad="$failed" -v all="${R[SENT]}" 'BEGIN { printf "%.3f", bad / all * 100 }')"

    record_result "Query loss" "$failure_pct% ($failed of $(thousands "${R[SENT]}"))" \
        "<= $TARGET_LOSS_MAX_PCT%" \
        "$(verdict_at_most "$failure_pct" "$TARGET_LOSS_MAX_PCT")"

    local rate_note=""
    if ! is_at_least "${R[QPS]}" "$(calc "$LOSS_TEST_QPS * 0.95")"; then
        rate_note="> **Note:** only $(thousands "${R[QPS]}") of the requested $(thousands "$LOSS_TEST_QPS") queries/s were reached, so named was pushed less than planned."
    fi

    add_details <<EOF
## Query loss

DNS normally uses UDP, which has no delivery guarantee: when a server is too
busy, queries are silently dropped and the client must wait and ask again.
Here $(thousands "$LOSS_TEST_QPS") queries per second were sent for $DURATION seconds.

| Item | Value |
|---|---:|
| Queries sent | $(thousands "${R[SENT]}") |
| Queries named received (its own counter) | $(thousands "$received") |
| Dropped by the kernel, receive buffer full (whole system) | $kernel_drops |
| Answered | $(thousands "${R[ANSWERED]}") |
| Lost (no answer within ${QUERY_TIMEOUT}s) | ${R[LOST]} |
| SERVFAIL / REFUSED answers | ${R[RCODE_SERVFAIL]} / ${R[RCODE_REFUSED]} |
| **Failed in total** | **$failed ($failure_pct%)** |
| Answer rate reached | $(thousands "${R[QPS]}") queries/s |

How to read it: if "received" is lower than "sent", the queries were dropped
before named saw them (kernel buffer full); if named received them but did
not answer, named itself was overloaded.

$rate_note

EOF
}

# -----------------------------------------------------------------------------
test_tcp() {
    section_title "TCP compared with UDP  ($OUTSTANDING in flight)"

    run_load "tcp-baseline-udp" "$MIXED_QUERIES"
    local udp_qps="${R[QPS]}" udp_p95="${R[LATENCY_P95_MS]}" udp_failed
    udp_failed="$(failed_queries)"

    run_load "tcp-reused-connections" "$MIXED_QUERIES" --tcp
    local reuse_qps="${R[QPS]}" reuse_p95="${R[LATENCY_P95_MS]}" reuse_failed
    reuse_failed="$(failed_queries)"

    run_load "tcp-new-connection-each" "$MIXED_QUERIES" --tcp --new-connection
    local new_qps="${R[QPS]}" new_p95="${R[LATENCY_P95_MS]}" new_failed
    new_failed="$(failed_queries)"

    record_result "TCP queries" "$(thousands "$new_qps") queries/s (new connection each)" \
        ">= $(thousands "$TARGET_TCP_MIN_QPS") queries/s" \
        "$(verdict_at_least "$new_qps" "$TARGET_TCP_MIN_QPS")"

    add_details <<EOF
## TCP compared with UDP

DNS switches to TCP for large answers (for example DNSSEC or long TXT records),
for zone transfers, and for clients that require it. TCP needs a connection
first, so it costs more than UDP.

| Transport | Queries/s | p95 (ms) | Failed | Compared with UDP |
|---|---:|---:|---:|---:|
| UDP | $(thousands "$udp_qps") | $udp_p95 | $udp_failed | 100% |
| TCP, connections reused | $(thousands "$reuse_qps") | $reuse_p95 | $reuse_failed | $(calc "$reuse_qps / $udp_qps * 100")% |
| TCP, new connection per query | **$(thousands "$new_qps")** | $new_p95 | $new_failed | $(calc "$new_qps / $udp_qps * 100")% |

"New connection per query" is the worst case and the one with a target.
The number of TCP connections named accepts at once is \`tcp-clients\`
($TCP_CLIENTS, set in settings.conf).

EOF
}

# -----------------------------------------------------------------------------
test_cache() {
    section_title "Cache  ($(thousands "$CACHE_TEST_NAMES") different names under $UPSTREAM_ZONE)"

    # Make sure forwarding works before measuring it.
    if [[ $(dig +short +time=2 +tries=1 -p "$DNS_PORT" "@$DNS_SERVER" "check.$UPSTREAM_ZONE" A) != 192.0.2.1 ]]; then
        die "Forwarding to the upstream view does not work. Re-run ./start-dns.sh"
    fi

    # A list of names that have never been asked for.
    local names_file="$RAW_DIR/queries-cache-names.txt"
    awk -v count="$CACHE_TEST_NAMES" -v zone="$UPSTREAM_ZONE" -v run="$(date +%s)" \
        'BEGIN { for (i = 1; i <= count; i++) printf "name%d-%d.%s A\n", i, run, zone }' > "$names_file"

    rndc flush                                   # start with an empty cache
    local hits_0 misses_0 hits_1 misses_1 hits_2 misses_2
    hits_0="$(named_counter cachestats QueryHits)"
    misses_0="$(named_counter cachestats QueryMisses)"

    # Phase 1: every name once. Nothing is cached yet, so every query is forwarded.
    run_load "cache-miss" "$names_file" --once
    local miss_qps="${R[QPS]}" miss_p50="${R[LATENCY_P50_MS]}" miss_p95="${R[LATENCY_P95_MS]}"
    local miss_failed
    miss_failed="$(failed_queries)"
    hits_1="$(named_counter cachestats QueryHits)"
    misses_1="$(named_counter cachestats QueryMisses)"

    # Phase 2: the same names again, for $DURATION seconds. All are now cached.
    run_load "cache-hit" "$names_file"
    local hit_qps="${R[QPS]}" hit_p50="${R[LATENCY_P50_MS]}" hit_p95="${R[LATENCY_P95_MS]}"
    local hit_failed
    hit_failed="$(failed_queries)"
    hits_2="$(named_counter cachestats QueryHits)"
    misses_2="$(named_counter cachestats QueryMisses)"

    local hit_ratio cache_nodes cache_memory_mb
    hit_ratio="$(awk -v h=$(( hits_2 - hits_1 )) -v m=$(( misses_2 - misses_1 )) \
        'BEGIN { printf "%.2f", (h + m > 0 ? h / (h + m) * 100 : 0) }')"
    cache_nodes="$(named_counter cachestats CacheNodes)"
    cache_memory_mb="$(calc "($(named_counter cachestats TreeMemInUse) + $(named_counter cachestats HeapMemInUse)) / 1048576")"

    record_result "Cache hit latency (p95)" "$hit_p95 ms (miss: $miss_p95 ms)" \
        "<= $TARGET_CACHE_HIT_P95_MAX_MS ms" \
        "$(verdict_at_most "$hit_p95" "$TARGET_CACHE_HIT_P95_MAX_MS")"

    add_details <<EOF
## Cache performance

A caching (recursive) DNS server remembers answers it fetched from other
servers. Phase 1 asked for $(thousands "$CACHE_TEST_NAMES") new names once each: every one was a
**cache miss** and had to be fetched from the upstream server. Phase 2 asked
for the same names again for $DURATION seconds: every one was a **cache hit**.

| Phase | Queries/s | p50 (ms) | p95 (ms) | Failed |
|---|---:|---:|---:|---:|
| 1 - cache miss (fetched upstream) | $(thousands "$miss_qps") | $miss_p50 | $miss_p95 | $miss_failed |
| 2 - cache hit (answered from memory) | **$(thousands "$hit_qps")** | $hit_p50 | **$hit_p95** | $hit_failed |

| named's own counters | Value |
|---|---:|
| Cache misses in phase 1 | $(thousands $(( misses_1 - misses_0 ))) |
| Cache hit ratio in phase 2 | $hit_ratio% |
| Names in the cache afterwards | $(thousands "$cache_nodes") |
| Cache memory afterwards | $cache_memory_mb MB |

Cache hits were **$(calc "$hit_qps / $miss_qps")x** as fast as misses. Here the "upstream"
server is on the same machine, so a miss costs only named's own work. On a real
network each miss also waits for the internet (typically 10 - 100 ms), so the
cache matters even more in production.

EOF
}

# -----------------------------------------------------------------------------
test_axfr() {
    section_title "Zone transfer (AXFR) of $TEST_ZONE  ($AXFR_ROUNDS rounds)"
    local round start_time end_time seconds records bytes table="" times=()

    for (( round = 1; round <= AXFR_ROUNDS; round++ )); do
        local output="$RAW_DIR/axfr-round-$round.txt"
        start_time="$(now_seconds)"
        dig +time=30 -p "$DNS_PORT" "@$DNS_SERVER" "$TEST_ZONE" AXFR > "$output" 2>&1 || true
        end_time="$(now_seconds)"

        # dig ends with a line like:  ;; XFR size: 200009 records (messages 52, bytes 4414377)
        records="$(awk '/XFR size:/ {print $4}' "$output")"
        bytes="$(awk '/XFR size:/ {gsub(/\)/, ""); print $NF}' "$output")"
        if [[ -z $records ]] || grep -q 'Transfer failed' "$output"; then
            die "Zone transfer failed. See $output"
        fi

        seconds="$(calc3 "$end_time - $start_time")"
        times+=("$seconds")
        table+="$(printf '| %5d | %7s | %7s | %4s | %9s |' "$round" "$seconds" \
            "$(thousands "$records")" "$(calc "$bytes / 1048576")" \
            "$(thousands "$(calc "$records / $seconds")")")"$'\n'
        printf '    round %d: %s s, %s records\n' "$round" "$seconds" "$records"
    done

    local average
    average="$(printf '%s\n' "${times[@]}" | awk '{s += $1} END {printf "%.3f", s / NR}')"

    record_result "Zone transfer (AXFR)" "$average s for $(thousands "$records") records" \
        "<= $TARGET_AXFR_MAX_S s" \
        "$(verdict_at_most "$average" "$TARGET_AXFR_MAX_S")"

    add_details <<EOF
## Zone transfer (AXFR)

A secondary DNS server copies a whole zone from the primary with a zone
transfer (AXFR, over TCP). Its speed decides how fast changes reach the
secondary servers and how long a new secondary needs to start.

| Round | Seconds | Records | MB | Records/s |
|------:|--------:|--------:|---:|----------:|
${table}
Average **$average s** for the zone $TEST_ZONE.

EOF
}

# -----------------------------------------------------------------------------
test_cpu() {
    section_title "CPU usage under load  ($OUTSTANDING in flight, $CPU_COUNT CPUs)"
    run_sampled_load

    local microseconds_per_query
    microseconds_per_query="$(calc "$LOAD_CPU_SECONDS * 1000000 / $LOAD_ANSWERED")"

    record_result "CPU usage" "$LOAD_CPU_AVG_PCT% average of all CPUs" \
        "<= $TARGET_CPU_MAX_PCT%" \
        "$(verdict_at_most "$LOAD_CPU_AVG_PCT" "$TARGET_CPU_MAX_PCT")"

    add_details <<EOF
## CPU usage

CPU used by named while it answered $(thousands "$LOAD_QPS") queries per second.
100% means every one of the $CPU_COUNT CPUs was fully busy with named.
(The load generator runs on the same machine and uses CPU too.)

| Item | Value |
|---|---:|
| Average CPU, % of all CPUs | **$LOAD_CPU_AVG_PCT%** |
| Busiest second, % of all CPUs | $LOAD_CPU_PEAK_PCT% |
| Same average, in CPU cores | $(calc "$LOAD_CPU_AVG_PCT * $CPU_COUNT / 100") cores |
| CPU time per query | $microseconds_per_query µs |
| named threads | $LOAD_THREADS |
| Load generator CPU (of $LOAD_PROCESSES processes) | $LOAD_CLIENT_CPU_PCT% |

Per-second samples: \`raw/resource-samples.csv\`

EOF
}

# -----------------------------------------------------------------------------
test_memory() {
    section_title "Memory usage  ($OUTSTANDING in flight)"
    run_sampled_load

    local cache_memory_mb
    cache_memory_mb="$(calc "($(named_counter cachestats TreeMemInUse) + $(named_counter cachestats HeapMemInUse)) / 1048576")"

    record_result "Memory usage" "$LOAD_MEM_PEAK_MB MB peak (idle $LOAD_MEM_IDLE_MB MB)" \
        "<= $TARGET_MEMORY_MAX_MB MB" \
        "$(verdict_at_most "$LOAD_MEM_PEAK_MB" "$TARGET_MEMORY_MAX_MB")"

    add_details <<EOF
## Memory usage

Memory of the named process (PSS: memory shared with other programs, such as
system libraries, is counted only by named's fair share).

| Item | Value |
|---|---:|
| Idle, before the load | $LOAD_MEM_IDLE_MB MB |
| Under load (peak) | **$LOAD_MEM_PEAK_MB MB** |
| Of which answer cache (named's own counter) | $cache_memory_mb MB |
| Cache size limit (max-cache-size) | $MAX_CACHE_SIZE |

Most of named's memory is the loaded zones and the answer cache. The cache
grows as more different names are looked up (run the \`cache\` test first to
see it filled), up to max-cache-size.

Per-second samples: \`raw/resource-samples.csv\`

EOF
}


# =============================================================================
#  PART 5 - preparing, and writing the final report
# =============================================================================

check_ready_to_test() {
    require_root "$@"
    command -v python3 >/dev/null || die "'python3' not found (part of every RHEL 9 installation)."
    command -v dig     >/dev/null || die "'dig' not found. Install bind-utils (./start-dns.sh does this)."
    command -v curl    >/dev/null || die "'curl' not found."
    dns_is_answering || die "named does not answer for $TEST_ZONE on $DNS_SERVER. Run ./start-dns.sh first."
    curl --noproxy '*' -sf "$STATS_URL" >/dev/null ||
        die "named statistics ($STATS_URL) not available. Run ./start-dns.sh first."
}

prepare_result_folder() {
    RESULT_DIR="$KIT_DIR/results/$(date +%Y%m%d-%H%M%S)"
    RAW_DIR="$RESULT_DIR/raw"
    DETAILS_FILE="$RAW_DIR/.details.md"
    mkdir -p "$RAW_DIR"
    : > "$DETAILS_FILE"
    ln -sfn "$(basename "$RESULT_DIR")" "$KIT_DIR/results/latest"
}

# The query list used by most tests: 100,000 queries in a fixed random order,
# mixed like everyday traffic (see PERFORMANCE-METRICS.md, section 2).
create_query_list() {
    MIXED_QUERIES="$RAW_DIR/queries-mixed.txt"
    awk -v hosts="$ZONE_RECORDS" -v zone="$TEST_ZONE" 'BEGIN {
        srand(1)                                    # same list on every run
        for (i = 0; i < 100000; i++) {
            host = int(rand() * hosts) + 1
            dice = rand() * 100
            if      (dice < 70) printf "host%d.%s A\n",    host, zone   # 70% IPv4 address
            else if (dice < 80) printf "host%d.%s AAAA\n", host, zone   # 10% IPv6 address
            else if (dice < 85) printf "%s MX\n",                zone   #  5% mail server
            else if (dice < 90) printf "%s TXT\n",               zone   #  5% text record
            else                printf "missing%d.%s A\n", host, zone   # 10% name does not exist
        }
    }' > "$MIXED_QUERIES"
}

# A short unmeasured run, so the first real test does not start "cold".
warm_up() {
    log "Warm-up: 3 seconds of light load (not measured)"
    run_load "warm-up" "$MIXED_QUERIES" --duration 3 --outstanding 10 > /dev/null
}

write_report() {
    local tests_run="$1" started="$2" finished="$3"
    local report="$RESULT_DIR/report.md"
    local named_version threads os_name selinux cpu_model memory_total

    named_version="$(named -v)"
    threads="$(rndc status 2>/dev/null | awk -F': ' '/worker threads/ {print $2}')"
    os_name="$(. /etc/os-release && echo "$PRETTY_NAME")"
    selinux="$(getenforce 2>/dev/null || echo 'not available')"
    cpu_model="$(awk -F': ' '/^model name/ {print $2; exit}' /proc/cpuinfo)"
    memory_total="$(awk '/^MemTotal/ {printf "%.1f GB", $2 / 1048576}' /proc/meminfo)"

    local overall="ALL PASSED"
    (( FAIL_COUNT > 0 )) && overall="$FAIL_COUNT TEST(S) FAILED"

    {
        echo "# DNS Server Performance Test Report"
        echo
        echo "**Result: $overall**"
        echo
        echo "| | |"
        echo "|---|---|"
        echo "| Test started  | $started |"
        echo "| Test finished | $finished |"
        echo "| Host | $(hostname) |"
        echo "| Operating system | $os_name (kernel $(uname -r)) |"
        echo "| CPU | $CPU_COUNT x $cpu_model |"
        echo "| Memory | $memory_total |"
        echo "| DNS server | $named_version, ${threads:-?} worker threads |"
        echo "| SELinux | $selinux |"
        echo "| Server address | $DNS_SERVER port $DNS_PORT |"
        echo "| Test zone | $TEST_ZONE, $ZONE_RECORDS hosts |"
        echo "| Tests run | $tests_run |"
        echo "| Load per run | $DURATION s, $OUTSTANDING queries in flight (unless the test says otherwise) |"
        echo "| Load generator | lib/dnsload.py, $LOAD_PROCESSES processes, same machine |"
        echo
        echo "## Summary"
        echo
        echo "PASS = target met, FAIL = target missed, INFO = measured only (no target)."
        echo "Targets are set in \`settings.conf\`. What each metric means: \`PERFORMANCE-METRICS.md\`."
        echo
        printf '| %-24s | %-46s | %-30s | %-8s |\n' "Test" "Measured" "Target" "Verdict"
        printf '|%s|%s|%s|%s|\n' "$(printf -- '-%.0s' {1..26})" "$(printf -- '-%.0s' {1..48})" \
                                 "$(printf -- '-%.0s' {1..32})" "$(printf -- '-%.0s' {1..10})"
        local row name measured target verdict
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            [[ $verdict == FAIL ]] && verdict="**FAIL**"
            printf '| %-24s | %-46s | %-30s | %-8s |\n' "$name" "$measured" "$target" "$verdict"
        done
        echo
        echo "# Details"
        echo
        cat -s "$DETAILS_FILE"          # -s: no double empty lines
        echo "## Files in this folder"
        echo
        echo "- \`report.md\` - this report"
        echo "- \`summary.csv\` - the summary table, for spreadsheets"
        echo "- \`raw/\` - the query lists, all numbers of every load run, and per-second samples"
    } > "$report"

    # The same summary as CSV.
    {
        echo "test,measured,target,verdict"
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            printf '"%s","%s","%s","%s"\n' "$name" "$measured" "$target" "$verdict"
        done
    } > "$RESULT_DIR/summary.csv"

    rm -f "$DETAILS_FILE"
}

print_usage() {
    # Print the comment block at the top of this file.
    sed -n '3,/^set -euo pipefail/{/^#/p}' "$0" | sed 's/^# \{0,1\}//'
}


# =============================================================================
#  MAIN
# =============================================================================
main() {
    case "${1:-}" in
        -h|--help) print_usage; exit 0 ;;
        --list)    printf '%s\n' "${ALL_TESTS[@]}"; exit 0 ;;
    esac

    # Which tests to run: the names given, or all of them.
    local tests=("$@")
    if (( ${#tests[@]} == 0 )) || [[ ${tests[0]} == all ]]; then
        tests=("${ALL_TESTS[@]}")
    fi
    local test_name
    for test_name in "${tests[@]}"; do
        if [[ " ${ALL_TESTS[*]} " != *" $test_name "* ]]; then
            echo "Unknown test: '$test_name'. Valid tests: ${ALL_TESTS[*]}" >&2
            exit 2
        fi
    done

    # Problems found before testing starts give exit code 2 ("could not run").
    ( check_ready_to_test ) || exit 2
    prepare_result_folder

    # Always stop the background sampler if the script is interrupted.
    trap 'kill $(jobs -p) 2>/dev/null || true' EXIT

    local started finished
    started="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    log "Testing DNS server $DNS_SERVER port $DNS_PORT - tests: ${tests[*]}"
    log "Results folder: $RESULT_DIR"
    create_query_list
    warm_up

    for test_name in "${tests[@]}"; do
        "test_$test_name"
    done

    finished="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    write_report "${tests[*]}" "$started" "$finished"

    echo
    sed -n '/^## Summary/,/^# Details/p' "$RESULT_DIR/report.md" | sed '$d'
    echo
    ok "Report saved: $RESULT_DIR/report.md"

    # Exit code: 0 when nothing failed, otherwise 1.
    if (( FAIL_COUNT > 0 )); then
        exit 1
    fi
}

main "$@"
