#!/usr/bin/env bash
# =============================================================================
#  test-apache.sh - measure Apache httpd performance, one metric at a time,
#                   and save an easy-to-read report
# =============================================================================
#
#  USAGE (as root, after ./start-apache.sh)
#      ./test-apache.sh                    run every test (about 3 minutes)
#      ./test-apache.sh throughput         run one test
#      ./test-apache.sh latency errors     run several tests
#      ./test-apache.sh --list             show the test names
#
#      DURATION=30 CONCURRENCY=100 ./test-apache.sh throughput
#                                          override settings.conf for one run
#
#  THE TESTS  (explained in detail in PERFORMANCE-METRICS.md)
#      startup      time from "start daemon" until the first page is served
#      throughput   requests served per second
#      latency      response time per request (average and percentiles)
#      concurrency  how throughput and latency change as clients increase
#      keepalive    gain from reusing TCP connections
#      transfer     download speed of a large file (MB/s)
#      errors       failed requests while Apache is overloaded
#      cpu          CPU used by Apache under load
#      memory       memory used by Apache, idle and under load
#      workers      busy / idle worker threads under load
#
#  OUTPUT
#      results/<date>-<time>/report.md    the human-readable report
#      results/<date>-<time>/summary.csv  one line per test (for spreadsheets)
#      results/<date>-<time>/raw/         unmodified tool output
#      results/latest                     link to the newest result folder
#
#  EXIT CODE
#      0 = every test passed,  1 = at least one FAIL,  2 = could not run
#
#  Load is generated by ApacheBench ("ab", from the httpd-tools RPM).
#  All traffic stays on this machine; no internet access is needed.
# =============================================================================

set -euo pipefail
source "$(dirname -- "${BASH_SOURCE[0]}")/lib/common.sh"

ALL_TESTS=(startup throughput latency concurrency keepalive transfer errors cpu memory workers)

SMALL_PAGE_URL="$BASE_URL/perf-test/small.html"    # 1 KB page
LARGE_FILE_URL="$BASE_URL/perf-test/large.bin"     # 1 MB file
APACHE_ERROR_LOG="/var/log/httpd/error_log"

SAMPLE_INTERVAL_SECONDS=1       # how often CPU / memory / workers are sampled
CPU_COUNT="$(nproc)"

# "ab" needs a request limit; it stops at DURATION or at this count,
# whichever comes first. 200,000 per second is far above what one host serves.
MAX_REQUESTS_PER_RUN=$(( DURATION * 200000 ))

# Filled in by the functions below.
RESULT_DIR=""
RAW_DIR=""
DETAILS_FILE=""
SUMMARY_ROWS=()     # "Test|Measured|Target|Verdict"
FAIL_COUNT=0


# =============================================================================
#  PART 1 - small helpers
# =============================================================================

# Floating-point math and comparisons (bash itself only knows whole numbers).
calc()          { awk "BEGIN { printf \"%.1f\", $* }"; }
is_at_most()    { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a <= b) }'; }
is_at_least()   { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a >= b) }'; }
now_seconds()   { date +%s.%N; }

# Turn a comparison into the words PASS / FAIL.
verdict_at_most()  { if is_at_most  "$1" "$2"; then echo PASS; else echo FAIL; fi; }
verdict_at_least() { if is_at_least "$1" "$2"; then echo PASS; else echo FAIL; fi; }

# Remember one line for the summary table at the top of the report.
#   record_result "Test name" "measured value" "target" PASS|FAIL|INFO
record_result() {
    local name="$1" measured="$2" target="$3" verdict="$4"
    SUMMARY_ROWS+=("$name|$measured|$target|$verdict")
    [[ $verdict == FAIL ]] && FAIL_COUNT=$(( FAIL_COUNT + 1 ))

    local colour='1;32'                       # green  = PASS
    [[ $verdict == FAIL ]] && colour='1;31'   # red    = FAIL
    [[ $verdict == INFO ]] && colour='1;36'   # cyan   = INFO
    printf "    => \033[${colour}m%-4s\033[0m  %s   (target: %s)\n" "$verdict" "$measured" "$target"
}

# Append Markdown text (read from stdin) to the "details" part of the report.
add_details() {
    cat >> "$DETAILS_FILE"
}

section_title() {
    echo
    printf '\033[1m==> %s\033[0m\n' "$*"
}


# =============================================================================
#  PART 2 - running ApacheBench and reading its output
# =============================================================================
#
#  run_ab <name> <url> <clients> <keepalive: yes|no> [seconds]
#
#  Runs one load test and saves:
#      raw/<name>.txt              the full ab output
#      raw/<name>-percentiles.csv  response time for every percentile 0..99
#
#  and sets these variables for the caller:
#      AB_COMPLETE  AB_FAILED  AB_FAILED_DETAIL  AB_NON_2XX  AB_SERVER_CLOSES
#      AB_RPS  AB_TRANSFER_KBPS  AB_MEAN_MS
#      AB_P50_MS  AB_P90_MS  AB_P95_MS  AB_P99_MS  AB_MAX_MS
#
#  A note on "failed requests" in keep-alive mode:
#  Apache sometimes answers a request completely but then closes the
#  keep-alive connection (the client simply reconnects). ApacheBench wrongly
#  counts each such answer as a "Length" failure. We count those closures
#  separately in AB_SERVER_CLOSES, and AB_FAILED holds only real failures.
# -----------------------------------------------------------------------------
run_ab() {
    local name="$1" url="$2" clients="$3" keepalive="$4" seconds="${5:-$DURATION}"
    local output="$RAW_DIR/$name.txt"
    local percentiles="$RAW_DIR/$name-percentiles.csv"

    local options=(
        -t "$seconds"                    # run for this many seconds ...
        -n "$MAX_REQUESTS_PER_RUN"       # ... or until this many requests
        -c "$clients"                    # simultaneous clients
        -r                               # keep going after socket errors
        -s 30                            # per-request timeout, seconds
        -e "$percentiles"                # save the percentile table
    )
    [[ $keepalive == yes ]] && options+=(-k)

    printf '    %-34s %4s clients, keep-alive %-3s, %ss ... ' \
        "$name" "$clients" "$keepalive" "$seconds"

    # ab returns non-zero on some socket errors; we still read its report.
    ab "${options[@]}" "$url" > "$output" 2>&1 || true

    AB_COMPLETE="$(awk '/^Complete requests:/      {print $3}' "$output")"
    AB_NON_2XX="$(awk  '/^Non-2xx responses:/      {print $3}' "$output")"
    local keepalive_answers
    keepalive_answers="$(awk '/^Keep-Alive requests:/ {print $3}' "$output")"
    AB_RPS="$(awk      '/^Requests per second:/    {print $4}' "$output")"
    AB_TRANSFER_KBPS="$(awk '/^Transfer rate:/     {print $3}' "$output")"
    AB_MEAN_MS="$(awk  '/^Time per request:/       {print $4; exit}' "$output")"
    AB_MAX_MS="$(awk   '/100%/                     {print $2}' "$output")"

    AB_P50_MS="$(awk -F, '$1 == 50 {printf "%.2f", $2}' "$percentiles" 2>/dev/null)"
    AB_P90_MS="$(awk -F, '$1 == 90 {printf "%.2f", $2}' "$percentiles" 2>/dev/null)"
    AB_P95_MS="$(awk -F, '$1 == 95 {printf "%.2f", $2}' "$percentiles" 2>/dev/null)"
    AB_P99_MS="$(awk -F, '$1 == 99 {printf "%.2f", $2}' "$percentiles" 2>/dev/null)"

    if [[ -z $AB_RPS || -z $AB_COMPLETE ]]; then
        echo "no result"
        die "ApacheBench did not produce a result. See $output"
    fi

    # ab only prints "Non-2xx responses" when there are some.
    AB_NON_2XX="${AB_NON_2XX:-0}"

    # ab prints the failure causes on the line after "Failed requests", e.g.
    #    (Connect: 0, Receive: 0, Length: 23, Exceptions: 0)
    # ("|| true": when there are no failures, awk prints nothing and read fails.)
    local connect="" receive="" length="" exceptions=""
    read -r connect receive length exceptions < <(awk '
        /^Failed requests:/ {
            getline                        # the line is only there when failures > 0
            if ($0 ~ /^ *\(Connect:/) {
                gsub(/[^0-9 ]/, "")        # keep only the four numbers
                print $1, $2, $3, $4
            }
        }' "$output") || true
    connect="${connect:-0}" receive="${receive:-0}"
    length="${length:-0}"   exceptions="${exceptions:-0}"

    # Answers that came back without keep-alive = connections Apache closed.
    AB_SERVER_CLOSES=0
    if [[ $keepalive == yes && -n $keepalive_answers ]]; then
        AB_SERVER_CLOSES=$(( AB_COMPLETE - keepalive_answers ))
    fi

    # Length errors caused by those closures are not real failures.
    local real_length_errors=$(( length - AB_SERVER_CLOSES ))
    (( real_length_errors < 0 )) && real_length_errors=0

    AB_FAILED=$(( connect + receive + real_length_errors + exceptions ))
    AB_FAILED_DETAIL="connect $connect, receive $receive, wrong length $real_length_errors, exceptions $exceptions"

    printf '%10.0f req/s\n' "$AB_RPS"
}


# =============================================================================
#  PART 3 - sampling Apache's CPU, memory and workers while load runs
# =============================================================================

# Total CPU time (seconds) that all Apache processes have used so far.
# With systemd the cgroup counter is exact, even for processes that exit.
# Without systemd we add up the counters of the httpd processes alive now.
apache_cpu_seconds() {
    local cgroup cpu_stat
    if has_systemd; then
        cgroup="$(systemctl show --property ControlGroup --value httpd 2>/dev/null)"
        cpu_stat="/sys/fs/cgroup${cgroup}/cpu.stat"
        if [[ -n $cgroup && -r $cpu_stat ]]; then
            awk '/^usage_usec/ {printf "%.3f", $2 / 1000000}' "$cpu_stat"
            return
        fi
    fi

    local ticks_per_second pid total_ticks=0 ticks
    ticks_per_second="$(getconf CLK_TCK)"
    for pid in $(httpd_pids); do
        # Fields 14 and 15 of /proc/<pid>/stat = user and system CPU ticks.
        ticks="$(awk '{print $14 + $15}' "/proc/$pid/stat" 2>/dev/null)" || continue
        total_ticks=$(( total_ticks + ${ticks:-0} ))
    done
    calc "$total_ticks / $ticks_per_second"
}

# Memory used by all Apache processes, in MB.
# PSS ("proportional set size") shares memory used by several processes fairly
# between them, so adding the PSS of all processes gives the true total.
apache_memory_mb() {
    local pid total_kb=0 kb
    for pid in $(httpd_pids); do
        kb="$(awk '/^Pss:/ {print $2}' "/proc/$pid/smaps_rollup" 2>/dev/null)" || continue
        total_kb=$(( total_kb + ${kb:-0} ))
    done
    calc "$total_kb / 1024"
}

apache_process_count() {
    httpd_pids | wc -l
}

# Busy and idle worker threads, as reported by Apache's own status page.
apache_workers() {
    curl --noproxy '*' --silent --max-time 1 "$STATUS_URL" 2>/dev/null |
        awk -F': ' '/^BusyWorkers/ {busy = $2} /^IdleWorkers/ {idle = $2}
                    END {print (busy == "" ? "?" : busy), (idle == "" ? "?" : idle)}'
}

# Runs in the background: writes one CSV line per second until killed.
sampler_loop() {
    local csv_file="$1"
    local start previous_time previous_cpu now cpu cpu_percent busy idle

    echo "elapsed_s,processes,memory_mb,cpu_percent_of_all_cpus,busy_workers,idle_workers" > "$csv_file"
    start="$(now_seconds)"
    previous_time="$start"
    previous_cpu="$(apache_cpu_seconds)"

    while true; do
        sleep "$SAMPLE_INTERVAL_SECONDS"
        now="$(now_seconds)"
        cpu="$(apache_cpu_seconds)"
        cpu_percent="$(calc "($cpu - $previous_cpu) / ($now - $previous_time) / $CPU_COUNT * 100")"
        read -r busy idle < <(apache_workers)

        printf '%s,%s,%s,%s,%s,%s\n' \
            "$(calc "$now - $start")" "$(apache_process_count)" "$(apache_memory_mb)" \
            "$cpu_percent" "$busy" "$idle" >> "$csv_file"

        previous_time="$now"
        previous_cpu="$cpu"
    done
}

# -----------------------------------------------------------------------------
#  run_sampled_load
#
#  One load run (small page, CONCURRENCY clients, keep-alive) while CPU,
#  memory and workers are sampled every second. The cpu, memory and workers
#  tests all read from this same run, so when you run more than one of them
#  the load is generated only once.
#
#  Sets: LOAD_CPU_AVG_PCT  LOAD_CPU_PEAK_PCT  LOAD_CPU_SECONDS
#        LOAD_MEM_IDLE_MB  LOAD_MEM_PEAK_MB   LOAD_PROCS_IDLE  LOAD_PROCS_PEAK
#        LOAD_BUSY_PEAK    LOAD_IDLE_MIN      LOAD_REQUESTS    LOAD_RPS
# -----------------------------------------------------------------------------
SAMPLED_LOAD_DONE=no

run_sampled_load() {
    [[ $SAMPLED_LOAD_DONE == yes ]] && return
    local samples="$RAW_DIR/resource-samples.csv"

    # Values while Apache is idle, before the load starts.
    LOAD_MEM_IDLE_MB="$(apache_memory_mb)"
    LOAD_PROCS_IDLE="$(apache_process_count)"

    sampler_loop "$samples" &
    local sampler_pid=$!

    local cpu_before time_before cpu_after time_after
    cpu_before="$(apache_cpu_seconds)"
    time_before="$(now_seconds)"

    run_ab "resource-load" "$SMALL_PAGE_URL" "$CONCURRENCY" yes

    cpu_after="$(apache_cpu_seconds)"
    time_after="$(now_seconds)"
    kill "$sampler_pid" 2>/dev/null || true
    wait "$sampler_pid" 2>/dev/null || true

    LOAD_REQUESTS="$AB_COMPLETE"
    LOAD_RPS="$AB_RPS"
    LOAD_CPU_SECONDS="$(calc "$cpu_after - $cpu_before")"
    LOAD_CPU_AVG_PCT="$(calc "$LOAD_CPU_SECONDS / ($time_after - $time_before) / $CPU_COUNT * 100")"

    # Column numbers: 2=processes 3=memory_mb 4=cpu% 5=busy 6=idle
    LOAD_CPU_PEAK_PCT="$(awk -F, 'NR > 1 && $4 > m {m = $4} END {printf "%.1f", m}' "$samples")"
    LOAD_MEM_PEAK_MB="$(awk  -F, -v m="$LOAD_MEM_IDLE_MB" 'NR > 1 && $3 > m {m = $3} END {printf "%.1f", m}' "$samples")"
    LOAD_PROCS_PEAK="$(awk   -F, -v m="$LOAD_PROCS_IDLE"  'NR > 1 && $2 > m {m = $2} END {print m}' "$samples")"
    LOAD_BUSY_PEAK="$(awk    -F, 'NR > 1 && $5 != "?" && $5 > m {m = $5} END {print m + 0}' "$samples")"
    LOAD_IDLE_MIN="$(awk     -F, 'NR > 1 && $6 != "?" && (min == "" || $6 < min) {min = $6} END {print min + 0}' "$samples")"

    SAMPLED_LOAD_DONE=yes
}


# =============================================================================
#  PART 4 - the tests, one function per metric
# =============================================================================

# -----------------------------------------------------------------------------
test_startup() {
    section_title "Startup time  ($STARTUP_ROUNDS restarts)"
    local round start_time end_time elapsed_ms times=()

    for (( round = 1; round <= STARTUP_ROUNDS; round++ )); do
        apache_stop
        start_time="$(now_seconds)"
        apache_start
        if ! wait_until_serving 60; then
            die "Apache did not come back after restart. See $APACHE_ERROR_LOG"
        fi
        end_time="$(now_seconds)"
        elapsed_ms="$(calc "($end_time - $start_time) * 1000")"
        times+=("$elapsed_ms")
        printf '    restart %d: %s ms\n' "$round" "$elapsed_ms"
    done

    local average fastest slowest
    average="$(printf '%s\n' "${times[@]}" | awk '{s += $1} END {printf "%.1f", s / NR}')"
    fastest="$(printf '%s\n' "${times[@]}" | sort -n | head -1)"
    slowest="$(printf '%s\n' "${times[@]}" | sort -n | tail -1)"

    record_result "Startup time" "$average ms average" "<= $TARGET_STARTUP_MAX_MS ms" \
        "$(verdict_at_most "$average" "$TARGET_STARTUP_MAX_MS")"

    add_details <<EOF
## Startup time

Apache was stopped and started $STARTUP_ROUNDS times. Each time was measured
from the start command until the first test page was served.

| Round | Time (ms) |
|------:|----------:|
$(for i in "${!times[@]}"; do printf '| %5d | %9s |\n' $(( i + 1 )) "${times[$i]}"; done)

Average **$average ms**, fastest $fastest ms, slowest $slowest ms.

EOF
}

# -----------------------------------------------------------------------------
test_throughput() {
    section_title "Throughput  (1 KB page, $CONCURRENCY clients, keep-alive)"
    run_ab "throughput" "$SMALL_PAGE_URL" "$CONCURRENCY" yes

    record_result "Throughput" "$(calc "$AB_RPS") requests/s" \
        ">= $TARGET_THROUGHPUT_MIN_RPS requests/s" \
        "$(verdict_at_least "$AB_RPS" "$TARGET_THROUGHPUT_MIN_RPS")"

    add_details <<EOF
## Throughput

$CONCURRENCY clients requested a 1 KB page as fast as possible for $DURATION seconds.

| Item | Value |
|---|---:|
| Requests per second | **$(calc "$AB_RPS")** |
| Requests completed | $AB_COMPLETE |
| Failed requests | $AB_FAILED |
| Keep-alive connections closed by Apache | $AB_SERVER_CLOSES |
| Average time per request | $AB_MEAN_MS ms |

Raw output: \`raw/throughput.txt\`

EOF
}

# -----------------------------------------------------------------------------
test_latency() {
    section_title "Latency  (1 KB page, $CONCURRENCY clients, keep-alive)"
    run_ab "latency" "$SMALL_PAGE_URL" "$CONCURRENCY" yes

    local verdict=PASS
    is_at_most "$AB_P95_MS" "$TARGET_LATENCY_P95_MAX_MS" || verdict=FAIL
    is_at_most "$AB_P99_MS" "$TARGET_LATENCY_P99_MAX_MS" || verdict=FAIL

    record_result "Latency (p95 / p99)" "$AB_P95_MS ms / $AB_P99_MS ms" \
        "<= $TARGET_LATENCY_P95_MAX_MS ms / <= $TARGET_LATENCY_P99_MAX_MS ms" "$verdict"

    add_details <<EOF
## Latency (response time)

How long one request took, from sending it to receiving the full answer,
with $CONCURRENCY clients active. "p95 = 3 ms" means 95 of every 100 requests
finished within 3 ms.

| Statistic | Time (ms) |
|---|---:|
| Average | $AB_MEAN_MS |
| p50 (median) | $AB_P50_MS |
| p90 | $AB_P90_MS |
| p95 | **$AB_P95_MS** |
| p99 | **$AB_P99_MS** |
| Slowest request | $AB_MAX_MS |

Every percentile from 0 to 99: \`raw/latency-percentiles.csv\`

EOF
}

# -----------------------------------------------------------------------------
test_concurrency() {
    section_title "Concurrency scaling  (1 KB page, clients: $CONCURRENCY_LEVELS)"
    local clients table="" peak_rps=0 peak_clients=0 last_rps=0 last_clients=0

    for clients in $CONCURRENCY_LEVELS; do
        run_ab "concurrency-${clients}-clients" "$SMALL_PAGE_URL" "$clients" yes
        table+="$(printf '| %7s | %10s | %8s | %8s | %6s | %21s |' \
            "$clients" "$(calc "$AB_RPS")" "$AB_MEAN_MS" "$AB_P95_MS" \
            "$AB_FAILED" "$AB_SERVER_CLOSES")"$'\n'

        if is_at_least "$AB_RPS" "$peak_rps"; then
            peak_rps="$AB_RPS"
            peak_clients="$clients"
        fi
        last_rps="$AB_RPS"
        last_clients="$clients"
    done

    local kept_pct
    kept_pct="$(calc "$last_rps / $peak_rps * 100")"

    record_result "Concurrency scaling" \
        "$kept_pct% of peak kept at $last_clients clients" \
        ">= $TARGET_SCALING_MIN_PCT% of peak" \
        "$(verdict_at_least "$kept_pct" "$TARGET_SCALING_MIN_PCT")"

    add_details <<EOF
## Concurrency scaling

The same test at several numbers of simultaneous clients. Healthy behaviour:
requests/second rises and then levels off; it should not collapse at the
highest level.

| Clients | Requests/s | Avg (ms) | p95 (ms) | Failed | Keep-alive closes (*) |
|--------:|-----------:|---------:|---------:|-------:|----------------------:|
${table}
(*) Connections Apache closed after a complete answer; the client reconnected.
These are not failures, but many of them mean Apache was short of workers.

Peak: **$(calc "$peak_rps") requests/s at $peak_clients clients**.
At $last_clients clients Apache still delivered **$kept_pct%** of that peak.

EOF
}

# -----------------------------------------------------------------------------
test_keepalive() {
    section_title "Keep-alive effect  (1 KB page, $CONCURRENCY clients)"

    run_ab "keepalive-on" "$SMALL_PAGE_URL" "$CONCURRENCY" yes
    local on_rps="$AB_RPS" on_mean="$AB_MEAN_MS" on_p95="$AB_P95_MS"

    run_ab "keepalive-off" "$SMALL_PAGE_URL" "$CONCURRENCY" no
    local off_rps="$AB_RPS" off_mean="$AB_MEAN_MS" off_p95="$AB_P95_MS"

    local gain
    gain="$(calc "$on_rps / $off_rps")"

    record_result "Keep-alive gain" "${gain}x more requests/s with keep-alive" \
        "information only" INFO

    add_details <<EOF
## Keep-alive effect

With keep-alive, a client sends many requests over one TCP connection.
Without it, every request opens and closes a new connection.

| Mode | Requests/s | Avg (ms) | p95 (ms) |
|---|---:|---:|---:|
| Keep-alive ON  | $(calc "$on_rps")  | $on_mean  | $on_p95  |
| Keep-alive OFF | $(calc "$off_rps") | $off_mean | $off_p95 |

Keep-alive served **${gain}x** as many requests per second.

EOF
}

# -----------------------------------------------------------------------------
test_transfer() {
    section_title "Transfer rate  (1 MB file, $CONCURRENCY clients, keep-alive)"
    run_ab "transfer" "$LARGE_FILE_URL" "$CONCURRENCY" yes

    local mb_per_second
    mb_per_second="$(calc "$AB_TRANSFER_KBPS / 1024")"

    record_result "Transfer rate" "$mb_per_second MB/s" \
        ">= $TARGET_TRANSFER_MIN_MBPS MB/s" \
        "$(verdict_at_least "$mb_per_second" "$TARGET_TRANSFER_MIN_MBPS")"

    add_details <<EOF
## Transfer rate (bandwidth)

$CONCURRENCY clients downloaded a 1 MB file repeatedly for $DURATION seconds.

| Item | Value |
|---|---:|
| Data sent per second | **$mb_per_second MB/s** ($(calc "$mb_per_second * 8") Mbit/s) |
| Files downloaded per second | $(calc "$AB_RPS") |
| Files downloaded in total | $AB_COMPLETE |
| Average time per download | $AB_MEAN_MS ms |

Over localhost the network is not a limit, so this shows how fast Apache
itself can push data. Over a real network, the network card is usually the limit.

EOF
}

# -----------------------------------------------------------------------------
test_errors() {
    section_title "Error rate under overload  ($ERROR_TEST_CONCURRENCY clients, no keep-alive)"

    local log_lines_before=0
    [[ -r $APACHE_ERROR_LOG ]] && log_lines_before="$(wc -l < "$APACHE_ERROR_LOG")"

    run_ab "errors-overload" "$SMALL_PAGE_URL" "$ERROR_TEST_CONCURRENCY" no

    local bad_requests error_rate new_log_lines="" new_log_count=0
    bad_requests=$(( AB_FAILED + AB_NON_2XX ))
    error_rate="$(awk -v bad="$bad_requests" -v all="$AB_COMPLETE" \
        'BEGIN { printf "%.3f", (all > 0 ? bad / all * 100 : 100) }')"

    if [[ -r $APACHE_ERROR_LOG ]]; then
        new_log_lines="$(tail -n +"$(( log_lines_before + 1 ))" "$APACHE_ERROR_LOG" |
                         grep -E '\[[a-z_]*:(error|crit|alert|emerg)\]|AH00484' || true)"
        [[ -n $new_log_lines ]] && new_log_count="$(printf '%s\n' "$new_log_lines" | wc -l)"
        printf '%s\n' "$new_log_lines" > "$RAW_DIR/errors-new-error_log-lines.txt"
    fi

    record_result "Error rate" "$error_rate% ($bad_requests of $AB_COMPLETE)" \
        "<= $TARGET_ERROR_RATE_MAX_PCT%" \
        "$(verdict_at_most "$error_rate" "$TARGET_ERROR_RATE_MAX_PCT")"

    add_details <<EOF
## Error rate under overload

$ERROR_TEST_CONCURRENCY clients, each opening a new connection for every request,
for $DURATION seconds. This pushes Apache harder than normal traffic.

| Item | Value |
|---|---:|
| Requests completed | $AB_COMPLETE |
| Failed requests (network / wrong length) | $AB_FAILED |
| &nbsp;&nbsp;broken down as | $AB_FAILED_DETAIL |
| Non-2xx answers (HTTP errors such as 503) | $AB_NON_2XX |
| **Error rate** | **$error_rate%** |
| New serious lines in $APACHE_ERROR_LOG | $new_log_count |

$(if [[ -n $new_log_lines ]]; then
    echo "First lines from the error log (all in \`raw/errors-new-error_log-lines.txt\`):"
    echo
    echo '```'
    printf '%s\n' "$new_log_lines" | head -10
    echo '```'
  fi)

EOF
}

# -----------------------------------------------------------------------------
test_cpu() {
    section_title "CPU usage under load  ($CONCURRENCY clients, $CPU_COUNT CPUs)"
    run_sampled_load

    local ms_per_1000
    ms_per_1000="$(calc "$LOAD_CPU_SECONDS * 1000 / $LOAD_REQUESTS * 1000")"

    record_result "CPU usage" "$LOAD_CPU_AVG_PCT% average of all CPUs" \
        "<= $TARGET_CPU_MAX_PCT%" \
        "$(verdict_at_most "$LOAD_CPU_AVG_PCT" "$TARGET_CPU_MAX_PCT")"

    add_details <<EOF
## CPU usage

CPU used by all Apache processes while serving $(calc "$LOAD_RPS") requests/s.
100% means every one of the $CPU_COUNT CPUs was fully busy with Apache.
(The load generator "ab" runs on the same machine and uses CPU too.)

| Item | Value |
|---|---:|
| Average CPU, % of all CPUs | **$LOAD_CPU_AVG_PCT%** |
| Busiest second, % of all CPUs | $LOAD_CPU_PEAK_PCT% |
| Same average, in CPU cores | $(calc "$LOAD_CPU_AVG_PCT * $CPU_COUNT / 100") cores |
| CPU time per 1000 requests | $ms_per_1000 ms |

Per-second samples: \`raw/resource-samples.csv\`

EOF
}

# -----------------------------------------------------------------------------
test_memory() {
    section_title "Memory usage  ($CONCURRENCY clients)"
    run_sampled_load

    record_result "Memory usage" "$LOAD_MEM_PEAK_MB MB peak (idle $LOAD_MEM_IDLE_MB MB)" \
        "<= $TARGET_MEMORY_MAX_MB MB" \
        "$(verdict_at_most "$LOAD_MEM_PEAK_MB" "$TARGET_MEMORY_MAX_MB")"

    add_details <<EOF
## Memory usage

Memory of all Apache processes together (PSS: memory shared between
processes is counted once, so this is the real total).

| Item | Idle | Under load (peak) |
|---|---:|---:|
| Memory (MB) | $LOAD_MEM_IDLE_MB | **$LOAD_MEM_PEAK_MB** |
| Apache processes | $LOAD_PROCS_IDLE | $LOAD_PROCS_PEAK |
| Memory per process (MB) | $(calc "$LOAD_MEM_IDLE_MB / $LOAD_PROCS_IDLE") | $(calc "$LOAD_MEM_PEAK_MB / $LOAD_PROCS_PEAK") |

Per-second samples: \`raw/resource-samples.csv\`

EOF
}

# -----------------------------------------------------------------------------
test_workers() {
    section_title "Worker usage  ($CONCURRENCY clients, MaxRequestWorkers $MAX_REQUEST_WORKERS)"
    run_sampled_load

    local used_pct
    used_pct="$(calc "$LOAD_BUSY_PEAK / $MAX_REQUEST_WORKERS * 100")"

    record_result "Worker usage" "$LOAD_BUSY_PEAK of $MAX_REQUEST_WORKERS busy ($used_pct%)" \
        "<= $TARGET_WORKERS_MAX_PCT%" \
        "$(verdict_at_most "$used_pct" "$TARGET_WORKERS_MAX_PCT")"

    add_details <<EOF
## Worker usage

A worker is one thread that handles one request at a time. When all
$MAX_REQUEST_WORKERS workers are busy, new clients must wait.
Values come from Apache's own status page (/server-status), once per second.

| Item | Value |
|---|---:|
| Most busy workers at one time | **$LOAD_BUSY_PEAK** |
| Fewest idle workers at one time | $LOAD_IDLE_MIN |
| Maximum allowed (MaxRequestWorkers) | $MAX_REQUEST_WORKERS |
| Peak usage | **$used_pct%** |

To find the real limit, raise CONCURRENCY (for example CONCURRENCY=400).

EOF
}


# =============================================================================
#  PART 5 - preparing, and writing the final report
# =============================================================================

check_ready_to_test() {
    require_root "$@"
    command -v ab   >/dev/null || die "'ab' not found. Install httpd-tools (./start-apache.sh does this)."
    command -v curl >/dev/null || die "'curl' not found."
    apache_is_serving || die "Apache is not serving $HEALTH_URL. Run ./start-apache.sh first."
    curl --noproxy '*' -sf "$STATUS_URL" | grep -q '^BusyWorkers' ||
        die "Status page $STATUS_URL not available. Run ./start-apache.sh first."
}

prepare_result_folder() {
    RESULT_DIR="$KIT_DIR/results/$(date +%Y%m%d-%H%M%S)"
    RAW_DIR="$RESULT_DIR/raw"
    DETAILS_FILE="$RAW_DIR/.details.md"
    mkdir -p "$RAW_DIR"
    : > "$DETAILS_FILE"
    ln -sfn "$(basename "$RESULT_DIR")" "$KIT_DIR/results/latest"
}

# A short unmeasured run, so the first real test does not start "cold".
warm_up() {
    log "Warm-up: 3 seconds of light load (not measured)"
    ab -k -t 3 -n 1000000 -c 10 "$SMALL_PAGE_URL" > "$RAW_DIR/warm-up.txt" 2>&1 || true
}

write_report() {
    local tests_run="$1" started="$2" finished="$3"
    local report="$RESULT_DIR/report.md"
    local httpd_version mpm os_name selinux cpu_model memory_total

    httpd_version="$(/usr/sbin/httpd -v | awk -F': ' '/version/ {print $2}')"
    mpm="$(/usr/sbin/httpd -V 2>/dev/null | awk -F': *' '/Server MPM/ {print $2}')"
    os_name="$(. /etc/os-release && echo "$PRETTY_NAME")"
    selinux="$(getenforce 2>/dev/null || echo 'not available')"
    cpu_model="$(awk -F': ' '/^model name/ {print $2; exit}' /proc/cpuinfo)"
    memory_total="$(awk '/^MemTotal/ {printf "%.1f GB", $2 / 1048576}' /proc/meminfo)"

    local overall="ALL PASSED"
    (( FAIL_COUNT > 0 )) && overall="$FAIL_COUNT TEST(S) FAILED"

    {
        echo "# Apache Performance Test Report"
        echo
        echo "**Result: $overall**"
        echo
        echo "| | |"
        echo "|---|---|"
        echo "| Test started  | $started |"
        echo "| Test finished | $finished |"
        echo "| Host | $(hostname) |"
        echo "| Operating system | $os_name (kernel $(uname -r)) |"
        echo "| CPU | $CPU_COUNT x $cpu_model |"
        echo "| Memory | $memory_total |"
        echo "| Apache | $httpd_version, $mpm MPM |"
        echo "| SELinux | $selinux |"
        echo "| Target URL | $BASE_URL |"
        echo "| Tests run | $tests_run |"
        echo "| Load per run | $DURATION s, $CONCURRENCY clients (unless the test says otherwise) |"
        echo
        echo "## Summary"
        echo
        echo "PASS = target met, FAIL = target missed, INFO = measured only (no target)."
        echo "Targets are set in \`settings.conf\`. What each metric means: \`PERFORMANCE-METRICS.md\`."
        echo
        printf '| %-22s | %-44s | %-30s | %-7s |\n' "Test" "Measured" "Target" "Verdict"
        printf '|%s|%s|%s|%s|\n' "$(printf -- '-%.0s' {1..24})" "$(printf -- '-%.0s' {1..46})" \
                                 "$(printf -- '-%.0s' {1..32})" "$(printf -- '-%.0s' {1..9})"
        local row name measured target verdict
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            [[ $verdict == FAIL ]] && verdict="**FAIL**"
            printf '| %-22s | %-44s | %-30s | %-7s |\n' "$name" "$measured" "$target" "$verdict"
        done
        echo
        echo "# Details"
        echo
        cat "$DETAILS_FILE"
        echo "## Files in this folder"
        echo
        echo "- \`report.md\` - this report"
        echo "- \`summary.csv\` - the summary table, for spreadsheets"
        echo "- \`raw/\` - unmodified ApacheBench output and per-second resource samples"
    } > "$report"

    # The same summary as CSV.
    {
        echo "test,measured,target,verdict"
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            printf '"%s","%s","%s","%s"\n' "$name" "$measured" "$target" "$verdict"
        done
    } > "$RESULT_DIR/summary.csv"

    rm -f "$DETAILS_FILE"
}

print_usage() {
    # Print the comment block at the top of this file.
    sed -n '3,/^set -euo pipefail/{/^#/p}' "$0" | sed 's/^# \{0,1\}//'
}


# =============================================================================
#  MAIN
# =============================================================================
main() {
    case "${1:-}" in
        -h|--help) print_usage; exit 0 ;;
        --list)    printf '%s\n' "${ALL_TESTS[@]}"; exit 0 ;;
    esac

    # Which tests to run: the names given, or all of them.
    local tests=("$@")
    if (( ${#tests[@]} == 0 )) || [[ ${tests[0]} == all ]]; then
        tests=("${ALL_TESTS[@]}")
    fi
    local test_name
    for test_name in "${tests[@]}"; do
        if [[ " ${ALL_TESTS[*]} " != *" $test_name "* ]]; then
            echo "Unknown test: '$test_name'. Valid tests: ${ALL_TESTS[*]}" >&2
            exit 2
        fi
    done

    check_ready_to_test
    prepare_result_folder

    # Always stop the background sampler if the script is interrupted.
    trap 'kill $(jobs -p) 2>/dev/null || true' EXIT

    local started finished
    started="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    log "Testing $BASE_URL - tests: ${tests[*]}"
    log "Results folder: $RESULT_DIR"
    warm_up

    for test_name in "${tests[@]}"; do
        "test_$test_name"
    done

    finished="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    write_report "${tests[*]}" "$started" "$finished"

    echo
    sed -n '/^## Summary/,/^# Details/p' "$RESULT_DIR/report.md" | sed '$d'
    echo
    ok "Report saved: $RESULT_DIR/report.md"

    # Exit code: 0 when nothing failed, otherwise 1.
    if (( FAIL_COUNT > 0 )); then
        exit 1
    fi
}

main "$@"
