#!/usr/bin/env bash
# =============================================================================
#  test-nfs.sh - measure NFS server performance, one metric at a time, and
#                save an easy-to-read report
# =============================================================================
#
#  USAGE (as root, after ./start-nfs.sh)
#      ./test-nfs.sh                     run every test (about 4 minutes)
#      ./test-nfs.sh seqread             run one test
#      ./test-nfs.sh latency errors      run several tests
#      ./test-nfs.sh --list              show the test names
#
#      DURATION=30 LOAD_RATE=1000 ./test-nfs.sh latency
#                                        override settings.conf for one run
#
#  THE TESTS  (explained in detail in PERFORMANCE-METRICS.md)
#      startup      time from "start server" until clients can work again
#      seqwrite     writing a big file, in MB/s
#      seqread      reading a big file, in MB/s
#      randread     4 KiB reads at random places, per second (IOPS)
#      randwrite    4 KiB writes at random places, per second (IOPS)
#      latency      time one request takes, at a steady realistic load
#      errors       requests that failed at that steady load
#      metadata     small files created (and deleted) per second
#      concurrency  how many clients can work at the same time, still fast
#      cpu          CPU used by the NFS server per request and per GB
#      memory       server memory used for each open file
#
#  OUTPUT
#      results/<date>-<time>/report.md    the human-readable report
#      results/<date>-<time>/summary.csv  one line per test (for spreadsheets)
#      results/<date>-<time>/raw/         unmodified measurement output
#      results/latest                     link to the newest result folder
#
#  EXIT CODE
#      0 = every test passed,  1 = at least one FAIL,  2 = could not run
#
#  The load comes from "fio" (the standard Linux storage benchmark) and from
#  lib/nfs-files.py. Both work on the test mount /mnt/nfs-perf-test, so every
#  request goes through the real NFS client and server of this machine.
#  No internet access is needed.
# =============================================================================

set -euo pipefail
source "$(dirname -- "${BASH_SOURCE[0]}")/lib/common.sh"

ALL_TESTS=(startup seqwrite seqread randread randwrite latency errors metadata concurrency cpu memory)

DATA_DIR="$MOUNT_POINT/fio"          # data.0, data.1, ... for the fio tests
META_DIR="$MOUNT_POINT/metadata"     # small files of the "metadata" test
HOLD_DIR="$MOUNT_POINT/open-files"   # open files of the "memory" test
GENERATOR_BUSY_PCT=90                # fio above this = it may be the limit

# Filled in by the functions below.
RESULT_DIR=""
RAW_DIR=""
DETAILS_FILE=""
SUMMARY_ROWS=()     # "Test|Measured|Target|Verdict"
FAIL_COUNT=0
GENERATOR_LIMITED_TESTS=()


# =============================================================================
#  PART 1 - small helpers
# =============================================================================

# Floating-point math and comparisons (bash itself only knows whole numbers).
calc()          { awk "BEGIN { printf \"%.1f\", $* }"; }
calc2()         { awk "BEGIN { printf \"%.2f\", $* }"; }
calc3()         { awk "BEGIN { printf \"%.3f\", $* }"; }
is_at_most()    { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a <= b) }'; }
is_at_least()   { awk -v a="$1" -v b="$2" 'BEGIN { exit !(a >= b) }'; }
now_seconds()   { date +%s.%N; }

# Turn a comparison into the words PASS / FAIL.
verdict_at_most()  { if is_at_most  "$1" "$2"; then echo PASS; else echo FAIL; fi; }
verdict_at_least() { if is_at_least "$1" "$2"; then echo PASS; else echo FAIL; fi; }

# Remember one line for the summary table at the top of the report.
#   record_result "Test name" "measured value" "target" PASS|FAIL|INFO
record_result() {
    local name="$1" measured="$2" target="$3" verdict="$4"
    SUMMARY_ROWS+=("$name|$measured|$target|$verdict")
    [[ $verdict == FAIL ]] && FAIL_COUNT=$(( FAIL_COUNT + 1 ))

    local colour='1;32'                       # green  = PASS
    [[ $verdict == FAIL ]] && colour='1;31'   # red    = FAIL
    [[ $verdict == INFO ]] && colour='1;36'   # cyan   = INFO
    printf "    => \033[${colour}m%-4s\033[0m  %s   (target: %s)\n" "$verdict" "$measured" "$target"
}

# Append Markdown text (read from stdin) to the "details" part of the report.
add_details() {
    cat >> "$DETAILS_FILE"
}

section_title() {
    echo
    printf '\033[1m==> %s\033[0m\n' "$*"
}


# =============================================================================
#  PART 2 - running fio and reading its results
# =============================================================================
#
#  run_fio <name> [fio options ...]
#
#  Runs fio once on the test mount and saves:
#      raw/<name>.json   fio's complete result
#      raw/<name>.txt    the values we use, as "key = value" lines
#
#  Next to fio's own numbers, <name>.txt also gets what the NFS server did
#  during the run: CPU time of the nfsd threads, and the RPC counters.
#
#  A run that already happened in this test session is not repeated (the
#  "cpu" test re-uses the "randread" and "seqread" runs, for example).
#
#  Read single values afterwards with:  value <key>
#  Example:  run_fio randread --rw=randread --bs=4k
#            value read_iops          ->  45210.713
# -----------------------------------------------------------------------------
LOAD_OUTPUT=""

# Options every fio run uses:
#   --direct=1             bypass the client's page cache, so every read and
#                          write really travels to the NFS server
#   --ioengine=libaio      keep several requests in flight (--iodepth)
#   --time_based           run for exactly DURATION seconds
#   --lat_percentiles=1    record the complete time of every request
#   --continue_on_error    count failed requests instead of stopping
fio_common_options() {
    printf '%s\n' \
        --directory="$DATA_DIR" \
        --ioengine=libaio \
        --direct=1 \
        --time_based \
        --runtime="$DURATION" \
        --group_reporting \
        --lat_percentiles=1 \
        --percentile_list=50:90:95:99:99.9 \
        --continue_on_error=all \
        --output-format=json
}

run_fio() {
    local name="$1"
    shift
    LOAD_OUTPUT="$RAW_DIR/$name.txt"
    [[ -s $LOAD_OUTPUT ]] && return 0             # already measured

    local common_options=()
    mapfile -t common_options < <(fio_common_options)

    printf '    %-28s ' "$name"
    local cpu_before counters_before
    cpu_before="$(nfsd_cpu_seconds)"
    counters_before="$(rpc_counters)"

    if ! fio --name="$name" "${common_options[@]}" "$@" \
            --output="$RAW_DIR/$name.json" 2> "$RAW_DIR/$name-errors.txt"; then
        echo "failed"
        die "fio failed. See $RAW_DIR/$name-errors.txt"
    fi
    [[ -s $RAW_DIR/$name-errors.txt ]] || rm -f "$RAW_DIR/$name-errors.txt"

    local cpu_after counters_after
    cpu_after="$(nfsd_cpu_seconds)"
    counters_after="$(rpc_counters)"

    {
        python3 "$FIO_TO_VALUES" "$RAW_DIR/$name.json"
        if [[ $cpu_before == n/a ]]; then
            echo "nfsd_cpu_seconds         = n/a"
        else
            echo "nfsd_cpu_seconds         = $(calc3 "$cpu_after - $cpu_before")"
        fi
        counter_differences "$counters_before" "$counters_after"
    } > "$LOAD_OUTPUT"

    printf '%10.0f IOPS %8.1f MB/s   %s failed\n' \
        "$(calc "$(value read_iops) + $(value write_iops)")" \
        "$(calc "$(value read_mb_per_s) + $(value write_mb_per_s)")" "$(value errors)"

    # Note it when fio itself was almost fully busy: then the result may
    # show the limit of the load generator, not of the NFS server.
    if is_at_least "$(calc "$(value fio_cpu_user_pct) + $(value fio_cpu_system_pct)")" "$GENERATOR_BUSY_PCT"; then
        GENERATOR_LIMITED_TESTS+=("$name")
        warn "fio was ${GENERATOR_BUSY_PCT}%+ busy in '$name'; the server may be faster than measured."
    fi
}

# One value from the last run output.  value <key>  (missing -> "0")
value() {
    awk -v key="$1" '$1 == key { print $3; found = 1 } END { if (!found) print 0 }' "$LOAD_OUTPUT"
}

# A latency value, rounded to 3 decimals (NFS requests often take < 1 ms).
ms() {
    calc3 "$(value "$1")"
}

# The first few error messages of the last run, as a Markdown list.
error_examples() {
    grep '^# error example:' "$LOAD_OUTPUT" | sed 's/^# error example: /- /' || true
}

# RPC counters of the NFS server and of the NFS client on this machine:
#   /proc/net/rpc/nfsd  "rpc <calls> <bad calls> ..."   (server)
#   /proc/net/rpc/nfs   "rpc <calls> <retransmissions> ..." (client)
# Printed as one line: server_calls server_bad_calls client_calls client_retrans
rpc_counters() {
    local server client
    server="$(awk '$1 == "rpc" { print $2, $3 }' /proc/net/rpc/nfsd 2>/dev/null)"
    client="$(awk '$1 == "rpc" { print $2, $3 }' /proc/net/rpc/nfs  2>/dev/null)"
    echo "${server:-0 0} ${client:-0 0}"
}

counter_differences() {
    local before=($1) after=($2)
    echo "server_rpc_calls         = $(( after[0] - before[0] ))"
    echo "server_rpc_bad_calls     = $(( after[1] - before[1] ))"
    echo "client_rpc_calls         = $(( after[2] - before[2] ))"
    echo "client_retransmissions   = $(( after[3] - before[3] ))"
}


# =============================================================================
#  PART 3 - test data and caches
# =============================================================================

# Create data.0 ... data.<SEQ_STREAMS-1> (DATA_FILE_MB each) if they are
# missing. fio only writes what is missing, so this is quick the next time.
ensure_data_files() {
    mkdir -p "$DATA_DIR"
    local wanted_bytes=$(( DATA_FILE_MB * 1024 * 1024 ))
    local stream missing=no
    for (( stream = 0; stream < SEQ_STREAMS; stream++ )); do
        [[ $(stat -c %s "$DATA_DIR/data.$stream" 2>/dev/null || echo 0) -ge $wanted_bytes ]] || missing=yes
    done
    [[ $missing == no ]] && return 0

    log "Creating the test data files ($SEQ_STREAMS x $DATA_FILE_MB MB) on the NFS mount"
    fio --name=create-data-files --directory="$DATA_DIR" \
        --filename_format='data.$jobnum' --numjobs="$SEQ_STREAMS" \
        --size="${DATA_FILE_MB}M" --rw=write --bs=1M --create_only=1 \
        > "$RAW_DIR/create-data-files.txt" 2>&1 ||
        die "Could not create the test data files. See $RAW_DIR/create-data-files.txt"
}

# With DROP_CACHES=yes: write all cached data to disk and empty the page
# cache, so that the next read test has to read from disk.
maybe_drop_caches() {
    [[ $DROP_CACHES == yes ]] || return 0
    sync
    echo 3 > /proc/sys/vm/drop_caches
    log "Page cache emptied (DROP_CACHES=yes): reads come from disk"
}

# Server state memory: the kernel keeps its NFSv4 bookkeeping (clients,
# open files, locks, delegations, reply cache) in "slab" caches whose names
# start with "nfsd". Total bytes in use, from /proc/slabinfo.
nfsd_slab_bytes() {
    awk '$1 ~ /^nfsd/ { total += $2 * $4 } END { print total + 0 }' /proc/slabinfo
}

# Memory of the whole kernel in slab caches (server AND client), in kB.
kernel_slab_kb() {
    awk '$1 == "Slab:" { print $2 }' /proc/meminfo
}

# Open-file "states" the NFSv4 server keeps for all clients.
nfsd_open_states() {
    cat /proc/fs/nfsd/clients/*/states 2>/dev/null | grep -c 'type: open' || true
}


# =============================================================================
#  PART 4 - the tests, one function per metric
# =============================================================================

# -----------------------------------------------------------------------------
test_startup() {
    section_title "Startup time  ($STARTUP_ROUNDS restart(s))"

    # A restart pauses every client of this server. Do not do that to a
    # server that also shares other folders, unless allowed.
    local others
    others="$(other_exports)"
    if [[ -n $others && $ALLOW_SERVER_RESTART != yes ]]; then
        warn "Skipped: the NFS server also exports other folders ($(echo $others))."
        warn "Set ALLOW_SERVER_RESTART=yes to restart it anyway."
        record_result "Startup time" "skipped (server has other exports)" \
            "<= $TARGET_STARTUP_MAX_MS ms" INFO
        add_details <<EOF
## Startup time

Skipped: this NFS server also exports $(echo $others), and restarting it
would pause those clients. Run \`ALLOW_SERVER_RESTART=yes ./test-nfs.sh startup\`
to measure it anyway.

EOF
        return
    fi

    local round start_time answer_time recovered_time
    local answer_ms=() recovery_s=() grace_notes=()
    for (( round = 1; round <= STARTUP_ROUNDS; round++ )); do
        nfs_stop
        # Remember how long the kernel log is, to read only the new lines later.
        local log_lines_before
        log_lines_before="$(dmesg 2>/dev/null | wc -l)"

        # 1) Start, and wait until the server answers an RPC "ping".
        start_time="$(now_seconds)"
        nfs_start
        until nfs_answers; do
            sleep 0.01
            if is_at_least "$(calc "$(now_seconds) - $start_time")" 60; then
                die "The NFS server did not answer within 60 s after restart."
            fi
        done
        answer_time="$(now_seconds)"

        # 2) The client (our mount) creates a new file. This works only
        #    after the client has noticed the restart, recovered its
        #    state, and the server has ended its "grace period".
        local probe="$MOUNT_POINT/.startup-probe-$round"
        if ! timeout "$RECOVERY_TIMEOUT" touch "$probe"; then
            die "The client could not create a file within $RECOVERY_TIMEOUT s after restart."
        fi
        recovered_time="$(now_seconds)"
        rm -f "$probe"

        # What the kernel said about the grace period during this round.
        local kernel_log="$RAW_DIR/startup-round-$round-kernel-log.txt" note
        dmesg 2>/dev/null | tail -n +"$(( log_lines_before + 1 ))" | grep -iE 'nfsd|nfs' > "$kernel_log" || true
        note="$(grep -oE 'NFSD: (starting [0-9]+-second grace period|no clients to reclaim[^(]*|all clients done reclaiming[^(]*)' "$kernel_log" | tail -1 || true)"

        answer_ms+=("$(calc "($answer_time - $start_time) * 1000")")
        recovery_s+=("$(calc2 "$recovered_time - $start_time")")
        grace_notes+=("${note:-no kernel message found}")
        printf '    restart %d: answers after %s ms, clients work again after %s s\n' \
            "$round" "${answer_ms[-1]}" "${recovery_s[-1]}"
    done

    local average_ms slowest_recovery
    average_ms="$(printf '%s\n' "${answer_ms[@]}" | awk '{ s += $1 } END { printf "%.1f", s / NR }')"
    slowest_recovery="$(printf '%s\n' "${recovery_s[@]}" | sort -n | tail -1)"

    record_result "Startup time" "$average_ms ms average" "<= $TARGET_STARTUP_MAX_MS ms" \
        "$(verdict_at_most "$average_ms" "$TARGET_STARTUP_MAX_MS")"
    record_result "Recovery after restart" "$slowest_recovery s (slowest)" \
        "<= $TARGET_RECOVERY_MAX_S s" \
        "$(verdict_at_most "$slowest_recovery" "$TARGET_RECOVERY_MAX_S")"

    add_details <<EOF
## Startup time and recovery after a restart

The NFS server was stopped and started $STARTUP_ROUNDS time(s) while the test
export stayed mounted. Two times were measured from the start command:

- **Answers**: the server answers an RPC "ping" (NULL call) on port 2049.
- **Clients work again**: the mounted client could create a new file. Before
  that, the client must notice the restart and recover its NFSv4 state, and
  the server must end its *grace period* (see PERFORMANCE-METRICS.md 4.1).

| Round | Answers (ms) | Clients work again (s) | Kernel message about the grace period |
|------:|-------------:|-----------------------:|---|
$(for i in "${!answer_ms[@]}"; do printf '| %5d | %12s | %22s | %s |\n' $(( i + 1 )) "${answer_ms[$i]}" "${recovery_s[$i]}" "${grace_notes[$i]}"; done)

Average time until the server answers: **$average_ms ms**.
Slowest time until clients could work again: **$slowest_recovery s**.

Kernel log of each round: \`raw/startup-round-*-kernel-log.txt\`

EOF
}

# -----------------------------------------------------------------------------
test_seqwrite() {
    section_title "Sequential write  ($SEQ_STREAMS stream(s), 1 MiB blocks, $SEQ_QUEUE_DEPTH in flight)"
    mkdir -p "$DATA_DIR"
    run_fio seqwrite --rw=write --bs=1M --iodepth="$SEQ_QUEUE_DEPTH" \
        --numjobs="$SEQ_STREAMS" --filename_format='data.$jobnum' \
        --size="${DATA_FILE_MB}M" --end_fsync=1

    local speed verdict
    speed="$(calc "$(value write_mb_per_s)")"
    verdict="$(verdict_at_least "$speed" "$TARGET_SEQWRITE_MIN_MBPS")"
    (( $(value errors) > 0 )) && verdict=FAIL

    record_result "Sequential write" "$speed MB/s" ">= $TARGET_SEQWRITE_MIN_MBPS MB/s" "$verdict"

    add_details <<EOF
## Sequential write

$SEQ_STREAMS stream(s) wrote ${DATA_FILE_MB} MB files from start to end, again and again,
for $DURATION seconds, in 1 MiB requests with $SEQ_QUEUE_DEPTH requests in flight.
This is like copying a big file (a backup, an ISO image) to the server.

| Item | Value |
|---|---:|
| **Speed** | **$speed MB/s** |
| Same, in network units | $(calc2 "$(value write_mb_per_s) * 8 * 1.048576 / 1000") Gbit/s |
| Data written | $(calc "$(value write_mb_total)") MB |
| Time per 1 MiB request: average / p99 | $(ms write_lat_avg_ms) / $(ms write_lat_p99_ms) ms |
| Failed requests | $(value errors) |
| nfsd CPU time during the run | $(value nfsd_cpu_seconds) s |

Export option: \`$EXPORT_OPTIONS\`. With \`sync\` the server writes the data to
disk before it answers, so this speed is limited by the disk.

EOF
}

# -----------------------------------------------------------------------------
test_seqread() {
    section_title "Sequential read  ($SEQ_STREAMS stream(s), 1 MiB blocks, $SEQ_QUEUE_DEPTH in flight)"
    ensure_data_files
    maybe_drop_caches
    run_fio seqread --rw=read --bs=1M --iodepth="$SEQ_QUEUE_DEPTH" \
        --numjobs="$SEQ_STREAMS" --filename_format='data.$jobnum' \
        --size="${DATA_FILE_MB}M"

    local speed verdict
    speed="$(calc "$(value read_mb_per_s)")"
    verdict="$(verdict_at_least "$speed" "$TARGET_SEQREAD_MIN_MBPS")"
    (( $(value errors) > 0 )) && verdict=FAIL

    record_result "Sequential read" "$speed MB/s" ">= $TARGET_SEQREAD_MIN_MBPS MB/s" "$verdict"

    add_details <<EOF
## Sequential read

$SEQ_STREAMS stream(s) read ${DATA_FILE_MB} MB files from start to end, again and again,
for $DURATION seconds, in 1 MiB requests with $SEQ_QUEUE_DEPTH requests in flight.

| Item | Value |
|---|---:|
| **Speed** | **$speed MB/s** |
| Same, in network units | $(calc2 "$(value read_mb_per_s) * 8 * 1.048576 / 1000") Gbit/s |
| Data read | $(calc "$(value read_mb_total)") MB |
| Time per 1 MiB request: average / p99 | $(ms read_lat_avg_ms) / $(ms read_lat_p99_ms) ms |
| Failed requests | $(value errors) |
| Page cache emptied before the test | $DROP_CACHES |

$(if [[ $DROP_CACHES == yes ]]; then
    echo "The page cache was emptied first, so the data came from disk (at least on"
    echo "the first pass through the file)."
  else
    echo "The data was served from the server's memory (page cache), so this is the"
    echo "speed of NFS itself, not of the disk. Run with \`DROP_CACHES=yes\` to include the disk."
  fi)
The test runs over the loopback interface, so no network card limits the
speed; over a real network the link usually decides (10 Gbit/s = about 1180 MB/s).

EOF
}

# -----------------------------------------------------------------------------
test_randread() {
    section_title "Random read  (4 KiB, $RANDOM_JOBS jobs x $RANDOM_QUEUE_DEPTH in flight)"
    ensure_data_files
    maybe_drop_caches
    run_fio randread --rw=randread --bs=4k --iodepth="$RANDOM_QUEUE_DEPTH" \
        --numjobs="$RANDOM_JOBS" --filename=data.0 --size="${DATA_FILE_MB}M"

    local iops verdict
    iops="$(calc "$(value read_iops)")"
    verdict="$(verdict_at_least "$iops" "$TARGET_RANDREAD_MIN_IOPS")"
    (( $(value errors) > 0 )) && verdict=FAIL

    record_result "Random read" "$iops IOPS" ">= $TARGET_RANDREAD_MIN_IOPS IOPS" "$verdict"

    add_details <<EOF
## Random read (IOPS)

$RANDOM_JOBS jobs read 4 KiB blocks at random places in a ${DATA_FILE_MB} MB file, each
job with $RANDOM_QUEUE_DEPTH requests in flight, for $DURATION seconds. Every block is
one NFS READ request, so this shows how many requests per second the server
can handle - like a database or many users opening small parts of files.

| Item | Value |
|---|---:|
| **Reads per second (IOPS)** | **$iops** |
| Data rate | $(calc "$(value read_mb_per_s)") MB/s |
| Time per request: average / p95 / p99 | $(ms read_lat_avg_ms) / $(ms read_lat_p95_ms) / $(ms read_lat_p99_ms) ms |
| Failed requests | $(value errors) |
| NFS requests the server received (RPC calls) | $(value server_rpc_calls) |
| Page cache emptied before the test | $DROP_CACHES |

EOF
}

# -----------------------------------------------------------------------------
test_randwrite() {
    section_title "Random write  (4 KiB, $RANDOM_JOBS jobs x $RANDOM_QUEUE_DEPTH in flight)"
    ensure_data_files
    run_fio randwrite --rw=randwrite --bs=4k --iodepth="$RANDOM_QUEUE_DEPTH" \
        --numjobs="$RANDOM_JOBS" --filename=data.0 --size="${DATA_FILE_MB}M"

    local iops verdict
    iops="$(calc "$(value write_iops)")"
    verdict="$(verdict_at_least "$iops" "$TARGET_RANDWRITE_MIN_IOPS")"
    (( $(value errors) > 0 )) && verdict=FAIL

    record_result "Random write" "$iops IOPS" ">= $TARGET_RANDWRITE_MIN_IOPS IOPS" "$verdict"

    add_details <<EOF
## Random write (IOPS)

$RANDOM_JOBS jobs wrote 4 KiB blocks at random places in a ${DATA_FILE_MB} MB file, each
job with $RANDOM_QUEUE_DEPTH requests in flight, for $DURATION seconds.

| Item | Value |
|---|---:|
| **Writes per second (IOPS)** | **$iops** |
| Data rate | $(calc "$(value write_mb_per_s)") MB/s |
| Time per request: average / p95 / p99 | $(ms write_lat_avg_ms) / $(ms write_lat_p95_ms) / $(ms write_lat_p99_ms) ms |
| Failed requests | $(value errors) |
| NFS requests the server received (RPC calls) | $(value server_rpc_calls) |

Each 4 KiB write is one *stable* NFS WRITE: the server saves it on disk
before it answers (export option \`${EXPORT_OPTIONS}\`). So this number depends a
lot on how fast the disk can make data safe (flush).

EOF
}

# -----------------------------------------------------------------------------
#  run_steady_load: LOAD_RATE requests per second (70 % reads, 30 % writes,
#  4 KiB, one at a time, arriving at random moments like real users).
#  Shared by the "latency" and "errors" tests, so it runs only once.
# -----------------------------------------------------------------------------
run_steady_load() {
    ensure_data_files
    local read_rate=$(( LOAD_RATE * 70 / 100 ))
    local write_rate=$(( LOAD_RATE - read_rate ))
    run_fio "steady-load-${LOAD_RATE}-per-s" --rw=randrw --rwmixread=70 --bs=4k \
        --iodepth=1 --numjobs=1 --filename=data.0 --size="${DATA_FILE_MB}M" \
        --rate_iops="$read_rate,$write_rate" --rate_process=poisson
}

test_latency() {
    section_title "Latency  ($LOAD_RATE requests per second, 70% read / 30% write)"
    run_steady_load

    local verdict=PASS
    is_at_most "$(value read_lat_p95_ms)"  "$TARGET_LATENCY_P95_MAX_MS" || verdict=FAIL
    is_at_most "$(value read_lat_p99_ms)"  "$TARGET_LATENCY_P99_MAX_MS" || verdict=FAIL
    is_at_most "$(value write_lat_p95_ms)" "$TARGET_LATENCY_P95_MAX_MS" || verdict=FAIL
    is_at_most "$(value write_lat_p99_ms)" "$TARGET_LATENCY_P99_MAX_MS" || verdict=FAIL

    record_result "Latency (p95 / p99)" \
        "read $(ms read_lat_p95_ms) / $(ms read_lat_p99_ms) ms, write $(ms write_lat_p95_ms) / $(ms write_lat_p99_ms) ms" \
        "<= $TARGET_LATENCY_P95_MAX_MS / $TARGET_LATENCY_P99_MAX_MS ms" "$verdict"

    add_details <<EOF
## Latency (how long one request takes)

A steady load of $LOAD_RATE requests per second for $DURATION seconds: 70% reads and
30% writes of 4 KiB at random places, one request at a time, arriving at
random moments (like many independent users). The server is far from full,
so this is the waiting time users feel in normal work.
"p95 = 0.5 ms" means 95 of every 100 requests were done within 0.5 ms.

| Statistic | Read (ms) | Write (ms) |
|---|---:|---:|
| Average      | $(ms read_lat_avg_ms) | $(ms write_lat_avg_ms) |
| p50 (median) | $(ms read_lat_p50_ms) | $(ms write_lat_p50_ms) |
| p90          | $(ms read_lat_p90_ms) | $(ms write_lat_p90_ms) |
| p95          | **$(ms read_lat_p95_ms)** | **$(ms write_lat_p95_ms)** |
| p99          | **$(ms read_lat_p99_ms)** | **$(ms write_lat_p99_ms)** |
| p99.9        | $(ms read_lat_p999_ms) | $(ms write_lat_p999_ms) |
| Slowest      | $(ms read_lat_max_ms) | $(ms write_lat_max_ms) |
| Requests     | $(value read_requests) | $(value write_requests) |

- **Read**: one NFS READ request; answered from the server's memory when the
  block is cached, otherwise from disk.
- **Write**: one stable NFS WRITE; the server answers after the data is safe on disk.

Full fio result: \`raw/steady-load-${LOAD_RATE}-per-s.json\`

EOF
}

# -----------------------------------------------------------------------------
test_errors() {
    section_title "Errors  ($LOAD_RATE requests per second)"
    run_steady_load

    local failed_pct
    failed_pct="$(calc3 "$(value error_pct)")"
    local requests=$(( $(value read_requests) + $(value write_requests) ))

    record_result "Errors (failed requests)" "$failed_pct% ($(value errors) of $requests)" \
        "<= $TARGET_ERRORS_MAX_PCT%" \
        "$(verdict_at_most "$failed_pct" "$TARGET_ERRORS_MAX_PCT")"

    add_details <<EOF
## Errors

The same steady load as the latency test ($LOAD_RATE requests per second for
$DURATION seconds). A request that returned an error counts as failed. The
test mount is a "hard" mount (the RHEL default): the client never gives up
and repeats a request the server did not answer, so problems show up as
**retransmissions** and slow requests rather than as errors.

| Item | Value |
|---|---:|
| Requests sent | $requests |
| **Failed requests** | **$(value errors) ($failed_pct%)** |
| First error code (0 = none) | $(value first_error_code) |
| Client: requests sent again (retransmissions, should be 0) | $(value client_retransmissions) |
| Server: malformed or refused requests (bad calls, should be 0) | $(value server_rpc_bad_calls) |
| Server: RPC requests received | $(value server_rpc_calls) |

Retransmissions mean the server did not answer in time (60 s by default,
mount option \`timeo\`), which usually points to an overloaded server or a
network problem.

EOF
}

# -----------------------------------------------------------------------------
test_metadata() {
    section_title "Metadata  ($WORKERS workers create ${SMALL_FILE_KB} KB files, then delete them)"
    LOAD_OUTPUT="$RAW_DIR/metadata.txt"
    rm -rf "$META_DIR"
    printf '    %-28s ' "metadata"
    if ! python3 "$FILE_TOOL" create --dir "$META_DIR" --workers "$WORKERS" \
            --duration "$DURATION" --file-kb "$SMALL_FILE_KB" \
            --output "$LOAD_OUTPUT" 2> "$RAW_DIR/metadata-errors.txt"; then
        echo "failed"
        die "lib/nfs-files.py failed. See $RAW_DIR/metadata-errors.txt"
    fi
    [[ -s $RAW_DIR/metadata-errors.txt ]] || rm -f "$RAW_DIR/metadata-errors.txt"
    rm -rf "$META_DIR"
    printf '%10.0f creates/s %8.0f deletes/s   %s failed\n' \
        "$(value creates_per_s)" "$(value deletes_per_s)" "$(value errors)"

    local rate verdict
    rate="$(calc "$(value creates_per_s)")"
    verdict="$(verdict_at_least "$rate" "$TARGET_METADATA_MIN_PER_S")"
    (( $(value errors) > 0 )) && verdict=FAIL

    record_result "Small files (metadata)" "$rate files created/s" \
        ">= $TARGET_METADATA_MIN_PER_S files/s" "$verdict"

    add_details <<EOF
## Small files (metadata operations)

$WORKERS workers created ${SMALL_FILE_KB} KB files as fast as they could for $DURATION seconds
(open a new file, write, close), then deleted all of them. For NFS every
small file costs several round trips - OPEN (create), WRITE, CLOSE, and later
REMOVE - and the server must save the new file and folder entry safely. Workloads with
many small files (source code, mail folders, home folders) depend on this
much more than on MB/s.

| Item | Create | Delete |
|---|---:|---:|
| **Files per second** | **$rate** | $(calc "$(value deletes_per_s)") |
| Files done | $(value creates) | $(value deletes) |
| Failed | $(value creates_failed) | $(value deletes_failed) |
| Time per file: average | $(ms create_avg_ms) ms | $(ms delete_avg_ms) ms |
| Time per file: p95 | $(ms create_p95_ms) ms | $(ms delete_p95_ms) ms |
| Time per file: p99 | $(ms create_p99_ms) ms | $(ms delete_p99_ms) ms |
| Slowest | $(ms create_max_ms) ms | $(ms delete_max_ms) ms |

$(error_examples)

EOF
}

# -----------------------------------------------------------------------------
test_concurrency() {
    section_title "Concurrency  (clients reading at the same time: $CONCURRENCY_LEVELS)"
    ensure_data_files
    local clients table="" capacity=0 all_kept_up=yes

    for clients in $CONCURRENCY_LEVELS; do
        run_fio "concurrency-${clients}-clients" --rw=randread --bs=4k --iodepth=1 \
            --numjobs="$clients" --filename=data.0 --size="${DATA_FILE_MB}M"

        local iops p99 average failed kept_up=yes
        iops="$(calc "$(value read_iops)")"
        average="$(ms read_lat_avg_ms)"
        p99="$(ms read_lat_p99_ms)"
        failed="$(value errors)"

        # A step "keeps up" when no request failed and 99% of the requests
        # were still answered within the latency target.
        (( failed == 0 )) || kept_up=no
        is_at_most "$p99" "$TARGET_LATENCY_P99_MAX_MS" || kept_up=no

        if [[ $kept_up == yes && $all_kept_up == yes ]]; then
            capacity="$clients"
        else
            all_kept_up=no
        fi

        table+="$(printf '| %7s | %10s | %9s | %9s | %6s | %-7s |' \
            "$clients" "$iops" "$average" "$p99" "$failed" "$kept_up")"$'\n'
    done

    local measured="$capacity clients with p99 <= $TARGET_LATENCY_P99_MAX_MS ms"
    [[ $all_kept_up == yes ]] && measured="every step OK (>= $capacity clients)"
    [[ $capacity == 0 ]] && measured="too slow already at the first step"

    record_result "Concurrency" "$measured" ">= $TARGET_CONCURRENCY_MIN clients" \
        "$(verdict_at_least "$capacity" "$TARGET_CONCURRENCY_MIN")"

    add_details <<EOF
## Concurrency

More and more clients read 4 KiB blocks at random places at the same time,
$DURATION s per step. Each client sends one request and waits for the answer
before it sends the next, like a program reading a file. The server has
**$(nfsd_thread_count) nfsd threads**, so at most that many requests are worked on at once;
the rest wait in a queue. A step "keeps up" when no request failed and the
p99 time stayed within $TARGET_LATENCY_P99_MAX_MS ms.

| Clients | Total IOPS | Avg (ms) | p99 (ms) | Failed | Kept up |
|--------:|-----------:|---------:|---------:|-------:|:--------|
${table}
Result: **$measured**.

- **Total IOPS** grows with the clients until the nfsd threads (or the CPU)
  are all busy; after that it stays flat and the time per request grows.
- If the time per request grows early, try more nfsd threads
  (\`NFSD_THREADS\` in settings.conf, see PERFORMANCE-METRICS.md section 7).
$(if [[ $all_kept_up == yes ]]; then
    echo
    echo "The server kept up at every step. To find its real limit, add more clients,"
    echo "for example: CONCURRENCY_LEVELS=\"$CONCURRENCY_LEVELS 256\" ./test-nfs.sh concurrency"
  fi)

All clients share one NFS mount (one TCP connection, unless NCONNECT > 1).

EOF
}

# -----------------------------------------------------------------------------
test_cpu() {
    section_title "CPU usage of the NFS server (nfsd threads)"
    ensure_data_files

    if [[ $(nfsd_cpu_seconds) == n/a ]]; then
        warn "The nfsd kernel threads are not visible (inside a container? use --pid=host)."
        record_result "CPU per request" "not measurable here" \
            "<= $TARGET_CPU_MAX_US_PER_OP us per 4 KiB read" INFO
        add_details <<EOF
## CPU usage

Not measured: the nfsd kernel threads are not visible from here. Inside a
container, start it with \`--pid=host\`.

EOF
        return
    fi

    # Re-use the randread and seqread runs (they are run now if needed).
    maybe_drop_caches
    run_fio randread --rw=randread --bs=4k --iodepth="$RANDOM_QUEUE_DEPTH" \
        --numjobs="$RANDOM_JOBS" --filename=data.0 --size="${DATA_FILE_MB}M"
    local random_cpu random_requests random_iops
    random_cpu="$(value nfsd_cpu_seconds)"
    random_requests="$(value read_requests)"
    random_iops="$(calc "$(value read_iops)")"

    maybe_drop_caches
    run_fio seqread --rw=read --bs=1M --iodepth="$SEQ_QUEUE_DEPTH" \
        --numjobs="$SEQ_STREAMS" --filename_format='data.$jobnum' --size="${DATA_FILE_MB}M"
    local sequential_cpu sequential_gb
    sequential_cpu="$(value nfsd_cpu_seconds)"
    sequential_gb="$(calc3 "$(value read_mb_total) / 1024")"

    local us_per_op seconds_per_gb cores_busy
    us_per_op="$(awk -v c="$random_cpu" -v n="$random_requests" 'BEGIN { printf "%.1f", (n > 0 ? c * 1000000 / n : 0) }')"
    seconds_per_gb="$(awk -v c="$sequential_cpu" -v g="$sequential_gb" 'BEGIN { printf "%.3f", (g > 0 ? c / g : 0) }')"
    cores_busy="$(calc2 "$random_cpu / $DURATION")"

    record_result "CPU per request" "$us_per_op us per 4 KiB read ($cores_busy cores busy at $random_iops IOPS)" \
        "<= $TARGET_CPU_MAX_US_PER_OP us" \
        "$(verdict_at_most "$us_per_op" "$TARGET_CPU_MAX_US_PER_OP")"

    add_details <<EOF
## CPU usage

CPU time of the nfsd kernel threads (the NFS server itself), measured during
the random-read run (many small requests) and the sequential-read run (big
requests). The NFS server does its work in the kernel, so it does not show
up as a normal process in \`top\`; its threads are called \`nfsd\`.

| Item | Value |
|---|---:|
| **nfsd CPU per 4 KiB read** | **$us_per_op microseconds** |
| nfsd CPU during the random-read run | $random_cpu s in $DURATION s = $cores_busy cores busy |
| Requests in that run | $random_requests ($random_iops per second) |
| nfsd CPU per GB read sequentially | $seconds_per_gb s |
| CPU cores in this machine | $(nproc) |

Estimate the CPU a server needs: **cores = requests per second x
microseconds per request / 1,000,000**. Big requests (1 MiB) cost much less
CPU per byte than small ones.

Not included: the network stack (softirq) and the client side, which both run
on this machine too in this test.

EOF
}

# -----------------------------------------------------------------------------
test_memory() {
    section_title "Memory usage  ($MEMORY_OPEN_FILES files open at the same time)"

    # Only NFSv4 servers remember which files a client has open.
    if [[ $(nfs_major_version) == 3 ]]; then
        record_result "Memory per open file" "not applicable to NFSv3" \
            "<= $TARGET_MEMORY_MAX_KB_PER_FILE KB per file" INFO
        add_details <<EOF
## Memory usage

Not measured: the test mount uses NFSv3, which keeps no "open state" on the
server, so open files cost the server no memory. Use NFS_VERSION=4.2.

EOF
        return
    fi
    local slab_before kernel_before states_before
    slab_before="$(nfsd_slab_bytes)"
    kernel_before="$(kernel_slab_kb)"
    states_before="$(nfsd_open_states)"

    # Open the files in the background and wait until all are open.
    local hold_output="$RAW_DIR/memory-open-files.txt"
    local ready_file="$RAW_DIR/.hold-ready"
    rm -rf "$HOLD_DIR" "$ready_file"
    python3 "$FILE_TOOL" hold --dir "$HOLD_DIR" --files "$MEMORY_OPEN_FILES" \
        --ready-file "$ready_file" --output "$hold_output" 2> "$RAW_DIR/memory-errors.txt" &
    local hold_pid=$!
    while [[ ! -e $ready_file ]] && kill -0 "$hold_pid" 2>/dev/null; do
        sleep 0.2
    done
    rm -f "$ready_file"
    [[ -s $RAW_DIR/memory-errors.txt ]] || rm -f "$RAW_DIR/memory-errors.txt"
    LOAD_OUTPUT="$hold_output"
    local opened
    opened="$(value files_open)"
    printf '    %-28s %10s files open\n' "memory-open-files" "$opened"

    sleep 1                                     # let the counters settle
    local slab_busy kernel_busy states_busy
    slab_busy="$(nfsd_slab_bytes)"
    kernel_busy="$(kernel_slab_kb)"
    states_busy="$(nfsd_open_states)"

    # Close the files again (the tool deletes them).
    kill "$hold_pid" 2>/dev/null || true
    wait "$hold_pid" 2>/dev/null || true
    rm -rf "$HOLD_DIR"

    local per_file_kb kernel_per_file_kb server_mb
    if (( opened > 0 )); then
        per_file_kb="$(calc2 "($slab_busy - $slab_before) / 1024 / $opened")"
        kernel_per_file_kb="$(calc2 "($kernel_busy - $kernel_before) / $opened")"
    else
        per_file_kb=0; kernel_per_file_kb=0
    fi
    server_mb="$(calc2 "$slab_busy / 1048576")"

    local verdict
    verdict="$(verdict_at_most "$per_file_kb" "$TARGET_MEMORY_MAX_KB_PER_FILE")"
    (( opened < MEMORY_OPEN_FILES )) && verdict=FAIL

    record_result "Memory per open file" "$per_file_kb KB ($opened files: $server_mb MB nfsd state)" \
        "<= $TARGET_MEMORY_MAX_KB_PER_FILE KB per file" "$verdict"

    add_details <<EOF
## Memory usage

The client opened $MEMORY_OPEN_FILES files and kept them open. For every open file an
NFSv4 server keeps a record ("open state") in kernel memory until the client
closes it. The nfsd kernel threads have no memory of their own that \`ps\`
could show, so the test measures the kernel caches the NFS server uses (the
slab caches named \`nfsd*\` in \`/proc/slabinfo\`).

| Item | Before | $opened files open |
|---|---:|---:|
| Open-file states the server keeps | $states_before | $states_busy |
| NFS server memory (nfsd slab caches) | $(calc2 "$slab_before / 1048576") MB | **$server_mb MB** |
| All kernel slab memory (server + client + everything else) | $(calc "$kernel_before / 1024") MB | $(calc "$kernel_busy / 1024") MB |

| Per open file | Value |
|---|---:|
| **NFS server memory** | **$per_file_kb KB** |
| All kernel slab memory (includes the NFS client on this machine; noisy) | $kernel_per_file_kb KB |

Estimate the memory for many clients: **open files x KB per file**. The file
data itself is cached in the page cache, which the kernel gives back when
programs need the memory.

$(error_examples)

EOF
}


# =============================================================================
#  PART 5 - preparing, and writing the final report
# =============================================================================

# Stop with exit code 2 ("could not run") and a clear message.
cannot_run() {
    printf '\033[1;31m[ERROR ]\033[0m %s\n' "$*" >&2
    exit 2
}

check_ready_to_test() {
    [[ $EUID -eq 0 ]]              || cannot_run "Please run as root, for example:  sudo $0"
    command -v fio >/dev/null      || cannot_run "'fio' not found. Run ./start-nfs.sh first."
    command -v python3 >/dev/null  || cannot_run "'python3' not found."
    nfs_server_is_running          || cannot_run "The NFS server is not running. Run ./start-nfs.sh first."
    nfs_answers                    || cannot_run "The NFS server does not answer. Run ./start-nfs.sh first."
    test_mount_is_mounted          || cannot_run "The test export is not mounted on $MOUNT_POINT. Run ./start-nfs.sh first."
    touch "$MOUNT_POINT/.test-nfs-probe" 2>/dev/null ||
        cannot_run "Cannot write to $MOUNT_POINT. Run ./start-nfs.sh again."
    rm -f "$MOUNT_POINT/.test-nfs-probe"
}

prepare_result_folder() {
    RESULT_DIR="$KIT_DIR/results/$(date +%Y%m%d-%H%M%S)"
    RAW_DIR="$RESULT_DIR/raw"
    DETAILS_FILE="$RAW_DIR/.details.md"
    mkdir -p "$RAW_DIR"
    : > "$DETAILS_FILE"
    ln -sfn "$(basename "$RESULT_DIR")" "$KIT_DIR/results/latest"
}

# Remove what an interrupted earlier run may have left behind.
remove_leftovers() {
    rm -rf "$META_DIR" "$HOLD_DIR"
}

write_report() {
    local tests_run="$1" started="$2" finished="$3"
    local report="$RESULT_DIR/report.md"
    local os_name selinux cpu_model memory_total export_fs

    os_name="$(. /etc/os-release && echo "$PRETTY_NAME")"
    selinux="$(getenforce 2>/dev/null || echo 'not available')"
    cpu_model="$(awk -F': ' '/^model name/ { print $2; exit }' /proc/cpuinfo)"
    memory_total="$(awk '/^MemTotal/ { printf "%.1f GB", $2 / 1048576 }' /proc/meminfo)"
    export_fs="$(df --output=source,fstype "$EXPORT_DIR" | tail -1 | awk '{ print $1 " (" $2 ")" }')"

    local overall="ALL PASSED"
    (( FAIL_COUNT > 0 )) && overall="$FAIL_COUNT TEST(S) FAILED"

    {
        echo "# NFS Server Performance Test Report"
        echo
        echo "**Result: $overall**"
        echo
        echo "| | |"
        echo "|---|---|"
        echo "| Test started  | $started |"
        echo "| Test finished | $finished |"
        echo "| Host | $(hostname) |"
        echo "| Operating system | $os_name (kernel $(uname -r)) |"
        echo "| CPU | $(nproc) x $cpu_model |"
        echo "| Memory | $memory_total |"
        echo "| NFS server | kernel nfsd, $(nfs_utils_version), $(nfsd_thread_count) threads |"
        echo "| SELinux | $selinux |"
        echo "| Export | \`$EXPORT_DIR\` on $export_fs, options \`$EXPORT_OPTIONS\` |"
        echo "| Test mount | \`$MOUNT_POINT\` (client on the same machine, over 127.0.0.1) |"
        echo "| Mount options | \`$(active_mount_options)\` |"
        echo "| Tests run | $tests_run |"
        echo "| Load per run | $DURATION s; test file $DATA_FILE_MB MB; steady load $LOAD_RATE requests/s; page cache emptied before reads: $DROP_CACHES |"
        echo "| Load generator | $(fio --version) |"
        echo
        echo "## Summary"
        echo
        echo "PASS = target met, FAIL = target missed, INFO = measured only (no target)."
        echo "Targets are set in \`settings.conf\`. What each metric means: \`PERFORMANCE-METRICS.md\`."
        echo
        printf '| %-26s | %-58s | %-28s | %-7s |\n' "Test" "Measured" "Target" "Verdict"
        printf '|%s|%s|%s|%s|\n' "$(printf -- '-%.0s' {1..28})" "$(printf -- '-%.0s' {1..60})" \
                                 "$(printf -- '-%.0s' {1..30})" "$(printf -- '-%.0s' {1..9})"
        local row name measured target verdict
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            [[ $verdict == FAIL ]] && verdict="**FAIL**"
            printf '| %-26s | %-58s | %-28s | %-7s |\n' "$name" "$measured" "$target" "$verdict"
        done
        echo
        if (( ${#GENERATOR_LIMITED_TESTS[@]} > 0 )); then
            echo "> **Note:** fio (the load generator) was at least ${GENERATOR_BUSY_PCT}% busy in:"
            echo "> ${GENERATOR_LIMITED_TESTS[*]}."
            echo "> In those runs the NFS server may be faster than the numbers show."
            echo
        fi
        echo "> **Remember:** client and server run on the same machine and talk over"
        echo "> 127.0.0.1. The results show what the server can do without a network;"
        echo "> real clients also pay the network's delay and speed limit."
        echo
        echo "# Details"
        echo
        cat "$DETAILS_FILE"
        echo "## Files in this folder"
        echo
        echo "- \`report.md\` - this report"
        echo "- \`summary.csv\` - the summary table, for spreadsheets"
        echo "- \`raw/\` - fio's full JSON result of every run, the values taken from it"
        echo "  (\"key = value\" lines) and kernel messages of the startup test"
    } > "$report"

    # The same summary as CSV.
    {
        echo "test,measured,target,verdict"
        for row in "${SUMMARY_ROWS[@]}"; do
            IFS='|' read -r name measured target verdict <<< "$row"
            printf '"%s","%s","%s","%s"\n' "$name" "$measured" "$target" "$verdict"
        done
    } > "$RESULT_DIR/summary.csv"

    rm -f "$DETAILS_FILE"
}

print_usage() {
    # Print the comment block at the top of this file.
    sed -n '3,/^set -euo pipefail/{/^#/p}' "$0" | sed 's/^# \{0,1\}//'
}


# =============================================================================
#  MAIN
# =============================================================================
main() {
    case "${1:-}" in
        -h|--help) print_usage; exit 0 ;;
        --list)    printf '%s\n' "${ALL_TESTS[@]}"; exit 0 ;;
    esac

    # Which tests to run: the names given, or all of them.
    local tests=("$@")
    if (( ${#tests[@]} == 0 )) || [[ ${tests[0]} == all ]]; then
        tests=("${ALL_TESTS[@]}")
    fi
    local test_name
    for test_name in "${tests[@]}"; do
        if [[ " ${ALL_TESTS[*]} " != *" $test_name "* ]]; then
            echo "Unknown test: '$test_name'. Valid tests: ${ALL_TESTS[*]}" >&2
            exit 2
        fi
    done

    check_ready_to_test
    prepare_result_folder

    # Always stop background helpers if the script is interrupted.
    trap 'kill $(jobs -p) 2>/dev/null || true' EXIT

    local started finished
    started="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    log "Testing the NFS server through $MOUNT_POINT - tests: ${tests[*]}"
    log "Results folder: $RESULT_DIR"
    remove_leftovers

    for test_name in "${tests[@]}"; do
        "test_$test_name"
    done

    finished="$(date '+%Y-%m-%d %H:%M:%S %Z')"
    write_report "${tests[*]}" "$started" "$finished"

    echo
    sed -n '/^## Summary/,/^# Details/p' "$RESULT_DIR/report.md" | sed '$d'
    echo
    ok "Report saved: $RESULT_DIR/report.md"

    # Exit code: 0 when nothing failed, otherwise 1.
    if (( FAIL_COUNT > 0 )); then
        exit 1
    fi
}

main "$@"
