#!/bin/bash
# ---------------------------------------------------------------------------
# Test what a kernel patch did.
#
# The idea is always the same: run the tests, apply the patch, run the same
# tests again, and report only the difference.  A single run tells you very
# little, because a RHEL kernel has selftests that already fail on a given
# machine; the difference is the part the patch is responsible for.
#
# Because a kernel patch usually needs a reboot, the work is split into
# phases and the progress is kept in a state file, so the run picks up where
# it left off after the machine comes back.
#
#   ./test-patch.sh start --mode rpm --patch-dir /srv/patch-rpms
#   (reboot)
#   ./test-patch.sh continue
#
# Exit status: 0 no regression, 1 regression found, 2 usage or environment
# problem, 3 the run is paused and waiting for you.
# ---------------------------------------------------------------------------
set -uo pipefail

KT_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
# shellcheck source=lib/common.sh
. "$KT_ROOT/lib/common.sh"

MODE=manual
PATCH_DIR=""
SOURCE_DIR=""
MODULE_SRC=""
AUTO_REBOOT=0
TEST_ARGS=()
CMD=""
# Whether each of these came from the command line this time round.  What the
# baseline recorded is the fallback, not the winner: running
# "test-patch.sh apply --mode rpm" after a baseline taken in the default
# manual mode has to do what the command line says.
MODE_SET=0; PATCH_DIR_SET=0; SOURCE_DIR_SET=0; MODULE_SRC_SET=0; TEST_ARGS_SET=0

usage() {
    sed -n '3,20p' "$0" | sed 's/^# \{0,1\}//'
    cat <<'EOF'

Commands:
  start       run the baseline, apply the patch, then pause or reboot
  continue    pick up after the reboot: run the tests again and compare
  baseline    only run the "before" tests
  apply       only apply the patch
  after       only run the "after" tests
  compare     only compare the two runs already recorded
  status      show where the run has got to
  abort       forget the current run (does not undo the patch)

Options:
  -m, --mode MODE        rpm | kpatch | module | source | manual
  -d, --patch-dir DIR    rpm/kpatch: directory of .rpm files
                         source:     directory of .patch files
  -s, --source-dir DIR   source: the kernel tree to patch and build
      --module-src DIR   module: out-of-tree module source to build and load
      --auto-reboot      reboot by itself and continue automatically afterwards
      --profile NAME     passed through to test-kernel.sh
      --engines "LIST"   passed through to test-kernel.sh
      --unsafe           passed through to test-kernel.sh
  -h, --help             this text

Modes:
  rpm      Install kernel RPMs from a directory (an erratum, a z-stream
           update, a locally rebuilt kernel).  Needs a reboot.
  kpatch   Install a kpatch / livepatch RPM.  Takes effect at once, no reboot.
  module   Build one out-of-tree module against kernel-devel and load it.
           No reboot.
  source   Apply .patch files to a kernel source tree, build it, install it.
           Needs a toolchain, so this is for a build machine, not a locked
           down production host.
  manual   The script stops after the baseline and waits while you apply the
           patch however you like.  Use this when nothing else fits.
EOF
}

[ $# -gt 0 ] || { usage; exit 2; }
# The first word is the command, but -h/--help is not a command, and asking
# for help must never be an error.
case $1 in -h|--help|help) usage; exit 0 ;; esac
CMD=$1; shift
while [ $# -gt 0 ]; do
    case $1 in
        -m|--mode)       MODE=$2; MODE_SET=1; shift 2 ;;
        -d|--patch-dir)  PATCH_DIR=$2; PATCH_DIR_SET=1; shift 2 ;;
        -s|--source-dir) SOURCE_DIR=$2; SOURCE_DIR_SET=1; shift 2 ;;
        --module-src)    MODULE_SRC=$2; MODULE_SRC_SET=1; shift 2 ;;
        --auto-reboot)   AUTO_REBOOT=1; shift ;;
        --profile)       TEST_ARGS+=(--profile "$2"); TEST_ARGS_SET=1; shift 2 ;;
        --engines)       TEST_ARGS+=(--engines "$2"); TEST_ARGS_SET=1; shift 2 ;;
        --unsafe)        TEST_ARGS+=(--unsafe); TEST_ARGS_SET=1; shift ;;
        -h|--help)       usage; exit 0 ;;
        *) kt_err "unknown option: $1"; usage; exit 2 ;;
    esac
done
kt_load_settings

STATE_DIR=$PATCH_STATE_DIR
STATE=$STATE_DIR/state
RESUME_UNIT=/etc/systemd/system/kernel-test-resume.service

# --- state -----------------------------------------------------------------
state_set() {
    mkdir -p "$STATE_DIR"
    local k=$1 v=$2
    # The state file is one KEY=VALUE per line, so a value that contains a
    # newline would corrupt it and the corruption would only surface much
    # later as "the baseline run directory is gone".  Refuse it here instead.
    case $v in
        *$'\n'*) kt_die "internal error: refusing to store a multi-line value for '$k'" ;;
    esac
    touch "$STATE"
    grep -v "^$k=" "$STATE" > "$STATE.new" 2>/dev/null
    printf '%s=%s\n' "$k" "$v" >> "$STATE.new"
    mv "$STATE.new" "$STATE"
}
state_get() {
    [ -r "$STATE" ] || return 1
    local line; line=$(grep "^$1=" "$STATE" | tail -1)
    [ -n "$line" ] || return 1
    local v=${line#*=}
    # phase_baseline records patch_dir=, source_dir= and module_src= even when
    # they are empty.  Returning success with an empty string would make every
    # "x=$(state_get k) || x=$DEFAULT" fall through with x unset.
    [ -n "$v" ] || return 1
    printf '%s' "$v"
}
state_show() {
    if [ ! -r "$STATE" ]; then
        kt_log "no patch run in progress"
        return 1
    fi
    kt_head "Patch run state"
    cat "$STATE"
}

require_state() {
    [ -r "$STATE" ] || kt_die "no patch run in progress; start one with: $0 start"
}

# Put back everything the baseline recorded, except what this invocation set
# on the command line.  Without this the run after a reboot uses different
# options from the baseline, and comparing a smoke run against a standard one
# buries the answer under hundreds of "new" and "gone" tests.
restore_from_state() {
    [ -r "$STATE" ] || return 0
    local v
    [ "$MODE_SET" = 0 ]        && v=$(state_get mode)       && MODE=$v
    [ "$PATCH_DIR_SET" = 0 ]   && v=$(state_get patch_dir)  && PATCH_DIR=$v
    [ "$SOURCE_DIR_SET" = 0 ]  && v=$(state_get source_dir) && SOURCE_DIR=$v
    [ "$MODULE_SRC_SET" = 0 ]  && v=$(state_get module_src) && MODULE_SRC=$v
    if v=$(state_get test_args); then
        if [ "$TEST_ARGS_SET" = 0 ]; then
            read -r -a TEST_ARGS <<< "$v"
            [ ${#TEST_ARGS[@]} -gt 0 ] && kt_log "using the baseline's test options: ${TEST_ARGS[*]}"
        elif [ "${TEST_ARGS[*]}" != "$v" ]; then
            kt_warn "the baseline ran with: $v"
            kt_warn "this run was given:    ${TEST_ARGS[*]}"
            kt_warn "The two runs will not test the same things, and the comparison"
            kt_warn "will show the difference as tests appearing and disappearing."
        fi
    fi
    return 0
}

# A phase that fails leaves the run resumable instead of stuck.
apply_die() {
    kt_err "$*"
    state_set phase apply_failed
    exit 2
}

# --- phases ----------------------------------------------------------------
# Sets RUN_TESTS_DIR rather than printing it: capturing the function's
# stdout would swallow the progress of a run that takes an hour, and the
# whole point of watching a patch test is seeing where it has got to.
RUN_TESTS_DIR=""
run_tests() {
    local tag=$1
    kt_head "Test run: $tag"
    "$KT_ROOT/test-kernel.sh" --tag "$tag" "${TEST_ARGS[@]}"
    local rc=$?
    # 2 means the run could not happen at all; anything else produced results.
    [ "$rc" = 2 ] && kt_die "test-kernel.sh could not run"
    RUN_TESTS_DIR=$(readlink -f "$KT_RESULTS_DIR/latest")
    [ -d "$RUN_TESTS_DIR" ] || kt_die "cannot find the run directory that was just written"
}

phase_baseline() {
    if [ -r "$STATE" ] && state_get before_dir >/dev/null 2>&1; then
        kt_warn "a baseline already exists: $(state_get before_dir)"
        kt_warn "run '$0 abort' first if you want to start over"
        exit 2
    fi
    mkdir -p "$STATE_DIR"
    : > "$STATE"
    state_set started       "$(date '+%Y-%m-%d %H:%M:%S %Z')"
    state_set mode          "$MODE"
    state_set patch_dir     "$PATCH_DIR"
    state_set source_dir    "$SOURCE_DIR"
    state_set module_src    "$MODULE_SRC"
    state_set test_args     "${TEST_ARGS[*]-}"
    state_set before_kernel "$(kt_kver)"
    state_set phase         baseline

    run_tests before-patch
    state_set before_dir "$RUN_TESTS_DIR"
    state_set phase baseline_done
    kt_ok "baseline recorded: $RUN_TESTS_DIR"
}

# Snapshot the kernel packages so "apply" can say what it changed.
snapshot_pkgs() {
    rpm -qa 'kernel*' 2>/dev/null | sort
}

phase_apply() {
    require_state
    restore_from_state
    local mode=$MODE
    kt_head "Applying the patch (mode: $mode)"
    snapshot_pkgs > "$STATE_DIR/pkgs-before.txt"
    state_set phase applying

    case $mode in
        rpm)      apply_rpm ;;
        kpatch)   apply_kpatch ;;
        module)   apply_module ;;
        source)   apply_source ;;
        manual)   apply_manual ;;
        *) kt_die "unknown mode: $mode" ;;
    esac
}

apply_rpm() {
    local dir=$PATCH_DIR
    [ -n "$dir" ] || kt_die "--patch-dir is required for mode rpm"
    [ -d "$dir" ] || kt_die "no such directory: $dir"
    local rpms=( "$dir"/*.rpm )
    [ -e "${rpms[0]}" ] || kt_die "no .rpm files in $dir"
    kt_log "${#rpms[@]} RPMs in $dir"

    # "dnf install <dir>/*.rpm" on an already installed package is an upgrade
    # request and fails on anything already at that version, so the wanted
    # verb is chosen per package.  --setopt=install_weak_deps=False keeps dnf
    # from reaching for a repository that is not there on an offline machine.
    local dnf_opts=(-y --disablerepo='*' --setopt=install_weak_deps=False --nogpgcheck)
    if dnf "${dnf_opts[@]}" install "${rpms[@]}" >> "$STATE_DIR/apply.log" 2>&1; then
        kt_ok "dnf install accepted the RPMs"
    else
        kt_warn "dnf install failed; trying 'dnf upgrade' for the ones already present"
        if ! dnf "${dnf_opts[@]}" upgrade "${rpms[@]}" >> "$STATE_DIR/apply.log" 2>&1; then
            kt_err "both install and upgrade failed -- see $STATE_DIR/apply.log"
            kt_err "On an offline host this is usually a missing dependency that"
            kt_err "is not in the patch directory.  Add it and run '$0 apply'."
            state_set phase apply_failed
            exit 2
        fi
    fi
    snapshot_pkgs > "$STATE_DIR/pkgs-after.txt"
    diff -u "$STATE_DIR/pkgs-before.txt" "$STATE_DIR/pkgs-after.txt" \
        > "$STATE_DIR/pkgs.diff" 2>/dev/null
    kt_log "package changes recorded in $STATE_DIR/pkgs.diff"

    local newk
    newk=$(rpm -q --qf '%{version}-%{release}.%{arch}\n' kernel-core 2>/dev/null | sort -V | tail -1)
    if [ -n "$newk" ] && [ "$newk" != "$(kt_kver)" ]; then
        state_set expect_kernel "$newk"
        kt_ok "new kernel installed: $newk (running: $(kt_kver))"
        need_reboot
    else
        kt_warn "no new kernel package appeared; nothing to reboot into"
        kt_warn "If the patch was userspace only, that is fine."
        state_set phase applied
    fi
}

apply_kpatch() {
    local dir=$PATCH_DIR
    [ -n "$dir" ] || kt_die "--patch-dir is required for mode kpatch"
    local rpms=( "$dir"/*.rpm )
    if [ -e "${rpms[0]}" ]; then
        dnf -y --disablerepo='*' --nogpgcheck install "${rpms[@]}" \
            >> "$STATE_DIR/apply.log" 2>&1 || apply_die "installing the kpatch RPM failed -- see $STATE_DIR/apply.log"
    fi
    command -v kpatch >/dev/null 2>&1 || kt_die "kpatch is not installed on this machine"
    kpatch list >> "$STATE_DIR/apply.log" 2>&1
    # A livepatch loaded through its RPM is active as soon as it is installed.
    local n
    n=$(kpatch list 2>/dev/null | awk '/Loaded patch modules/{f=1;next} /^$/{f=0} f' | grep -c . )
    kt_ok "$n livepatch module(s) loaded"
    kpatch list > "$STATE_DIR/kpatch-list.txt" 2>&1
    [ "$n" -gt 0 ] || kt_warn "no livepatch is loaded; the 'after' run will test an unpatched kernel"
    state_set phase applied
}

apply_module() {
    local src=$MODULE_SRC
    [ -n "$src" ] || kt_die "--module-src is required for mode module"
    [ -d "$src" ] || kt_die "no such directory: $src"
    local kdir=/lib/modules/$(uname -r)/build
    [ -d "$kdir" ] || kt_die "kernel-devel for $(uname -r) is not installed ($kdir missing)"
    command -v make >/dev/null 2>&1 || kt_die "make is not installed; cannot build a module here"

    kt_log "building in $src against $kdir"
    if ! make -C "$kdir" M="$(readlink -f "$src")" modules >> "$STATE_DIR/apply.log" 2>&1; then
        kt_err "module build failed -- see $STATE_DIR/apply.log"
        state_set phase apply_failed
        exit 2
    fi
    local kos=( "$src"/*.ko )
    [ -e "${kos[0]}" ] || kt_die "the build produced no .ko"
    if kt_module_sig_enforced; then
        kt_err "This kernel refuses unsigned modules (module signature enforcement"
        kt_err "is on, usually because Secure Boot is enabled).  Sign the module"
        kt_err "with a key this machine trusts, or turn Secure Boot off."
        state_set phase apply_failed
        exit 2
    fi
    for ko in "${kos[@]}"; do
        kt_log "insmod $ko"
        insmod "$ko" >> "$STATE_DIR/apply.log" 2>&1 || \
            apply_die "insmod $ko failed -- see $STATE_DIR/apply.log"
    done
    printf '%s\n' "${kos[@]}" > "$STATE_DIR/modules-loaded.txt"
    kt_ok "${#kos[@]} module(s) loaded"
    state_set phase applied
}

apply_source() {
    local tree=$SOURCE_DIR
    local dir=$PATCH_DIR
    [ -n "$tree" ] || kt_die "--source-dir is required for mode source"
    [ -d "$tree" ] || kt_die "no such directory: $tree"
    [ -n "$dir" ]  || kt_die "--patch-dir (of .patch files) is required for mode source"
    for t in gcc make flex bison bc; do
        command -v "$t" >/dev/null 2>&1 || kt_die "$t is missing; mode source needs a toolchain"
    done

    local patches=( "$dir"/*.patch )
    [ -e "${patches[0]}" ] || kt_die "no .patch files in $dir"
    kt_log "applying ${#patches[@]} patches to $tree"
    for p in "${patches[@]}"; do
        # --dry-run first: a half applied series is much harder to undo.
        if ! patch -d "$tree" -p1 --dry-run < "$p" >> "$STATE_DIR/apply.log" 2>&1; then
            kt_err "$p does not apply cleanly; nothing has been changed"
            state_set phase apply_failed
            exit 2
        fi
    done
    for p in "${patches[@]}"; do
        patch -d "$tree" -p1 < "$p" >> "$STATE_DIR/apply.log" 2>&1 || \
            apply_die "applying $p failed even though the dry run passed -- see $STATE_DIR/apply.log"
        kt_ok "applied $(basename "$p")"
    done

    # Build from the running kernel's configuration so the result is
    # comparable with the baseline.
    if [ -r "/boot/config-$(uname -r)" ] && [ ! -r "$tree/.config" ]; then
        cp "/boot/config-$(uname -r)" "$tree/.config"
        kt_log "seeded .config from the running kernel"
        ( cd "$tree" && make olddefconfig ) >> "$STATE_DIR/apply.log" 2>&1
    fi
    kt_log "building (this takes a while; log: $STATE_DIR/apply.log)"
    if ! ( cd "$tree" && make -j"$(nproc)" ) >> "$STATE_DIR/apply.log" 2>&1; then
        kt_err "the kernel build failed -- see $STATE_DIR/apply.log"
        state_set phase apply_failed
        exit 2
    fi
    ( cd "$tree" && make modules_install && make install ) >> "$STATE_DIR/apply.log" 2>&1 || \
        apply_die "installing the new kernel failed -- see $STATE_DIR/apply.log"
    local newk
    newk=$(cd "$tree" && make -s kernelrelease 2>/dev/null)
    [ -n "$newk" ] && state_set expect_kernel "$newk"
    kt_ok "built and installed $newk"
    need_reboot
}

apply_manual() {
    cat <<EOF

  The baseline is recorded.  Apply your patch now, however you normally do
  it.  When the machine is running the patched kernel, come back and run:

      $0 continue

EOF
    state_set phase awaiting_manual
    exit 3
}

# --- reboot handling -------------------------------------------------------
need_reboot() {
    state_set phase awaiting_reboot
    if [ "$AUTO_REBOOT" = 1 ]; then
        install_resume_unit
        kt_warn "rebooting in 10 seconds; the run continues by itself afterwards"
        kt_warn "press Ctrl-C to stop"
        sleep 10
        systemctl reboot
        exit 3
    fi
    cat <<EOF

  The patched kernel is installed but the machine is still running
  $(kt_kver).  Reboot into it, then run:

      $0 continue

  Check the boot entry first if this machine does not boot the newest
  kernel by default:  grubby --default-kernel

EOF
    exit 3
}

install_resume_unit() {
    cat > "$RESUME_UNIT" <<EOF
[Unit]
Description=Continue the kernel-test patch run after a reboot
After=multi-user.target network-online.target
Wants=network-online.target

[Service]
Type=oneshot
# No options: "continue" reads them back out of the state file, so the run
# after the reboot tests exactly what the baseline tested.
ExecStart=$KT_ROOT/test-patch.sh continue
StandardOutput=append:$STATE_DIR/resume.log
StandardError=append:$STATE_DIR/resume.log
RemainAfterExit=yes

[Install]
WantedBy=multi-user.target
EOF
    systemctl daemon-reload
    systemctl enable kernel-test-resume.service >/dev/null 2>&1
    state_set resume_unit installed
    kt_log "installed $RESUME_UNIT; it removes itself once the run finishes"
}

remove_resume_unit() {
    if [ -e "$RESUME_UNIT" ]; then
        systemctl disable kernel-test-resume.service >/dev/null 2>&1
        rm -f "$RESUME_UNIT"
        systemctl daemon-reload
        kt_log "removed $RESUME_UNIT"
    fi
}

phase_verify() {
    require_state
    local want running
    running=$(kt_kver)
    if want=$(state_get expect_kernel); then
        if [ "$want" != "$running" ]; then
            kt_err "expected to be running $want but this is $running"
            kt_err "The machine booted the wrong kernel.  Fix the boot entry"
            kt_err "(grubby --set-default /boot/vmlinuz-$want), reboot, and run"
            kt_err "'$0 continue' again."
            exit 2
        fi
        kt_ok "running the patched kernel: $running"
    else
        local before; before=$(state_get before_kernel) || before=""
        if [ -n "$before" ] && [ "$before" = "$running" ]; then
            kt_log "same kernel as the baseline ($running) -- expected for a"
            kt_log "livepatch, a module or a userspace patch"
        fi
    fi
    state_set after_kernel "$running"
}

phase_after() {
    require_state
    restore_from_state
    run_tests after-patch
    state_set after_dir "$RUN_TESTS_DIR"
    state_set phase after_done
    kt_ok "post-patch run recorded: $RUN_TESTS_DIR"
}

phase_compare() {
    require_state
    local b a
    b=$(state_get before_dir) || kt_die "no baseline recorded"
    a=$(state_get after_dir)  || kt_die "no post-patch run recorded"
    [ -d "$b" ] || kt_die "the baseline run directory is gone: $b"
    [ -d "$a" ] || kt_die "the post-patch run directory is gone: $a"

    local out; out=$(kt_new_run_dir "$KT_RESULTS_DIR" compare) || exit 2
    local flaky=$KT_ROOT/$PATCH_FLAKY_LIST
    [ -r "$flaky" ] || flaky=""

    kt_head "Comparison"
    if [ -n "$flaky" ]; then
        "$KT_PYTHON" "$KT_ROOT/lib/compare.py" "$b" "$a" "$out" --flaky "$flaky"
    else
        "$KT_PYTHON" "$KT_ROOT/lib/compare.py" "$b" "$a" "$out"
    fi
    local rc=$?

    # Carry the evidence across so the comparison folder stands on its own.
    cp "$b/report.md" "$out/report-before.md" 2>/dev/null
    cp "$a/report.md" "$out/report-after.md"  2>/dev/null
    cp "$STATE" "$out/patch-state.txt" 2>/dev/null
    for f in pkgs.diff apply.log kpatch-list.txt; do
        [ -r "$STATE_DIR/$f" ] && cp "$STATE_DIR/$f" "$out/$f"
    done

    state_set compare_dir "$out"
    state_set phase done
    remove_resume_unit

    kt_info "comparison : $out/comparison.md"
    kt_info "before     : $b/report.md"
    kt_info "after      : $a/report.md"
    if [ "$rc" = 0 ]; then
        kt_ok "No regression."
    else
        kt_err "Regressions found.  Read $out/comparison.md"
    fi
    return $rc
}

# --- commands --------------------------------------------------------------
case $CMD in
    start)
        kt_require_root
        phase_baseline
        phase_apply
        # Modes that need neither a reboot nor a pause fall through.
        phase_verify
        phase_after
        phase_compare
        exit $?
        ;;
    continue)
        kt_require_root
        require_state
        p=$(state_get phase) || p=""
        case $p in
            baseline_done|apply_failed|applying)
                # "applying" means the machine stopped in the middle of the
                # apply phase.  Running it again is the only way forward; for
                # RPMs that is a no-op, for a module insmod will say it is
                # already there.
                [ "$p" = applying ] && kt_warn "the apply phase did not finish last time; running it again"
                phase_apply; phase_verify; phase_after; phase_compare; exit $? ;;
            awaiting_reboot|awaiting_manual|applied)
                phase_verify; phase_after; phase_compare; exit $? ;;
            after_done)
                phase_compare; exit $? ;;
            baseline)
                kt_err "the baseline run did not finish, so there is nothing to"
                kt_err "compare against.  Start again:  $0 abort && $0 start ..."
                exit 2 ;;
            done)
                kt_log "this run has already finished: $(state_get compare_dir)"; exit 0 ;;
            *)
                kt_die "do not know how to continue from phase '$p'" ;;
        esac
        ;;
    baseline) kt_require_root; phase_baseline ;;
    apply)    kt_require_root; phase_apply ;;
    after)    kt_require_root; phase_verify; phase_after ;;
    compare)  phase_compare; exit $? ;;
    status)   state_show ;;
    abort)
        remove_resume_unit
        if [ -r "$STATE" ]; then
            mv "$STATE" "$STATE.aborted-$(date +%Y%m%d-%H%M%S)"
            kt_ok "run forgotten; the patch itself was NOT undone"
        else
            kt_log "nothing to abort"
        fi
        ;;
    *) kt_err "unknown command: $CMD"; usage; exit 2 ;;
esac
