#!/bin/bash
# ---------------------------------------------------------------------------
# Check the kit itself, not the kernel.
#
# Everything here is fast and harmless.  A check whose prerequisite is
# missing reports SKIP rather than FAIL, so this is also a useful way to see
# what a given machine can and cannot do.
#
#   ./selftest.sh
# ---------------------------------------------------------------------------
set -uo pipefail

KT_ROOT=$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)
# shellcheck source=lib/common.sh
. "$KT_ROOT/lib/common.sh"
kt_load_settings

PASS=0; FAIL=0; SKIP=0
WORK=$(mktemp -d /tmp/kernel-test-selftest.XXXXXX)
trap 'rm -rf "$WORK"' EXIT

ok()   { PASS=$(( PASS + 1 )); printf '  %s%-6s%s %s\n' "$C_GRN" "PASS" "$C_OFF" "$1"; }
bad()  { FAIL=$(( FAIL + 1 )); printf '  %s%-6s%s %s\n' "$C_RED" "FAIL" "$C_OFF" "$1"; }
skip() { SKIP=$(( SKIP + 1 )); printf '  %s%-6s%s %s  %s(%s)%s\n' "$C_YEL" "SKIP" "$C_OFF" "$1" "$C_DIM" "$2" "$C_OFF"; }

check() {  # check <name> <command...>
    local name=$1; shift
    if "$@" >/dev/null 2>&1; then ok "$name"; else bad "$name"; fi
}

kt_head "Files"
for f in test-kernel.sh test-patch.sh install-offline.sh build-ltp.sh \
         download-rpms.sh make-bundle.sh selftest.sh \
         tests/run-kunit.sh tests/run-kselftest.sh tests/run-ltp.sh \
         lib/common.sh lib/parse_results.py lib/report.py lib/compare.py \
         settings.conf README.md; do
    if [ -e "$KT_ROOT/$f" ]; then ok "$f exists"; else bad "$f is missing"; fi
done

kt_head "Syntax"
for f in "$KT_ROOT"/*.sh "$KT_ROOT"/tests/*.sh "$KT_ROOT"/lib/*.sh; do
    [ -e "$f" ] || continue
    check "bash -n $(basename "$f")" bash -n "$f"
done
for f in "$KT_ROOT"/lib/*.py; do
    [ -e "$f" ] || continue
    check "python syntax $(basename "$f")" "$KT_PYTHON" -c \
        "import ast,sys;ast.parse(open(sys.argv[1]).read())" "$f"
done

kt_head "Executable bits"
for f in "$KT_ROOT"/*.sh "$KT_ROOT"/tests/*.sh; do
    [ -e "$f" ] || continue
    if [ -x "$f" ]; then ok "$(basename "$f") is executable"; else bad "$(basename "$f") is not executable"; fi
done

kt_head "Settings"
check "settings.conf parses" bash -c ". '$KT_ROOT/lib/common.sh'; kt_load_settings"
for v in KT_PROFILE KT_ENGINES KT_RESULTS_DIR KT_SAFE KUNIT_TIMEOUT \
         KSELFTEST_DIR KSELFTEST_TIMEOUT LTP_DIR LTP_TIMEOUT LTP_SUITE_TIMEOUT \
         LTP_TMPDIR PATCH_STATE_DIR PATCH_FLAKY_LIST; do
    if [ -n "${!v:-}" ]; then ok "$v is set (${!v})"; else bad "$v is not set"; fi
done
# The environment must win over the file, or nothing can be overridden.
got=$(KT_PROFILE=smoke bash -c ". '$KT_ROOT/lib/common.sh'; kt_load_settings; printf '%s' \"\$KT_PROFILE\"")
if [ "$got" = smoke ]; then ok "environment overrides settings.conf"; else bad "environment did not override settings.conf (got '$got')"; fi

kt_head "Profile lists"
for p in smoke standard full; do
    for e in kselftest ltp; do
        f="$KT_ROOT/conf/$e-$p.list"
        if [ -r "$f" ]; then ok "conf/$e-$p.list exists"; else bad "conf/$e-$p.list is missing"; fi
    done
done
if [ -d "$LTP_DIR/runtest" ]; then
    bad_names=0
    for p in smoke standard full; do
        while read -r s; do
            case $s in ''|\#*) continue ;; esac
            [ -f "$LTP_DIR/runtest/$s" ] || { bad "conf/ltp-$p.list names a scenario LTP does not have: $s"; bad_names=1; }
        done < "$KT_ROOT/conf/ltp-$p.list"
    done
    [ "$bad_names" = 0 ] && ok "every LTP scenario named in the profiles exists"
else
    skip "LTP scenario names" "LTP is not installed"
fi
if [ -r "$KSELFTEST_DIR/kselftest-list.txt" ]; then
    avail=$(grep -oE '^[A-Za-z0-9_./-]+:' "$KSELFTEST_DIR/kselftest-list.txt" | tr -d : | sort -u)
    bad_names=0
    for p in smoke standard; do
        while read -r c; do
            case $c in ''|\#*|all) continue ;; esac
            grep -qx "$c" <<< "$avail" || { bad "conf/kselftest-$p.list names a collection that is not installed: $c"; bad_names=1; }
        done < "$KT_ROOT/conf/kselftest-$p.list"
    done
    [ "$bad_names" = 0 ] && ok "every kselftest collection named in the profiles exists"
else
    skip "kselftest collection names" "kernel-selftests-internal is not installed"
fi

kt_head "Parsers"
# KUnit: a nested KTAP document with a container result, a skip directive and
# a parameterised sub-suite -- the three things that trip a naive parser.
cat > "$WORK/kunit.ktap" <<'EOF'
KTAP version 1
1..1
    KTAP version 1
    # Subtest: demo
    1..3
    ok 1 alpha
    not ok 2 beta
        KTAP version 1
        # Subtest: gamma
        ok 1 case one
        ok 2 case two # SKIP not applicable
    ok 3 gamma
ok 1 demo
EOF
rows=$("$KT_PYTHON" "$KT_ROOT/lib/parse_results.py" kunit "$WORK/kunit.ktap")
n=$(printf '%s\n' "$rows" | grep -c .)
if [ "$n" = 4 ]; then ok "KUnit parser: 4 leaf cases, containers dropped"
else bad "KUnit parser returned $n rows, expected 4"; printf '%s\n' "$rows" | sed 's/^/        /'; fi
grep -q '"demo","alpha","PASS"'                      <<< "$rows" && ok "KUnit parser: pass"    || bad "KUnit parser: pass"
grep -q '"demo","beta","FAIL"'                       <<< "$rows" && ok "KUnit parser: fail"    || bad "KUnit parser: fail"
grep -q '"demo","gamma.case two","SKIP"'             <<< "$rows" && ok "KUnit parser: nested skip keeps its sub-suite" || bad "KUnit parser: nested skip"

# Two suites, each with something called "shared_name": one a real failing
# leaf, the other a container.  Matching containers by bare name instead of
# by full path silently threw the failure away.
cat > "$WORK/collide.ktap" <<'EOF'
KTAP version 1
1..2
    KTAP version 1
    # Subtest: suite_a
    not ok 1 shared_name
ok 1 suite_a
    KTAP version 1
    # Subtest: suite_b
        KTAP version 1
        # Subtest: shared_name
        ok 1 inner
    ok 1 shared_name
ok 2 suite_b
EOF
rows=$("$KT_PYTHON" "$KT_ROOT/lib/parse_results.py" kunit "$WORK/collide.ktap")
grep -q '"suite_a","shared_name","FAIL"' <<< "$rows" \
    && ok "KUnit parser: a failure is not eaten by a same-named subtest elsewhere" \
    || { bad "KUnit parser: lost a failure to a same-named subtest in another suite"; printf '%s\n' "$rows" | sed 's/^/        /'; }
grep -q '"suite_b","shared_name.inner","PASS"' <<< "$rows" \
    && ok "KUnit parser: the real container is still dropped" || bad "KUnit parser: container handling"

# KUnit also prints KTAP to the kernel log, where every line carries a
# "[  1234.567890] " prefix.  The fallback must strip those before looking.
cat > "$WORK/dmesg.ktap" <<'EOF'
[ 1008177.462264] KTAP version 1
[ 1008177.462495] 1..1
[ 1008177.463259]     KTAP version 1
[ 1008177.463522]     # Subtest: from_dmesg
[ 1008177.463746]     ok 1 case_one
[ 1008177.464000] ok 1 from_dmesg
EOF
sed -E 's/^\[[^]]*\] ?//' "$WORK/dmesg.ktap" > "$WORK/dmesg.stripped"
grep -q '^[[:space:]]*KTAP version' "$WORK/dmesg.stripped" \
    && ok "KUnit dmesg fallback: the KTAP check matches once timestamps are stripped" \
    || bad "KUnit dmesg fallback: the KTAP check does not match"
grep -q '^[[:space:]]*KTAP version' "$WORK/dmesg.ktap" \
    && bad "the raw kernel log should NOT match before stripping" \
    || ok "KUnit dmesg fallback: raw log correctly does not match (this was the bug)"
rows=$("$KT_PYTHON" "$KT_ROOT/lib/parse_results.py" kunit "$WORK/dmesg.stripped" --collection mod)
grep -q '"mod","case_one","PASS"' <<< "$rows" && ok "KUnit dmesg fallback parses" || bad "KUnit dmesg fallback parses"

cat > "$WORK/kself.log" <<'EOF'
TAP version 13
1..3
# selftests: memfd: memfd_test
ok 1 selftests: memfd: memfd_test
not ok 2 selftests: cgroup: test_memcontrol # exit=1
ok 3 selftests: net: rtnetlink.sh # SKIP needs two interfaces
EOF
rows=$("$KT_PYTHON" "$KT_ROOT/lib/parse_results.py" kselftest "$WORK/kself.log")
grep -q '"memfd","memfd_test","PASS"'        <<< "$rows" && ok "kselftest parser: pass" || bad "kselftest parser: pass"
grep -q '"cgroup","test_memcontrol","FAIL"'  <<< "$rows" && ok "kselftest parser: fail" || bad "kselftest parser: fail"
grep -q '"net","rtnetlink.sh","SKIP"'        <<< "$rows" && ok "kselftest parser: skip" || bad "kselftest parser: skip"

cat > "$WORK/kirk.json" <<'EOF'
{"results":[
 {"test_fqn":"access01","status":"pass","test":{"command":"access01","duration":0.03,"passed":9,"failed":0,"broken":0,"skipped":0,"warnings":0,"retval":["0"]}},
 {"test_fqn":"broken01","status":"brok","test":{"command":"broken01","duration":0.1,"passed":0,"failed":0,"broken":1,"skipped":0,"warnings":0,"retval":["2"]}},
 {"test_fqn":"conf01","status":"conf","test":{"command":"conf01","duration":0.0,"passed":0,"failed":0,"broken":0,"skipped":1,"warnings":0,"retval":["32"]}},
 {"test_fqn":"fail01","status":"fail","test":{"command":"fail01","duration":1.5,"passed":1,"failed":2,"broken":0,"skipped":0,"warnings":0,"retval":["1"]}}
],"stats":{"runtime":1.6}}
EOF
rows=$("$KT_PYTHON" "$KT_ROOT/lib/parse_results.py" ltp-json "$WORK/kirk.json" --collection syscalls)
grep -q '"syscalls","access01","PASS"'  <<< "$rows" && ok "LTP parser: pass"                  || bad "LTP parser: pass"
grep -q '"syscalls","broken01","ERROR"' <<< "$rows" && ok "LTP parser: brok becomes ERROR"    || bad "LTP parser: brok"
grep -q '"syscalls","conf01","SKIP"'    <<< "$rows" && ok "LTP parser: conf becomes SKIP"     || bad "LTP parser: conf"
grep -q '"syscalls","fail01","FAIL"'    <<< "$rows" && ok "LTP parser: fail"                  || bad "LTP parser: fail"

kt_head "Report and comparison"
mkdir -p "$WORK/before" "$WORK/after"
printf '%s\n' "$KT_CSV_HEADER" > "$WORK/before/results.csv"
cat >> "$WORK/before/results.csv" <<'EOF'
"kunit","demo","alpha","PASS","0.1",""
"kunit","demo","beta","PASS","0.1",""
"ltp","syscalls","gone01","PASS","0.1",""
"ltp","syscalls","old_fail","FAIL","0.1",""
"ltp","syscalls","leapsec01","PASS","0.1",""
"kselftest","memfd","(collection)","SKIP","",""
EOF
cp "$WORK/before/results.csv" "$WORK/after/results.csv"
"$KT_PYTHON" - "$WORK/after/results.csv" <<'PY'
import sys
p=sys.argv[1]
out=[]
for l in open(p).read().splitlines():
    if '"beta"' in l: l=l.replace('"PASS"','"FAIL"',1)     # a regression
    if '"old_fail"' in l: l=l.replace('"FAIL"','"PASS"',1) # a fix
    if '"gone01"' in l: continue                            # disappeared
    if '"leapsec01"' in l: l=l.replace('"PASS"','"FAIL"',1) # flaky regression
    out.append(l)
out.append('"ltp","syscalls","new01","PASS","0.1",""')
open(p,'w').write("\n".join(out)+"\n")
PY
printf 'kernel: 5.14.0-1.el9_6.x86_64\nprofile: smoke\nengines: kunit ltp\n' > "$WORK/before/env.txt"
printf 'kernel: 5.14.0-2.el9_6.x86_64\nprofile: smoke\nengines: kunit ltp\n' > "$WORK/after/env.txt"

# A run with nothing but passes: the "before" fixture deliberately holds a
# pre-existing failure, so it cannot be used for this one.
mkdir -p "$WORK/clean"
printf '%s\n' "$KT_CSV_HEADER" > "$WORK/clean/results.csv"
printf '%s\n' '"kunit","demo","alpha","PASS","0.1",""' >> "$WORK/clean/results.csv"
printf '%s\n' '"ltp","syscalls","conf01","SKIP","0.1",""' >> "$WORK/clean/results.csv"
if "$KT_PYTHON" "$KT_ROOT/lib/report.py" "$WORK/clean" >/dev/null 2>&1; then
    ok "report.py exits 0 when nothing failed"
else bad "report.py should have exited 0 on an all-pass run"; fi
if "$KT_PYTHON" "$KT_ROOT/lib/report.py" "$WORK/after" >/dev/null 2>&1; then
    bad "report.py should have exited 1 when a test failed"
else ok "report.py exits 1 when a test failed"; fi
[ -s "$WORK/after/report.md" ]    && ok "report.md written"    || bad "report.md missing"
[ -s "$WORK/after/summary.json" ] && ok "summary.json written" || bad "summary.json missing"
check "summary.json is valid json" "$KT_PYTHON" -c \
    "import json,sys;json.load(open(sys.argv[1]))" "$WORK/after/summary.json"

out=$("$KT_PYTHON" "$KT_ROOT/lib/compare.py" "$WORK/before" "$WORK/after" "$WORK/cmp" \
      --flaky "$KT_ROOT/conf/flaky.list" 2>&1)
rc=$?
[ "$rc" = 1 ] && ok "compare.py exits 1 on a regression" || bad "compare.py exit was $rc, expected 1"
j=$WORK/cmp/comparison.json
n() { "$KT_PYTHON" -c "import json,sys;print(len(json.load(open('$j'))['$1']))"; }
[ "$(n regressions)" = 1 ]       && ok "comparison: 1 regression"                  || bad "comparison: regressions = $(n regressions), expected 1"
[ "$(n flaky_regressions)" = 1 ] && ok "comparison: flaky regression kept apart"   || bad "comparison: flaky_regressions = $(n flaky_regressions), expected 1"
[ "$(n fixes)" = 1 ]             && ok "comparison: 1 fix"                         || bad "comparison: fixes = $(n fixes), expected 1"
[ "$(n new_tests)" = 1 ]         && ok "comparison: 1 new test"                    || bad "comparison: new_tests = $(n new_tests), expected 1"
[ "$(n missing_tests)" = 1 ]     && ok "comparison: 1 test disappeared"            || bad "comparison: missing_tests = $(n missing_tests), expected 1"
grep -q '(collection)' "$j" && bad "comparison counted a bookkeeping row as a test" \
                            || ok "comparison ignores bookkeeping rows"

kt_head "Kernel log marks"
# The mark has to be exact.  A timestamp with one-second resolution cannot
# tell forty KUnit modules apart -- they all load inside the same second,
# and one module's stack trace gets blamed on the next one.
m1=$(kt_dmesg_mark)
if [ "$m1" -eq "$m1" ] 2>/dev/null; then ok "kt_dmesg_mark returns a line count"
else bad "kt_dmesg_mark returned '$m1', which is not a number"; fi
[ "$(kt_dmesg_since "$m1" | wc -l)" = 0 ] \
    && ok "nothing is attributed to a window that has not happened yet" \
    || bad "kt_dmesg_since returned lines from before the mark"
[ "$(kt_dmesg_since $(( m1 > 5 ? m1 - 5 : 0 )) | wc -l)" -le 6 ] \
    && ok "kt_dmesg_since returns exactly the lines after the mark" \
    || bad "kt_dmesg_since returned too many lines"
grep -q 'wc -l' <<< "$(declare -f kt_dmesg_mark)" \
    && ok "kt_dmesg_mark does not use a one-second timestamp" \
    || bad "kt_dmesg_mark is time-based again; module attribution will be wrong"

kt_head "An engine that stops working"
# One engine reports an error and takes all its tests with it.  Filing that
# under "tests that disappeared" and calling the run OK would green-light a
# patch that broke a whole suite.
mkdir -p "$WORK/eb" "$WORK/ea"
printf '%s\n' "$KT_CSV_HEADER" > "$WORK/eb/results.csv"
printf '%s\n' '"kunit","example","alpha","PASS","0.1",""' >> "$WORK/eb/results.csv"
printf '%s\n' '"ltp","syscalls","getpid01","PASS","0.1",""' >> "$WORK/eb/results.csv"
printf '%s\n' "$KT_CSV_HEADER" > "$WORK/ea/results.csv"
printf '%s\n' '"kunit","","(engine)","ERROR","","modprobe kunit enable=1 failed"' >> "$WORK/ea/results.csv"
printf '%s\n' '"ltp","syscalls","getpid01","PASS","0.1",""' >> "$WORK/ea/results.csv"
printf 'kernel: k1\nprofile: smoke\n' > "$WORK/eb/env.txt"
printf 'kernel: k2\nprofile: smoke\n' > "$WORK/ea/env.txt"
if "$KT_PYTHON" "$KT_ROOT/lib/compare.py" "$WORK/eb" "$WORK/ea" "$WORK/ecmp" >/dev/null 2>&1; then
    bad "compare.py reported OK when a whole engine stopped working"
else
    ok "compare.py exits 1 when a whole engine stopped working"
fi
v=$("$KT_PYTHON" -c "import json;print(json.load(open('$WORK/ecmp/comparison.json'))['verdict'])" 2>/dev/null)
[ "$v" = REGRESSION ] && ok "comparison verdict is REGRESSION when an engine failed" \
                      || bad "comparison verdict was '$v', expected REGRESSION"
n=$("$KT_PYTHON" -c "import json;print(len(json.load(open('$WORK/ecmp/comparison.json'))['engine_problems']))" 2>/dev/null)
[ "$n" = 1 ] && ok "comparison names the engine that failed" || bad "engine_problems = $n, expected 1"

kt_head "Patch run state"
# state_get must fail, not return an empty string, when a key is present but
# empty -- otherwise every "x=$(state_get k) || x=$DEFAULT" fallback is dead.
ST=$WORK/state
printf 'mode=manual\npatch_dir=\n' > "$ST"
g() { local line; line=$(grep "^$1=" "$ST" | tail -1); [ -n "$line" ] || return 1
      local v=${line#*=}; [ -n "$v" ] || return 1; printf '%s' "$v"; }
if g patch_dir >/dev/null 2>&1; then bad "state_get returns success on an empty value"
else ok "state_get fails on an empty value"; fi
g mode >/dev/null 2>&1 && ok "state_get works on a set value" || bad "state_get failed on a set value"
# The options the baseline used must survive to the run after the reboot.
grep -q 'restore_from_state' "$KT_ROOT/test-patch.sh" \
    && ok "test-patch.sh restores the baseline's test options" \
    || bad "test-patch.sh never reads test_args back, so the two runs differ"
for ph in applying baseline apply_failed awaiting_reboot awaiting_manual applied after_done done; do
    grep -q "$ph" "$KT_ROOT/test-patch.sh" || bad "continue cannot resume from phase '$ph'"
done
ok "the continue dispatcher knows every phase apply can leave behind"

kt_head "Drivers"
check "test-kernel.sh --help"  "$KT_ROOT/test-kernel.sh" --help
check "test-patch.sh --help"   "$KT_ROOT/test-patch.sh" --help
check "install-offline.sh --help" "$KT_ROOT/install-offline.sh" --help
check "build-ltp.sh --help"    "$KT_ROOT/build-ltp.sh" --help
check "download-rpms.sh --help" "$KT_ROOT/download-rpms.sh" --help
if "$KT_ROOT/test-kernel.sh" --profile nonsense --check >/dev/null 2>&1; then
    bad "test-kernel.sh accepted an unknown profile"
else ok "test-kernel.sh rejects an unknown profile"; fi
if "$KT_ROOT/test-kernel.sh" --profile full --check >/dev/null 2>&1; then
    bad "the full profile ran without --unsafe"
else ok "the full profile refuses to run without --unsafe"; fi
if kt_is_root; then
    check "test-kernel.sh --check" "$KT_ROOT/test-kernel.sh" --profile smoke --check
else
    skip "test-kernel.sh --check" "not root"
fi

kt_head "This machine's engines"
if find "/lib/modules/$(uname -r)" -name 'kunit.ko*' -print -quit 2>/dev/null | grep -q .; then
    ok "KUnit: kunit.ko present for $(uname -r)"
    if kt_is_root; then
        if modprobe kunit enable=1 2>/dev/null; then
            ok "KUnit: the core module loads"
            modprobe -r kunit 2>/dev/null
        else
            bad "KUnit: kunit.ko is there but will not load"
        fi
    else
        skip "KUnit: load test" "not root"
    fi
else
    skip "KUnit" "kernel-modules-internal not installed for $(uname -r)"
fi
if [ -x "$KSELFTEST_DIR/run_kselftest.sh" ]; then
    ok "kselftest: installed at $KSELFTEST_DIR"
    check "kselftest: the runner lists its tests" "$KSELFTEST_DIR/run_kselftest.sh" -l
else
    skip "kselftest" "kernel-selftests-internal not installed"
fi
if [ -d "$LTP_DIR/runtest" ]; then
    ok "LTP: installed at $LTP_DIR"
    if [ -f "$LTP_DIR/kirk" ]; then
        check "LTP: kirk starts" "$KT_PYTHON" "$LTP_DIR/kirk" --version
    else
        skip "LTP: kirk" "this build has no kirk; runltp will be used"
    fi
else
    skip "LTP" "not installed at $LTP_DIR"
fi

kt_head "Result"
printf '  %s%d passed%s, %s%d failed%s, %s%d skipped%s\n' \
    "$C_GRN" "$PASS" "$C_OFF" "$C_RED" "$FAIL" "$C_OFF" "$C_YEL" "$SKIP" "$C_OFF"
[ "$FAIL" = 0 ] || exit 1
