#!/usr/bin/python3
"""Turn KUnit / kselftest / LTP output into one common CSV.

Every engine ends up as rows of
    engine,collection,test,status,duration_s,message
so that compare.py can diff two runs without knowing which engine made them.

Usage:
    parse_results.py kunit     <results-file> [--collection NAME]
    parse_results.py kselftest <output.log>
    parse_results.py ltp-json  <kirk-report.json>
    parse_results.py ltp-text  <runltp-log>
Rows are written to stdout as CSV without a header.
"""

import csv
import json
import re
import sys

STATUS_PASS = "PASS"
STATUS_FAIL = "FAIL"
STATUS_SKIP = "SKIP"
STATUS_ERROR = "ERROR"


def emit(rows):
    w = csv.writer(sys.stdout, quoting=csv.QUOTE_ALL, lineterminator="\n")
    for r in rows:
        w.writerow(r)


# --------------------------------------------------------------------------
# KUnit -- nested KTAP, as found in /sys/kernel/debug/kunit/<suite>/results
# --------------------------------------------------------------------------
# A result line may be indented; KUnit uses four spaces per nesting level.
RE_RESULT = re.compile(
    r"^(?P<indent>[ \t]*)(?P<not>not )?ok\s+(?P<num>\d+)\s*(?P<name>.*?)\s*$"
)
RE_SUBTEST = re.compile(r"^(?P<indent>[ \t]*)#\s*Subtest:\s*(?P<name>.+?)\s*$")
RE_DIAG = re.compile(r"^(?P<indent>[ \t]*)#\s*(?P<text>.*)$")


def _depth(indent):
    return len(indent.expandtabs(8)) // 4


def _subtest_paths(lines):
    """Full path of every "# Subtest: X" declaration in the document.

    A subtest declared at indent depth k has its own result line at depth
    k-1, and its enclosing suites are whatever is on the stack above it.
    Recording the whole path, not just the name, is what keeps two suites
    that happen to contain a same-named subtest apart.
    """
    stack = {}
    paths = set()
    for ln in lines:
        m = RE_SUBTEST.match(ln)
        if not m:
            continue
        d = _depth(m.group("indent"))
        name = m.group("name")
        # A declaration at depth d replaces anything at d or deeper: those
        # belonged to the sibling we have just left.
        for k in [k for k in stack if k >= d]:
            del stack[k]
        paths.add(tuple(stack[k] for k in sorted(stack) if k < d) + (name,))
        stack[d] = name
    return paths


def parse_kunit(path, collection=None):
    """Emit one row per leaf test case.

    A line like "ok 2 example" that has a "# Subtest: example" one level
    deeper is a container: its children were already emitted, so it is
    dropped.  Keeping both would double count every suite.
    """
    with open(path, "r", errors="replace") as fh:
        lines = fh.read().splitlines()

    declared = _subtest_paths(lines)

    # Path of enclosing subtest names, indexed by depth.
    stack = {}
    rows = []
    for ln in lines:
        m = RE_SUBTEST.match(ln)
        if m:
            d = _depth(m.group("indent"))
            for k in [k for k in stack if k >= d]:
                del stack[k]
            stack[d] = m.group("name")
            continue
        m = RE_RESULT.match(ln)
        if not m:
            continue
        d = _depth(m.group("indent"))
        raw = m.group("name")

        # A trailing directive: "# SKIP reason" / "# TODO ..."
        status = STATUS_FAIL if m.group("not") else STATUS_PASS
        msg = ""
        if "#" in raw:
            name, _, directive = raw.partition("#")
            name = name.strip()
            directive = directive.strip()
            if directive.upper().startswith("SKIP"):
                status = STATUS_SKIP
                msg = directive[4:].strip()
            else:
                msg = directive
        else:
            name = raw.strip()

        enclosing = [stack[k] for k in sorted(stack) if k <= d]

        # Container result -> its children already produced rows.  Compared
        # by full path, because a bare name is not unique across suites: a
        # real "not ok shared_name" in one suite would otherwise be thrown
        # away because a different suite declares a subtest of that name.
        if tuple(enclosing) + (name,) in declared:
            continue

        # The outermost "# Subtest:" is the KUnit suite; that is the
        # collection.  Anything nested below it (parameterised sub-suites)
        # becomes part of the test name, so that identically named cases in
        # different sub-suites do not collide in the comparison.
        coll = collection or (enclosing[0] if enclosing else "kunit")
        test = ".".join(enclosing[1:] + [name])
        rows.append(["kunit", coll, test, status, "", msg])
    return rows


# --------------------------------------------------------------------------
# kselftest -- TAP 13 from run_kselftest.sh
# --------------------------------------------------------------------------
# "ok 3 selftests: cgroup: test_memcontrol"
# "not ok 4 selftests: net: reuseaddr_ports_exhausted.sh # exit=1"
RE_KSELF = re.compile(
    r"^(?P<not>not )?ok\s+(?P<num>\d+)\s+selftests:\s+"
    r"(?P<coll>\S+):\s+(?P<test>\S+)\s*(?P<rest>.*)$"
)
# "# TIMEOUT 45 seconds" is printed by the runner just before a timed-out test
RE_TIMEOUT = re.compile(r"^#\s*TIMEOUT\b")


def parse_kselftest(path):
    rows = []
    timed_out = False
    with open(path, "r", errors="replace") as fh:
        for ln in fh:
            ln = ln.rstrip("\n")
            if RE_TIMEOUT.match(ln):
                timed_out = True
                continue
            m = RE_KSELF.match(ln)
            if not m:
                continue
            rest = m.group("rest").strip()
            status = STATUS_FAIL if m.group("not") else STATUS_PASS
            msg = rest.lstrip("#").strip()
            upper = msg.upper()
            if upper.startswith("SKIP"):
                status = STATUS_SKIP
                msg = msg[4:].strip()
            elif timed_out and status == STATUS_FAIL:
                status = "TIMEOUT"
            timed_out = False
            rows.append(["kselftest", m.group("coll"), m.group("test"),
                         status, "", msg])
    return rows


# --------------------------------------------------------------------------
# LTP -- kirk --json-report
# --------------------------------------------------------------------------
def parse_ltp_json(path, suite="ltp"):
    """kirk writes {"results":[{test_fqn,status,test:{...}}], "stats":{...}}.

    "status" is kirk's own verdict and is authoritative; the pass/fail/brok
    counters sit under the nested "test" object, not at the top level.
    """
    with open(path, "r", errors="replace") as fh:
        data = json.load(fh)

    # kirk vocabulary (libkirk/export.py): pass brok warn conf fail.
    # conf means "test does not apply to this kernel", which is a skip.
    status_map = {
        "pass": STATUS_PASS,
        "fail": STATUS_FAIL,
        "brok": STATUS_ERROR,
        "conf": STATUS_SKIP,
        "warn": STATUS_PASS,
    }

    rows = []
    for res in data.get("results", []):
        t = res.get("test", {}) or {}
        fqn = res.get("test_fqn") or (t.get("command") or "?")
        verdict = (res.get("status") or t.get("result") or "").lower()
        status = status_map.get(verdict)
        if status is None:
            # Unknown verdict: fall back to the counters.
            if int(t.get("failed", 0) or 0):
                status = STATUS_FAIL
            elif int(t.get("broken", 0) or 0):
                status = STATUS_ERROR
            elif int(t.get("passed", 0) or 0):
                status = STATUS_PASS
            else:
                status = STATUS_SKIP

        dur = t.get("duration")
        dur = "%.2f" % float(dur) if dur is not None else ""
        retval = t.get("retval")
        msg = "%s pass=%s fail=%s brok=%s skip=%s warn=%s rc=%s" % (
            verdict or "?",
            t.get("passed", "?"), t.get("failed", "?"), t.get("broken", "?"),
            t.get("skipped", "?"), t.get("warnings", "?"),
            ",".join(str(x) for x in retval) if isinstance(retval, list) else retval)
        rows.append(["ltp", suite, fqn, status, dur, msg])
    return rows


# --------------------------------------------------------------------------
# LTP -- the plain runltp log, used when kirk cannot run
# --------------------------------------------------------------------------
# "abort01                            PASS       0"
RE_LTP_TEXT = re.compile(
    r"^(?P<test>\S+)\s+(?P<res>PASS|FAIL|BROK|CONF|WARN|RETR|TBROK|TCONF)\s+"
    r"(?P<rc>-?\d+)\s*$"
)
LTP_MAP = {
    "PASS": STATUS_PASS,
    "FAIL": STATUS_FAIL,
    "BROK": STATUS_ERROR,
    "TBROK": STATUS_ERROR,
    "CONF": STATUS_SKIP,
    "TCONF": STATUS_SKIP,
    "WARN": STATUS_PASS,
    "RETR": STATUS_ERROR,
}


def parse_ltp_text(path, suite="ltp"):
    rows = []
    with open(path, "r", errors="replace") as fh:
        for ln in fh:
            m = RE_LTP_TEXT.match(ln.strip())
            if not m:
                continue
            res = m.group("res")
            rows.append(["ltp", suite, m.group("test"),
                         LTP_MAP.get(res, STATUS_ERROR), "",
                         "%s rc=%s" % (res, m.group("rc"))])
    return rows


def main(argv):
    if len(argv) < 3:
        sys.stderr.write(__doc__)
        return 2
    mode, path = argv[1], argv[2]
    collection = None
    if "--collection" in argv:
        collection = argv[argv.index("--collection") + 1]

    if mode == "kunit":
        rows = parse_kunit(path, collection)
    elif mode == "kselftest":
        rows = parse_kselftest(path)
    elif mode == "ltp-json":
        rows = parse_ltp_json(path, collection or "ltp")
    elif mode == "ltp-text":
        rows = parse_ltp_text(path, collection or "ltp")
    else:
        sys.stderr.write("unknown mode: %s\n" % mode)
        return 2
    emit(rows)
    return 0


if __name__ == "__main__":
    sys.exit(main(sys.argv))
