#!/usr/bin/python3 """Compare two test runs and say what the change did. compare.py [--flaky FILE] Writes comparison.md and comparison.json into . Exit status 0 when there is no regression, 1 when there is, 2 on error. """ import csv import fnmatch import json import os import sys GOOD = ("PASS",) BAD = ("FAIL", "ERROR", "TIMEOUT") # "(engine)", "(collection)", "(module)", "(scenario)" rows are bookkeeping # from the runners, not test cases, and must not show up as regressions. PSEUDO = ("(engine)", "(collection)", "(module)", "(scenario)") def load(run_dir): """Return (real test cases, engine-level bookkeeping rows). The runners emit "(engine)", "(scenario)", "(module)" and "(collection)" rows to record that a whole engine or suite could not run. Those are not test cases, so they must not appear as regressions -- but they must not be thrown away either: an engine that stops working is the most serious thing a patch can do, and it shows up ONLY in those rows. """ path = os.path.join(run_dir, "results.csv") if not os.path.exists(path): sys.stderr.write("no results.csv in %s\n" % run_dir) sys.exit(2) out = {} pseudo = {} with open(path, newline="") as fh: rd = csv.reader(fh) next(rd, None) for r in rd: if len(r) < 6: r = r + [""] * (6 - len(r)) key = "%s|%s|%s" % (r[0], r[1], r[2]) if r[2] in PSEUDO: pseudo[key] = {"engine": r[0], "collection": r[1], "test": r[2], "status": r[3], "message": r[5]} continue # A test that ran more than once keeps its worst result: a # flapping test must not be hidden by a later pass. prev = out.get(key) if prev and prev["status"] in BAD: continue out[key] = {"engine": r[0], "collection": r[1], "test": r[2], "status": r[3], "duration_s": r[4], "message": r[5]} return out, pseudo def load_env(run_dir): env = {} p = os.path.join(run_dir, "env.txt") if os.path.exists(p): for line in open(p, errors="replace"): if ":" in line: k, _, v = line.partition(":") env[k.strip()] = v.strip() return env def load_flaky(path): pats = [] if path and os.path.exists(path): for line in open(path, errors="replace"): line = line.strip() if line and not line.startswith("#"): pats.append(line) return pats def is_flaky(key, pats): # key is "engine|collection|test"; patterns are written "engine:collection:test" dotted = key.replace("|", ":") return any(fnmatch.fnmatch(dotted, p) for p in pats) def md_table(headers, rows): out = ["| " + " | ".join(headers) + " |", "|" + "|".join(["---"] * len(headers)) + "|"] for r in rows: out.append("| " + " | ".join(str(c).replace("|", "\\|") for c in r) + " |") return "\n".join(out) def split_key(k): return k.split("|", 2) def main(argv): if len(argv) < 4: sys.stderr.write(__doc__) return 2 before_dir, after_dir, out_dir = argv[1], argv[2], argv[3] flaky_path = None if "--flaky" in argv: flaky_path = argv[argv.index("--flaky") + 1] before, before_pseudo = load(before_dir) after, after_pseudo = load(after_dir) flaky = load_flaky(flaky_path) # An engine or suite that reports a problem after the patch but did not # before it. Counting the tests it took with it is what stops "the whole # of KUnit stopped working" being filed under "tests that disappeared" # and reported as OK. engine_problems = [] for k, a in sorted(after_pseudo.items()): if a["status"] not in BAD: continue b = before_pseudo.get(k) if b is not None and b["status"] in BAD: continue # already broken before the patch lost = sum(1 for kk in before if kk not in after and kk.split("|", 1)[0] == a["engine"]) engine_problems.append({"engine": a["engine"], "collection": a["collection"], "what": a["test"], "status": a["status"], "message": a["message"], "tests_lost": lost}) regressions, flaky_regressions, fixes, flaky_fixes = [], [], [], [] still_failing, added, removed, changed_other = [], [], [], [] for k, a in sorted(after.items()): b = before.get(k) if b is None: added.append((k, a)) continue if b["status"] in GOOD and a["status"] in BAD: (flaky_regressions if is_flaky(k, flaky) else regressions).append((k, b, a)) elif b["status"] in BAD and a["status"] in GOOD: # A flaky test that happened to pass this time is not a fix. # Crediting the patch for it is as wrong as blaming it for the # failure in the other direction. (flaky_fixes if is_flaky(k, flaky) else fixes).append((k, b, a)) elif b["status"] in BAD and a["status"] in BAD: still_failing.append((k, b, a)) elif b["status"] != a["status"]: changed_other.append((k, b, a)) for k, b in sorted(before.items()): if k not in after: removed.append((k, b)) env_b, env_a = load_env(before_dir), load_env(after_dir) result = { "before": {"dir": os.path.abspath(before_dir), "environment": env_b, "tests": len(before)}, "after": {"dir": os.path.abspath(after_dir), "environment": env_a, "tests": len(after)}, "regressions": [{"test": k, "before": b["status"], "after": a["status"], "message": a["message"]} for k, b, a in regressions], "flaky_regressions": [{"test": k, "before": b["status"], "after": a["status"], "message": a["message"]} for k, b, a in flaky_regressions], "flaky_fixes": [{"test": k, "before": b["status"], "after": a["status"]} for k, b, a in flaky_fixes], "fixes": [{"test": k, "before": b["status"], "after": a["status"]} for k, b, a in fixes], "still_failing": [k for k, _b, _a in still_failing], "new_tests": [{"test": k, "status": a["status"]} for k, a in added], "missing_tests": [{"test": k, "status": b["status"]} for k, b in removed], "other_changes": [{"test": k, "before": b["status"], "after": a["status"]} for k, b, a in changed_other], "engine_problems": engine_problems, } result["verdict"] = "REGRESSION" if (regressions or engine_problems) else "OK" L = [] L.append("# Patch comparison") L.append("") L.append(md_table( ["", "before", "after"], [["run", os.path.basename(before_dir.rstrip("/")), os.path.basename(after_dir.rstrip("/"))], ["kernel", env_b.get("kernel", "?"), env_a.get("kernel", "?")], ["profile", env_b.get("profile", "?"), env_a.get("profile", "?")], ["engines", env_b.get("engines", "?"), env_a.get("engines", "?")], ["started", env_b.get("started", "?"), env_a.get("started", "?")], ["tests", len(before), len(after)], ["taint", env_b.get("taint_after", "?"), env_a.get("taint_after", "?")]])) L.append("") if env_b.get("kernel") == env_a.get("kernel"): L.append("> Both runs report the same kernel release. If the patch was " "supposed to change the kernel, it is not the one running.") L.append("") if env_b.get("profile") != env_a.get("profile") or \ env_b.get("engines") != env_a.get("engines"): L.append("> The two runs did not test the same things. Differences " "below may be down to that, not to the patch.") L.append("") L.append("## Verdict") L.append("") if engine_problems: L.append("**REGRESSION -- a whole engine or suite stopped working after " "the patch.**") L.append("") L.append(md_table( ["engine", "collection", "what", "result", "tests lost", "detail"], [[e["engine"], e["collection"] or "-", e["what"], e["status"], e["tests_lost"], (e["message"] or "")[:80]] for e in engine_problems])) L.append("") if regressions: L.append("There are also %d individual regression(s) below." % len(regressions)) elif regressions: L.append("**REGRESSION -- %d test(s) passed before the patch and do not " "pass after it.**" % len(regressions)) else: L.append("**OK -- nothing that passed before the patch fails after it.**") L.append("") L.append(md_table(["", "count"], [ ["regressions", len(regressions)], ["regressions in known-flaky tests", len(flaky_regressions)], ["known-flaky tests that flipped the other way", len(flaky_fixes)], ["fixed by the patch", len(fixes)], ["failing before and after", len(still_failing)], ["new tests (not in the before run)", len(added)], ["tests that disappeared", len(removed)], ["other status changes", len(changed_other)], ["engines or suites that stopped working", len(engine_problems)], ])) L.append("") def section(title, rows, note=None): L.append("## %s" % title) L.append("") if note: L.append(note) L.append("") if not rows: L.append("None.") else: L.append(md_table(["engine", "collection", "test", "before", "after", "detail"], rows)) L.append("") section("Regressions", [ split_key(k) + [b["status"], a["status"], (a["message"] or "")[:100]] for k, b, a in regressions]) section("Regressions in known-flaky tests", [ split_key(k) + [b["status"], a["status"], (a["message"] or "")[:100]] for k, b, a in flaky_regressions], "These match a pattern in the flaky list, so they are reported apart " "from the real regressions and do not affect the exit status. Confirm " "them by hand before dismissing them." + ("" if not flaky_fixes else " %d other flaky test(s) went the other way (failing before, " "passing after), which is the same noise seen from the other side." % len(flaky_fixes))) section("Fixed by the patch", [ split_key(k) + [b["status"], a["status"], ""] for k, b, a in fixes]) section("Failing before and after", [ split_key(k) + [b["status"], a["status"], (a["message"] or "")[:100]] for k, b, a in still_failing], "Already broken before the patch. Not this patch's doing, but worth " "knowing about.") L.append("## Tests that appeared or disappeared") L.append("") if not added and not removed: L.append("The same set of tests ran in both runs.") else: L.append("A test that disappears is usually a selftest package or LTP " "version that changed with the patch, not a test that was " "deleted. Check that before reading anything into it.") L.append("") L.append(md_table(["what", "engine", "collection", "test", "status"], [["new"] + split_key(k) + [a["status"]] for k, a in added] + [["gone"] + split_key(k) + [b["status"]] for k, b in removed])) L.append("") os.makedirs(out_dir, exist_ok=True) with open(os.path.join(out_dir, "comparison.md"), "w") as fh: fh.write("\n".join(L) + "\n") with open(os.path.join(out_dir, "comparison.json"), "w") as fh: json.dump(result, fh, indent=2, sort_keys=True) fh.write("\n") print("%s: %d regression(s), %d engine problem(s), %d fix(es), %d new, %d gone" % (result["verdict"], len(regressions), len(engine_problems), len(fixes), len(added), len(removed))) return 1 if (regressions or engine_problems) else 0 if __name__ == "__main__": sys.exit(main(sys.argv))