#!/bin/sh # Compare a Ruby dnstraverse capture against an ExploreDNS capture, ignoring # run-to-run noise that is not a behaviour difference: # - ANSI colour codes # - which root server was picked (the reference picks one at random) # - TTL drift between the two runs # - RR ordering within one RRset / answer block # - per-run child ordering (the reference shuffles children, so refids are # assigned to different servers each run; refids are reduced to their # structural shape and progress lines compared as a sorted multiset) # What IS compared: aggregated Results percentages + statuses + server(ip) # attribution, Summary lines, all wording, refid structure and the shape of # every progress line. Server-version fingerprints are ignored (different # fingerprint databases), but the servers-encountered name:IP rows compare. # # Usage: tools/golden/compare.sh REFERENCE_CAPTURE GO_CAPTURE # Exit 0 when equivalent; exit 1 with a unified diff of the normalized forms. set -eu if [ $# -ne 2 ]; then echo "usage: $0 REFERENCE_CAPTURE GO_CAPTURE" >&2 exit 2 fi tmpdir=$(mktemp -d) trap 'rm -rf "$tmpdir"' EXIT normalize() { python3 - "$1" <<'PYEOF' import re import sys path = sys.argv[1] with open(path, encoding="utf-8", errors="replace") as f: text = f.read() # Strip ANSI escape sequences. text = re.sub(r"\x1b\[[0-9;]*[A-Za-z]", "", text) lines = text.split("\n") # The initial root is a per-run random choice: find it and substitute # placeholders everywhere (progress line 1, servers rows, results). rootname = rootip = None m_root = re.compile(r"^Using (\S+) \(([^)]+)\) as initial root$") for ln in lines: m = m_root.match(ln) if m: rootname, rootip = m.group(1), m.group(2) break def subst_root(s): if rootname: s = re.sub(re.escape(rootname), "ROOTNAME", s, flags=re.IGNORECASE) s = s.replace(rootip, "ROOTIP") return s def norm_rr(s): # dig-style RR: name ttl class type rdata... -> collapse whitespace and # blank the TTL so it cannot drift between the two runs. parts = s.split() if len(parts) >= 4 and parts[1].isdigit() and parts[2] in ("IN", "CH", "HS"): parts[1] = "TTL" return " ".join(parts) def refid_shape(refid): # Reduce a refid to its structure: every component is 'd' except the # literal 0 that marks a glue-resolution subtree (1.2.0.1 -> d.d.0.d). return ".".join("0" if p == "0" else "d" for p in refid.split(".")) re_progress = re.compile(r"^(\d+(?:\.\d+)*) (.*)$") re_result_head = re.compile(r"^\s*\d+(?:\.\d+)?%: ") re_summary_head = re.compile(r"^\s*\d+(?:\.\d+)?% ") re_completed = re.compile(r"completed earlier \((\d+(?:\.\d+)*)\)") header, progress, servers, results, summary = [], [], [], [], [] section = "header" i = 0 while i < len(lines): ln = lines[i].rstrip() if ln == "The following servers were encountered:": section = "servers" i += 1 continue if ln == "Results:": section = "results" i += 1 continue if ln == "Summary Results:": section = "summary" i += 1 continue if section == "header": m = re_progress.match(ln) if m: section = "progress" continue # reprocess as progress if ln: header.append(subst_root(ln)) i += 1 continue if section == "progress": m = re_progress.match(ln) if m: rest = m.group(2) # sort the IP list inside "(ip1,ip2,...)" -- RRset order noise def sort_ips(mm): return "(" + ",".join(sorted(mm.group(1).split(","))) + ")" rest = re.sub(r"\(([^)]*)\)$", sort_ips, rest) rest = re.sub( r"\(([^)]*)\)( -- .*)$", lambda mm: "(" + ",".join(sorted(mm.group(1).split(","))) + ")" + mm.group(2), rest, ) rest = re_completed.sub( lambda mm: "completed earlier (%s)" % refid_shape(mm.group(1)), rest ) progress.append("%s %s" % (refid_shape(m.group(1)), subst_root(rest))) elif ln: progress.append(subst_root(ln)) i += 1 continue if section == "servers": if ln: # " name: ip version-text" -> keep name:ip, drop versions m = re.match(r"^\s*(\S+): (\S+)", ln) if m: servers.append(subst_root("%s: %s" % (m.group(1), m.group(2)))) else: servers.append(subst_root(ln.strip())) i += 1 continue if section == "results": if re_result_head.match(ln): head = subst_root(" ".join(ln.split())) rrs, extra = [], [] i += 1 while i < len(lines): nxt = lines[i].rstrip() if not nxt or re_result_head.match(nxt) or nxt in ( "Results:", "Summary Results:", "The following servers were encountered:", ): break if nxt.lstrip().startswith("While querying"): extra.append(" ".join(nxt.split())) else: rrs.append(norm_rr(nxt)) i += 1 block = [head] + sorted(rrs) + extra results.append("\n".join(" " + b if n else b for n, b in enumerate(block))) continue i += 1 continue if section == "summary": if re_summary_head.match(ln): flat = " ".join(ln.split()) m = re.match(r"^(\d+(?:\.\d+)?% answered with )(.*)$", flat) if m: rrs = [norm_rr(m.group(2))] i += 1 while i < len(lines): nxt = lines[i].rstrip() if not nxt or re_summary_head.match(nxt): break rrs.append(norm_rr(nxt)) i += 1 summary.append(m.group(1) + " | ".join(sorted(rrs))) continue summary.append(flat) i += 1 continue i += 1 out = [] out.append("== HEADER ==") out.extend(header) out.append("== PROGRESS (sorted shapes) ==") out.extend(sorted(progress)) if servers: out.append("== SERVERS (name: ip) ==") out.extend(sorted(servers)) out.append("== RESULTS (sorted blocks) ==") out.extend(sorted(results)) out.append("== SUMMARY (sorted) ==") out.extend(sorted(summary)) # Sanity: aggregated Results percentages should sum to ~100. total = 0.0 for r in results: m = re.match(r"^\s*(\d+(?:\.\d+)?)%:", r) if m: total += float(m.group(1)) if results: out.append("== RESULTS TOTAL ~ %d%% ==" % round(total)) print("\n".join(out)) PYEOF } normalize "$1" > "$tmpdir/ref.norm" normalize "$2" > "$tmpdir/go.norm" if diff -u --label "reference:$1" --label "exploredns:$2" \ "$tmpdir/ref.norm" "$tmpdir/go.norm"; then echo "MATCH: $1 == $2 (normalized)" else exit 1 fi