#!/usr/bin/env python3 """ check.py -- verify the offline mirror is self-contained. Run it on the air-gapped machine, with no network and no server needed: ./check.py It walks every mirrored page, collects every local URL the page depends on (stylesheets, scripts, images, fonts, ES-module imports, CSS url(), RHDS icon modules) and reports the ones that are not on disk. "The pages load" is not the same as "the pages are complete" -- a missing web-component module just means that element silently never upgrades. """ import argparse import json import os import re import sys from urllib.parse import urlsplit, urljoin, unquote ROOT = os.path.dirname(os.path.abspath(__file__)) SITE = os.path.join(ROOT, "site") TRACKER_RE = re.compile( r"(/dtm\.js|trustarc|trustecm|googletagmanager|google-analytics|hotjar" r"|munchkin|marketo|6sense|rh\.mktg\.js)", re.I) ASSET_RELS = {"stylesheet", "preload", "modulepreload", "prefetch", "icon", "shortcut icon", "apple-touch-icon", "manifest", "mask-icon"} RHDS_ICONS = "/modules/contrib/red_hat_shared_libs/dist/rhds-elements/icons" def unescape(v): return v.replace("&", "&").replace("&", "&") def to_file(url): """Local URL -> path under site/, or None if it is not ours to check.""" p = urlsplit(url) if p.scheme or p.netloc: return None # absolute, deliberately left external path = p.path if not path.startswith("/"): return None rel = path.lstrip("/") if path.endswith("/"): rel += "index.html" segs = [unquote(seg).replace("/", "_").replace("\\", "_") for seg in rel.split("/")] return os.path.join(SITE, *segs) class Checker: def __init__(self, verbose): self.verbose = verbose self.missing = {} # url -> set of referrers self.checked = set() self.pages = 0 self.refs = 0 self.trackers = [] def note(self, url, referrer): if url is None: return self.refs += 1 target = to_file(url) if target is None: return if not os.path.isfile(target): self.missing.setdefault(url, set()).add(referrer) # -- extraction ------------------------------------------------------- def from_html(self, text, referrer): if TRACKER_RE.search(text): for m in TRACKER_RE.finditer(text): self.trackers.append((referrer, m.group(1))) # for m in re.finditer(r']*)>', text, re.I): attrs = m.group(1) rel = re.search(r'rel=["\']([^"\']+)', attrs, re.I) href = re.search(r'href=["\']([^"\']+)', attrs, re.I) if rel and href and rel.group(1).strip().lower() in ASSET_RELS: self.note(unescape(href.group(1)), referrer) # ', text, re.S | re.I) if m: try: importmap = json.loads(m.group(1)).get("imports", {}) break except ValueError: pass if not importmap: print("WARNING: no