#!/usr/bin/env python3
"""
check.py -- verify the offline mirror is self-contained.
Run it on the air-gapped machine, with no network and no server needed:
./check.py
It walks every mirrored page, collects every local URL the page depends on
(stylesheets, scripts, images, fonts, ES-module imports, CSS url(), RHDS
icon modules) and reports the ones that are not on disk.
"The pages load" is not the same as "the pages are complete" -- a missing
web-component module just means that element silently never upgrades.
"""
import argparse
import json
import os
import re
import sys
from urllib.parse import urlsplit, urljoin, unquote
ROOT = os.path.dirname(os.path.abspath(__file__))
SITE = os.path.join(ROOT, "site")
TRACKER_RE = re.compile(
r"(/dtm\.js|trustarc|trustecm|googletagmanager|google-analytics|hotjar"
r"|munchkin|marketo|6sense|rh\.mktg\.js)", re.I)
ASSET_RELS = {"stylesheet", "preload", "modulepreload", "prefetch", "icon",
"shortcut icon", "apple-touch-icon", "manifest", "mask-icon"}
RHDS_ICONS = "/modules/contrib/red_hat_shared_libs/dist/rhds-elements/icons"
def unescape(v):
return v.replace("&", "&").replace("&", "&")
def to_file(url):
"""Local URL -> path under site/, or None if it is not ours to check."""
p = urlsplit(url)
if p.scheme or p.netloc:
return None # absolute, deliberately left external
path = p.path
if not path.startswith("/"):
return None
rel = path.lstrip("/")
if path.endswith("/"):
rel += "index.html"
segs = [unquote(seg).replace("/", "_").replace("\\", "_")
for seg in rel.split("/")]
return os.path.join(SITE, *segs)
class Checker:
def __init__(self, verbose):
self.verbose = verbose
self.missing = {} # url -> set of referrers
self.checked = set()
self.pages = 0
self.refs = 0
self.trackers = []
def note(self, url, referrer):
if url is None:
return
self.refs += 1
target = to_file(url)
if target is None:
return
if not os.path.isfile(target):
self.missing.setdefault(url, set()).add(referrer)
# -- extraction -------------------------------------------------------
def from_html(self, text, referrer):
if TRACKER_RE.search(text):
for m in TRACKER_RE.finditer(text):
self.trackers.append((referrer, m.group(1)))
#
for m in re.finditer(r']*)>', text, re.I):
attrs = m.group(1)
rel = re.search(r'rel=["\']([^"\']+)', attrs, re.I)
href = re.search(r'href=["\']([^"\']+)', attrs, re.I)
if rel and href and rel.group(1).strip().lower() in ASSET_RELS:
self.note(unescape(href.group(1)), referrer)
# ',
text, re.S | re.I)
if m:
try:
importmap = json.loads(m.group(1)).get("imports", {})
break
except ValueError:
pass
if not importmap:
print("WARNING: no