#!/usr/bin/env python3
"""
fetch.py -- build an offline mirror of https://www.redhat.com/en/topics/
Run this on a host that HAS internet access. It writes everything under
./site/, which ./serve.py then serves on the air-gapped machine.
Python standard library only -- no pip installs, nothing to vendor.
Why not wget:
* Red Hat's Drupal aggregates CSS/JS behind ~530-character query strings.
wget turns those into filenames and blows past the 255-byte limit.
The basename already carries a content hash, so we simply drop the query.
* Pages load Red Hat Design System components as ES modules through a
',
text, re.S | re.I):
try:
data = json.loads(m.group(1))
except ValueError:
continue
for bare, target in (data.get("imports") or {}).items():
if bare not in self.importmap:
self.importmap[bare] = target
def resolve_bare(self, spec, base):
"""Resolve a bare module specifier through the page's import map."""
best = None
for prefix in self.importmap:
if spec.startswith(prefix) and (best is None or len(prefix) > len(best)):
best = prefix
if best is None:
return None
target = self.importmap[best] + spec[len(best):]
return norm_url(target, base)
def js_imports(self, text, base):
"""Static import / re-export specifiers plus literal dynamic imports."""
out = []
patterns = (
r'\bimport\s+[^\'"()]*?from\s*[\'"]([^\'"]+)[\'"]',
r'\bimport\s*[\'"]([^\'"]+)[\'"]',
r'\bexport\s+[^\'"()]*?from\s*[\'"]([^\'"]+)[\'"]',
r'\bimport\s*\(\s*[\'"]([^\'"${}]+)[\'"]\s*\)',
)
for pat in patterns:
for m in re.finditer(pat, text):
spec = m.group(1)
if spec.startswith((".", "/")) or "://" in spec:
url = norm_url(spec, base)
else:
url = self.resolve_bare(spec, base)
if url:
out.append(url)
return out
def css_refs(self, text, base):
out = []
for m in re.finditer(r'url\(\s*[\'"]?([^\'")]+)[\'"]?\s*\)', text):
url = norm_url(m.group(1), base)
if url:
out.append(url)
for m in re.finditer(r'@import\s+[\'"]([^\'"]+)[\'"]', text):
url = norm_url(m.group(1), base)
if url:
out.append(url)
return out
def rhds_component_urls(self, text, base):
""" -> the element module; -> the glyph."""
out = []
for tag in set(re.findall(r'<(rh-[a-z0-9-]+)[\s>]', text, re.I)):
tag = tag.lower()
# rh-footer-block and friends ship inside the rh-footer package,
# so fall back to progressively shorter package names.
parts = tag.split("-")
for cut in range(len(parts), 1, -1):
pkg = "-".join(parts[:cut])
url = self.resolve_bare("@rhds/elements/%s/%s.js" % (pkg, pkg), base)
if url:
out.append(("component", url))
for m in re.finditer(r']*)>', text, re.I):
attrs = m.group(1)
icon = re.search(r'\bicon=["\']([^"\']+)', attrs)
iset = re.search(r'\bset=["\']([^"\']+)', attrs)
if icon and iset:
url = "https://%s%s/icons/%s/%s.js" % (
SITE_HOST, RHDS_DIST, iset.group(1), icon.group(1))
out.append(("icon", url))
return out
# -- rewriting --------------------------------------------------------
def map_page(self, raw, base):
""": rewrite only links that stay inside the topic tree.
Everything else (products, blog, other languages) keeps the URL it
had. Root-relative ones land on serve.py's "not mirrored" page;
absolute ones still work if the browser happens to be online.
"""
url = norm_url(raw, base)
if url is None or not is_page(url):
return raw
self.enqueue(url)
return local_href(url)
def map_asset(self, raw, base):
url = norm_url(raw, base)
if url is None or TRACKER_RE.search(url) or not mirrored_host(url):
return raw
self.enqueue(url)
return local_href(url)
def map_srcset(self, value, base):
parts = []
for item in value.split(","):
item = item.strip()
if not item:
continue
bits = item.split(None, 1)
new = self.map_asset(bits[0], base)
parts.append(new if len(bits) == 1 else "%s %s" % (new, bits[1]))
return ", ".join(parts)
def rewrite_attrs(self, tag, attrs, base):
"""Rewrite the URL-valued attributes of one start tag."""
rel = ""
m = re.search(r'\brel\s*=\s*["\']([^"\']+)', attrs, re.I)
if m:
rel = m.group(1).strip().lower()
def one(m):
name, quote, value = m.group(1).lower(), m.group(2), m.group(3)
if tag in ("a", "area") and name == "href":
return '%s=%s%s%s' % (m.group(1), quote,
self.map_page(value, base), quote)
if tag == "link" and name == "href":
if rel in ASSET_LINK_RELS:
return '%s=%s%s%s' % (m.group(1), quote,
self.map_asset(value, base), quote)
return m.group(0) # canonical, alternate, ... left alone
if tag in ASSET_TAGS and name in ASSET_ATTRS:
if name in ("srcset", "data-srcset"):
return '%s=%s%s%s' % (m.group(1), quote,
self.map_srcset(value, base), quote)
return '%s=%s%s%s' % (m.group(1), quote,
self.map_asset(value, base), quote)
return m.group(0)
return re.sub(r'([a-zA-Z_:][-a-zA-Z0-9_:.]*)\s*=\s*(["\'])(.*?)\2',
one, attrs, flags=re.S)
def strip_trackers(self, text):
"""Cut out everything that can only reach for the network.
Offline these hosts never resolve, so the page stalls on DNS until the
browser gives up -- the primary nav and the consent banner are the
visible casualties. Three kinds go:
* analytics / consent / marketing tags (Adobe DTM, TrustArc, ...);
* any script or stylesheet on a host this mirror does not hold,
which offline can only hang (one topic page pulls jQuery and
Font Awesome from third-party CDNs);
* every preconnect / dns-prefetch hint, including ones for hosts we
do mirror -- their assets are served from here now.
Inline snippets are left alone: they only push onto arrays and guard
their own globals, so they are harmless without the network.
"""
def kill_script(m):
tag = m.group(0)
if TRACKER_RE.search(tag):
return ""
src = re.search(r'\bsrc=["\']([^"\']+)', tag, re.I)
if src:
url = norm_url(src.group(1), START_URL)
if url and not mirrored_host(url):
return ""
return tag
text = re.sub(r'',
kill_script, text, flags=re.I)
def kill_link(m):
tag = m.group(0)
if TRACKER_RE.search(tag):
return ""
rel = re.search(r'rel=["\']([^"\']+)', tag, re.I)
rel = rel.group(1).strip().lower() if rel else ""
if rel in ("preconnect", "dns-prefetch"):
return ""
if rel in ASSET_LINK_RELS:
href = re.search(r'href=["\']([^"\']+)', tag, re.I)
if href:
url = norm_url(href.group(1), START_URL)
if url and not mirrored_host(url):
return ""
return tag
text = re.sub(r']*>', kill_link, text, flags=re.I)
# speculative prerender of /en, which is not part of the mirror
text = re.sub(r'',
"", text, flags=re.S | re.I)
return text
def rewrite_html(self, text, base):
text = self.strip_trackers(text)
# Park