#!/usr/bin/env python3 """ fetch.py -- build an offline mirror of https://www.redhat.com/en/topics/ Run this on a host that HAS internet access. It writes everything under ./site/, which ./serve.py then serves on the air-gapped machine. Python standard library only -- no pip installs, nothing to vendor. Why not wget: * Red Hat's Drupal aggregates CSS/JS behind ~530-character query strings. wget turns those into filenames and blows past the 255-byte limit. The basename already carries a content hash, so we simply drop the query. * Pages load Red Hat Design System components as ES modules through a ', text, re.S | re.I): try: data = json.loads(m.group(1)) except ValueError: continue for bare, target in (data.get("imports") or {}).items(): if bare not in self.importmap: self.importmap[bare] = target def resolve_bare(self, spec, base): """Resolve a bare module specifier through the page's import map.""" best = None for prefix in self.importmap: if spec.startswith(prefix) and (best is None or len(prefix) > len(best)): best = prefix if best is None: return None target = self.importmap[best] + spec[len(best):] return norm_url(target, base) def js_imports(self, text, base): """Static import / re-export specifiers plus literal dynamic imports.""" out = [] patterns = ( r'\bimport\s+[^\'"()]*?from\s*[\'"]([^\'"]+)[\'"]', r'\bimport\s*[\'"]([^\'"]+)[\'"]', r'\bexport\s+[^\'"()]*?from\s*[\'"]([^\'"]+)[\'"]', r'\bimport\s*\(\s*[\'"]([^\'"${}]+)[\'"]\s*\)', ) for pat in patterns: for m in re.finditer(pat, text): spec = m.group(1) if spec.startswith((".", "/")) or "://" in spec: url = norm_url(spec, base) else: url = self.resolve_bare(spec, base) if url: out.append(url) return out def css_refs(self, text, base): out = [] for m in re.finditer(r'url\(\s*[\'"]?([^\'")]+)[\'"]?\s*\)', text): url = norm_url(m.group(1), base) if url: out.append(url) for m in re.finditer(r'@import\s+[\'"]([^\'"]+)[\'"]', text): url = norm_url(m.group(1), base) if url: out.append(url) return out def rhds_component_urls(self, text, base): """ -> the element module; -> the glyph.""" out = [] for tag in set(re.findall(r'<(rh-[a-z0-9-]+)[\s>]', text, re.I)): tag = tag.lower() # rh-footer-block and friends ship inside the rh-footer package, # so fall back to progressively shorter package names. parts = tag.split("-") for cut in range(len(parts), 1, -1): pkg = "-".join(parts[:cut]) url = self.resolve_bare("@rhds/elements/%s/%s.js" % (pkg, pkg), base) if url: out.append(("component", url)) for m in re.finditer(r']*)>', text, re.I): attrs = m.group(1) icon = re.search(r'\bicon=["\']([^"\']+)', attrs) iset = re.search(r'\bset=["\']([^"\']+)', attrs) if icon and iset: url = "https://%s%s/icons/%s/%s.js" % ( SITE_HOST, RHDS_DIST, iset.group(1), icon.group(1)) out.append(("icon", url)) return out # -- rewriting -------------------------------------------------------- def map_page(self, raw, base): """: rewrite only links that stay inside the topic tree. Everything else (products, blog, other languages) keeps the URL it had. Root-relative ones land on serve.py's "not mirrored" page; absolute ones still work if the browser happens to be online. """ url = norm_url(raw, base) if url is None or not is_page(url): return raw self.enqueue(url) return local_href(url) def map_asset(self, raw, base): url = norm_url(raw, base) if url is None or TRACKER_RE.search(url) or not mirrored_host(url): return raw self.enqueue(url) return local_href(url) def map_srcset(self, value, base): parts = [] for item in value.split(","): item = item.strip() if not item: continue bits = item.split(None, 1) new = self.map_asset(bits[0], base) parts.append(new if len(bits) == 1 else "%s %s" % (new, bits[1])) return ", ".join(parts) def rewrite_attrs(self, tag, attrs, base): """Rewrite the URL-valued attributes of one start tag.""" rel = "" m = re.search(r'\brel\s*=\s*["\']([^"\']+)', attrs, re.I) if m: rel = m.group(1).strip().lower() def one(m): name, quote, value = m.group(1).lower(), m.group(2), m.group(3) if tag in ("a", "area") and name == "href": return '%s=%s%s%s' % (m.group(1), quote, self.map_page(value, base), quote) if tag == "link" and name == "href": if rel in ASSET_LINK_RELS: return '%s=%s%s%s' % (m.group(1), quote, self.map_asset(value, base), quote) return m.group(0) # canonical, alternate, ... left alone if tag in ASSET_TAGS and name in ASSET_ATTRS: if name in ("srcset", "data-srcset"): return '%s=%s%s%s' % (m.group(1), quote, self.map_srcset(value, base), quote) return '%s=%s%s%s' % (m.group(1), quote, self.map_asset(value, base), quote) return m.group(0) return re.sub(r'([a-zA-Z_:][-a-zA-Z0-9_:.]*)\s*=\s*(["\'])(.*?)\2', one, attrs, flags=re.S) def strip_trackers(self, text): """Cut out everything that can only reach for the network. Offline these hosts never resolve, so the page stalls on DNS until the browser gives up -- the primary nav and the consent banner are the visible casualties. Three kinds go: * analytics / consent / marketing tags (Adobe DTM, TrustArc, ...); * any script or stylesheet on a host this mirror does not hold, which offline can only hang (one topic page pulls jQuery and Font Awesome from third-party CDNs); * every preconnect / dns-prefetch hint, including ones for hosts we do mirror -- their assets are served from here now. Inline snippets are left alone: they only push onto arrays and guard their own globals, so they are harmless without the network. """ def kill_script(m): tag = m.group(0) if TRACKER_RE.search(tag): return "" src = re.search(r'\bsrc=["\']([^"\']+)', tag, re.I) if src: url = norm_url(src.group(1), START_URL) if url and not mirrored_host(url): return "" return tag text = re.sub(r']*\bsrc=["\'][^"\']+["\'][^>]*>\s*', kill_script, text, flags=re.I) def kill_link(m): tag = m.group(0) if TRACKER_RE.search(tag): return "" rel = re.search(r'rel=["\']([^"\']+)', tag, re.I) rel = rel.group(1).strip().lower() if rel else "" if rel in ("preconnect", "dns-prefetch"): return "" if rel in ASSET_LINK_RELS: href = re.search(r'href=["\']([^"\']+)', tag, re.I) if href: url = norm_url(href.group(1), START_URL) if url and not mirrored_host(url): return "" return tag text = re.sub(r']*>', kill_link, text, flags=re.I) # speculative prerender of /en, which is not part of the mirror text = re.sub(r']*type=["\']speculationrules["\'][^>]*>.*?', "", text, flags=re.S | re.I) return text def rewrite_html(self, text, base): text = self.strip_trackers(text) # Park