#!/usr/bin/env python3 """Mirror https://docs.eazybi.com/eazybi (server-rendered Rails docs) for offline viewing. Output goes to ../site (relative to this script): site/eazybi/index.html home page site/eazybi//index.html one file per documentation page site/eazybi/attachments/..., thumbnails/..., download/... page images and files site/assets/... the site's CSS/JS/images site/_ext//... external assets (Font Awesome CSS + webfonts) site/404.html the site's not-found page site/_redirects.json old page paths that redirect on the live site A URL with a query string (attachments carry ?version=...&modificationDate=...) is saved as "@"; the file without the suffix is also written for the first variant seen. Files are stored exactly as downloaded; ../serve.py rewrites external URLs. Stdlib only; runs on Python 3.9+. """ import concurrent.futures as cf import gzip import hashlib import html import json import os import re import sys import threading import time import urllib.error import urllib.parse import urllib.request BASE = "https://docs.eazybi.com" SPACE = "/eazybi" EXT_HOSTS = ("eazybi.com",) # Font Awesome kit lives at https://eazybi.com/static/fontawesome/ UA = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/140.0 Safari/537.36" WORKERS = 1 DELAY = 1.5 # seconds between requests; the site rate-limits (HTTP 429) aggressive crawlers HERE = os.path.dirname(os.path.abspath(__file__)) SITE = os.path.join(os.path.dirname(HERE), "site") ASSET_PREFIXES = ("/assets/", SPACE + "/attachments/", SPACE + "/thumbnails/", SPACE + "/download/") NOT_PAGES = (SPACE + "/search", SPACE + "/sitemap.xml") RE_ATTR = re.compile(r'(?:href|src|data-src|poster|data-image-src)\s*=\s*"([^"]+)"') RE_SRCSET = re.compile(r'srcset\s*=\s*"([^"]+)"') RE_CSS_URL = re.compile(r"url\(\s*['\"]?([^'\")]+)['\"]?\s*\)") RE_LOC = re.compile(r"([^<]+)") RESUME = "--fresh" not in sys.argv # reuse pages already on disk unless --fresh lock = threading.Lock() save_lock = threading.Lock() failures = [] redirects = {} # old page path -> new path (live site answers 301) _last = [0.0] def fetch(url): req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept-Encoding": "identity"}) for attempt in range(8): with lock: # global politeness delay wait = _last[0] + DELAY - time.time() if wait > 0: time.sleep(wait) _last[0] = time.time() try: with urllib.request.urlopen(req, timeout=60) as r: data = r.read() # eazybi.com sends gzip even for Accept-Encoding: identity if r.headers.get("Content-Encoding") == "gzip" or data[:2] == b"\x1f\x8b" and not url.endswith(".gz"): data = gzip.decompress(data) return r.status, data, r.geturl() except urllib.error.HTTPError as e: if e.code in (404, 410, 403): return e.code, e.read(), url err = e if e.code == 429: # rate limited: honour Retry-After, else back off ra = e.headers.get("Retry-After", "") pause = int(ra) if ra.isdigit() else min(600, 60 * 2 ** attempt) print(f" 429 on {url}; pausing {pause}s", flush=True) time.sleep(pause) continue except Exception as e: # noqa: BLE001 - retry on network errors err = e time.sleep(2 * (attempt + 1)) raise err def local_path(url): """Map an absolute URL to a path under SITE, or None if out of scope.""" u = urllib.parse.urlsplit(url) path = urllib.parse.unquote(u.path) if u.netloc == "docs.eazybi.com": if path.startswith(ASSET_PREFIXES): rel = path.lstrip("/") elif path == SPACE or path.startswith(SPACE + "/"): rel = path.lstrip("/").rstrip("/") + "/index.html" else: return None elif u.netloc in EXT_HOSTS and path.startswith("/static/"): rel = "_ext/" + u.netloc + path else: return None if ".." in rel.split("/"): return None full = os.path.join(SITE, rel) if u.query and not rel.endswith("/index.html"): return full, full + "@" + hashlib.sha1(u.query.encode()).hexdigest()[:12] return full, None def save(path, data): os.makedirs(os.path.dirname(path), exist_ok=True) tmp = path + ".part" with open(tmp, "wb") as f: f.write(data) os.replace(tmp, path) def is_page(url): u = urllib.parse.urlsplit(url) p = u.path.rstrip("/") return (u.netloc == "docs.eazybi.com" and (p == SPACE or p.startswith(SPACE + "/")) and not p.startswith(ASSET_PREFIXES) and not p.startswith(NOT_PAGES)) def links(base, text): out = set() for raw in RE_ATTR.findall(text): out.add(raw) for s in RE_SRCSET.findall(text): out.update(part.strip().split(" ")[0] for part in s.split(",")) for raw in RE_CSS_URL.findall(text): out.add(raw) res = set() for raw in out: raw = html.unescape(raw).strip() if not raw or raw.startswith(("data:", "mailto:", "javascript:", "#")): continue res.add(urllib.parse.urljoin(base, raw).split("#")[0]) return res def crawl(): status, body, _ = fetch(BASE + SPACE + "/sitemap.xml") queue = {html.unescape(u).rstrip("/") for u in RE_LOC.findall(body.decode())} queue.add(BASE + SPACE) print(f"sitemap: {len(queue)} pages", flush=True) seen = set() assets = set() pages = 0 def do_page(url): target, _ = local_path(url) if RESUME and os.path.exists(target): data = open(target, "rb").read() return url, 200, links(url, data.decode("utf-8", "replace")) try: st, data, final = fetch(url) except Exception as e: # noqa: BLE001 return url, repr(e)[:60], set() text = data.decode("utf-8", "replace") if st != 200: return url, st, set() target, _ = local_path(final) or local_path(url) save(target, data) old, new = urllib.parse.urlsplit(url).path, urllib.parse.urlsplit(final).path if old.rstrip("/") != new.rstrip("/") and is_page(final): redirects[old.rstrip("/")] = new.rstrip("/") return url, st, links(final, text) with cf.ThreadPoolExecutor(WORKERS) as ex: while queue: batch = [u for u in queue if u not in seen] queue = set() seen.update(batch) for url, st, found in ex.map(do_page, batch): if st != 200: failures.append((st, url)) continue pages += 1 for l in found: u = urllib.parse.urlsplit(l) if is_page(l): clean = urllib.parse.urlunsplit(("https", u.netloc, u.path.rstrip("/"), "", "")) if clean not in seen: queue.add(clean) elif local_path(l): assets.add(l) print(f"pages: {pages} done, {len(queue)} newly found", flush=True) rpath = os.path.join(SITE, "_redirects.json") if os.path.exists(rpath): with open(rpath) as f: redirects.update({k: v for k, v in json.load(f).items() if k not in redirects}) with open(rpath, "w") as f: json.dump(redirects, f, indent=1, sort_keys=True) # 404 page (for the local server) st, data, _ = fetch(BASE + SPACE + "/__offline_404__") save(os.path.join(SITE, "404.html"), data) # assets, then anything referenced from CSS (webfonts) done = set() while assets: batch = sorted(assets - done) assets = set() done.update(batch) print(f"assets: fetching {len(batch)}", flush=True) def do_asset(url): full, variant = local_path(url) if RESUME and os.path.exists(variant or full): data = open(variant or full, "rb").read() else: try: st, data, _ = fetch(url) except Exception as e: # noqa: BLE001 return url, repr(e)[:60], set() if st != 200: return url, st, set() if variant: save(variant, data) with save_lock: if not os.path.exists(full): save(full, data) else: save(full, data) found = set() if full.endswith(".css"): found = {l for l in links(url, data.decode("utf-8", "replace")) if local_path(l)} return url, 200, found with cf.ThreadPoolExecutor(WORKERS) as ex: for url, st, found in ex.map(do_asset, batch): if st != 200: failures.append((st, url)) assets.update(found - done) print(f"done: {pages} pages, {len(done)} assets, {len(failures)} failures") for st, url in sorted(failures, key=str): print(f" FAIL {st} {url}") if __name__ == "__main__": os.makedirs(SITE, exist_ok=True) crawl() sys.exit(0)