#!/usr/bin/env python3 """Mirror https://docs.semgrep.dev (a Mintlify / Next.js site) for offline viewing. Output goes to ../site (relative to this script): site/.html one file per docs page ("/" -> index.html) site/404.html the site's not-found page site/mintlify-assets/... Next.js JS/CSS/font chunks (incl. lazily loaded chunks) site/_ext//... assets from CDNs (mintcdn.com, cloudfront, b-cdn) site/_redirects.json internal links that redirect on the live site Files are stored exactly as downloaded. The CDN URLs inside them are rewritten when they are served (see ../serve.py), because the site's JavaScript calls new URL() on them and crashes if they are rewritten to relative paths on disk. Serve the mirror from the root of a web server. Stdlib only; runs on Python 3.9+. """ import concurrent.futures as cf import json import os import re import sys import time import urllib.error import urllib.parse import urllib.request BASE = "https://docs.semgrep.dev" ASSET_HOSTS = ( "mintcdn.com", "d3gk2c5xim1je2.cloudfront.net", "d4tuoctqmanu0.cloudfront.net", "mintlify.b-cdn.net", ) ASSET_EXT = { "js", "mjs", "css", "woff", "woff2", "ttf", "otf", "eot", "svg", "png", "jpg", "jpeg", "gif", "webp", "avif", "ico", "json", "mp4", "webm", "pdf", "txt", "xml", } TEXT_EXT = {"html", "css", "js", "mjs", "json", "txt", "xml"} UA = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/140.0 Safari/537.36" WORKERS = 8 HERE = os.path.dirname(os.path.abspath(__file__)) SITE = os.path.join(os.path.dirname(HERE), "site") HOSTS_RE = "|".join(re.escape(h) for h in ASSET_HOSTS) RE_REMOTE = re.compile(r"https:(?:\\?/){2}(" + HOSTS_RE + r")((?:\\?/[^\"'\s<>()\\&?#`${}]*)+)") RE_NEXT_STATIC = re.compile(r"static/(?:chunks|media|css)/[A-Za-z0-9_.@-]+\.[a-z0-9]+") RE_LOCAL_ASSET = re.compile( r"[\"'(=](/(?!_mintlify/api|_ext/)[A-Za-z0-9_.@%/-]+\.(?:" + "|".join(sorted(ASSET_EXT)) + r"))[\"')?#\\]" ) RE_PAGE_LINK = re.compile(r"href=(?:\\?\")?\\?\"?(/[^\"'\\#?\s<>]*)|\\\"href\\\":\\\"(/[^\"\\#?]*)") RE_CSS_URL = re.compile(r"url\(\s*['\"]?([^'\")]+)['\"]?\s*\)") failures = [] redirects = {} def log(msg): print(msg, flush=True) def fetch(url, allow_error_body=False): """Return (status, final_url, body). Follows 308 redirects (urllib only handles them from Python 3.11) and retries transient failures, including CDN throttling that shows up as 403/429. """ last = None attempt = hops = 0 while attempt < 4: req = urllib.request.Request(url, headers={"User-Agent": UA, "Accept": "*/*"}) try: with urllib.request.urlopen(req, timeout=60) as r: return r.status, r.geturl(), r.read() except urllib.error.HTTPError as e: loc = e.headers.get("Location") if e.code == 308 and loc and hops < 5: url = urllib.parse.urljoin(url, loc) hops += 1 continue if e.code in (403, 429, 500, 502, 503, 504) and attempt < 3: attempt += 1 time.sleep(3 * attempt) continue if allow_error_body: return e.code, url, e.read() last = "HTTP %d" % e.code break except Exception as e: # network hiccup last = repr(e) attempt += 1 time.sleep(2 * attempt) failures.append("%s %s" % (last, url)) return None, url, None def ext_of(path): last = path.rstrip("/").rsplit("/", 1)[-1] return last.rsplit(".", 1)[-1].lower() if "." in last else "" def page_file(path): path = urllib.parse.unquote(path).rstrip("/") return os.path.join(SITE, "index.html") if not path else os.path.join(SITE, path.lstrip("/") + ".html") def asset_file(url): u = urllib.parse.urlsplit(url) path = urllib.parse.unquote(u.path).lstrip("/") if u.netloc == "docs.semgrep.dev": return os.path.join(SITE, path) return os.path.join(SITE, "_ext", u.netloc, path) def save(dest, body): os.makedirs(os.path.dirname(dest), exist_ok=True) with open(dest, "wb") as f: f.write(body) def discover(url, body, is_css): """Find further pages and assets referenced by a downloaded text file.""" text = body.decode("utf-8", "replace") pages, assets = set(), set() for host, path in RE_REMOTE.findall(text): path = path.replace("\\/", "/") if ext_of(path) in ASSET_EXT: assets.add("https://%s%s" % (host, path)) for m in RE_NEXT_STATIC.findall(text): assets.add(BASE + "/mintlify-assets/_next/" + m) for m in RE_LOCAL_ASSET.findall(text): assets.add(BASE + m) if is_css: for ref in RE_CSS_URL.findall(text): if ref.startswith("data:") or ref.startswith("#"): continue absu = urllib.parse.urljoin(url, ref) u = urllib.parse.urlsplit(absu) if u.netloc in ASSET_HOSTS or u.netloc == "docs.semgrep.dev": assets.add(urllib.parse.urlunsplit((u.scheme, u.netloc, u.path, "", ""))) if url.startswith(BASE) and not is_css and ext_of(urllib.parse.urlsplit(url).path) in ("", "html"): for a, b in RE_PAGE_LINK.findall(text): p = a or b if not p or p.startswith(("/_mintlify", "/mintlify-assets", "/_ext", "/cdn-cgi", "//")): continue if ext_of(p) in ASSET_EXT: assets.add(BASE + p) elif ext_of(p) == "": pages.add(p.rstrip("/") or "/") return pages, assets def get_page(path): status, final, body = fetch(BASE + urllib.parse.quote(path, safe="/%@:+-_.~")) if body is None: return path, set(), set() fpath = urllib.parse.urlsplit(final).path.rstrip("/") or "/" if urllib.parse.unquote(fpath) != urllib.parse.unquote(path): redirects[path] = fpath return path, {fpath}, set() save(page_file(path), body) return (path,) + discover(final, body, False) def get_asset(url): status, final, body = fetch(url) if body is None: return url, set(), set() save(asset_file(url), body) ext = ext_of(urllib.parse.urlsplit(url).path) if ext in TEXT_EXT: return (url,) + discover(url, body, ext == "css") return url, set(), set() def crawl(): sm = fetch(BASE + "/sitemap.xml")[2] if sm is None: sys.exit("cannot fetch sitemap.xml") save(os.path.join(SITE, "sitemap.xml"), sm) pages_todo = {"/"} | { urllib.parse.urlsplit(u).path.rstrip("/") or "/" for u in re.findall(r"([^<]+)", sm.decode()) } log("sitemap: %d pages" % len(pages_todo)) status, _, body = fetch(BASE + "/offline-mirror-not-found-page", allow_error_body=True) if body: save(os.path.join(SITE, "404.html"), body) _, assets_todo = discover(BASE + "/404", body, False) else: assets_todo = set() pages_seen, assets_seen = set(), set() rnd = 0 with cf.ThreadPoolExecutor(WORKERS) as ex: while pages_todo or assets_todo: rnd += 1 pages_todo -= pages_seen assets_todo -= assets_seen if not pages_todo and not assets_todo: break log("round %d: %d pages, %d assets" % (rnd, len(pages_todo), len(assets_todo))) pages_seen |= pages_todo assets_seen |= assets_todo jobs = [ex.submit(get_page, p) for p in pages_todo] + [ex.submit(get_asset, a) for a in assets_todo] pages_todo, assets_todo = set(), set() for i, j in enumerate(cf.as_completed(jobs), 1): _, np, na = j.result() pages_todo |= np assets_todo |= na if i % 200 == 0: log(" %d/%d" % (i, len(jobs))) log("downloaded %d page URLs, %d asset URLs" % (len(pages_seen), len(assets_seen))) backfill_katex_fonts() def backfill_katex_fonts(): """Mintlify's KaTeX CDN only hosts a few fonts (the rest are 403 on the live site too); fill the gaps from the matching KaTeX release on jsDelivr.""" host = "d4tuoctqmanu0.cloudfront.net" css = os.path.join(SITE, "_ext", host, "katex.min.css") if not os.path.isfile(css): return with open(css, encoding="utf-8", errors="replace") as f: text = f.read() ver = re.search(r'"(\d+\.\d+\.\d+)"', text) if not ver: return missing = [n for n in sorted(set(re.findall(r"fonts/(KaTeX_[A-Za-z0-9_-]+\.(?:woff2|woff|ttf))", text))) if not os.path.isfile(os.path.join(SITE, "_ext", host, "fonts", n))] got = 0 for name in missing: body = fetch("https://cdn.jsdelivr.net/npm/katex@%s/dist/fonts/%s" % (ver.group(1), name))[2] if body: save(os.path.join(SITE, "_ext", host, "fonts", name), body) got += 1 failures[:] = [x for x in failures if not (host + "/fonts/" in x and os.path.isfile(os.path.join(SITE, "_ext", host, "fonts", x.rsplit("/", 1)[-1])))] log("KaTeX %s: backfilled %d/%d fonts from jsDelivr" % (ver.group(1), got, len(missing))) def main(): if os.path.exists(SITE) and os.listdir(SITE): sys.exit("%s is not empty; move it aside before re-mirroring" % SITE) os.makedirs(SITE, exist_ok=True) crawl() with open(os.path.join(SITE, "_redirects.json"), "w") as f: json.dump(redirects, f, indent=1, sort_keys=True) with open(os.path.join(HERE, "mirror-failures.log"), "w") as f: f.write("\n".join(sorted(failures)) + "\n") log("%d redirects, %d failed URLs (see tools/mirror-failures.log)" % (len(redirects), len(failures))) if __name__ == "__main__": main()