"""Polite same-domain BFS crawler for tuneecu.net -- mirrors the docs site locally so we can grep/read full raw pages instead of relying on one-page- at-a-time AI-summarized fetches (which have already been shown to drop detail, e.g. the glossary needing a second full pass). No robots.txt on tuneecu.net (404s). Still polite: single-threaded, delay between requests, capped page count, HTML only (binaries/zips are recorded as leaf links but not fetched/recursed into -- there's no reason to mirror the whole map-download tree, just discover what's there). Usage: python3 tuneecu_crawl.py # default seeds, depth 10 python3 tuneecu_crawl.py --max-pages 300 """ from __future__ import annotations import argparse import re import time import urllib.error import urllib.request from collections import deque from pathlib import Path from urllib.parse import urljoin, urlparse DOMAIN = "tuneecu.net" OUT_DIR = Path(__file__).parent / "tuneecu_site_mirror" SEEDS = [ "https://tuneecu.net/TuneECU_En/index.html", "https://tuneecu.net/TuneECU_En/links.html", "https://tuneecu.net/TuneECU_En/glossary.html", "https://tuneecu.net/TuneECU_En/mapedit.html", "https://tuneecu.net/Map_Database.html", ] HREF_RE = re.compile(r'href\s*=\s*["\']([^"\']+)["\']', re.I) HTML_EXTS = (".html", ".htm", "") # "" covers dir-style URLs, treated as html NON_HTML_EXTS = ( ".zip", ".hex", ".pdf", ".mp4", ".jpg", ".jpeg", ".png", ".gif", ".dat", ".md5", ".exe", ".bin", ".rar", ".7z", ".doc", ".docx", ) # Leaf file tree -- thousands of per-model tune files/checksums, not docs. # Record links into it (so we know what's there) but never recurse. NO_RECURSE_PREFIXES = ("https://tuneecu.net/tunes_in_hex_and_dat/",) def _is_html(url: str) -> bool: path = urlparse(url).path.lower() if any(path.endswith(ext) for ext in NON_HTML_EXTS): return False return True def _no_recurse(url: str) -> bool: return url.lower().startswith(NO_RECURSE_PREFIXES) def _same_domain(url: str) -> bool: return urlparse(url).netloc.lower().endswith(DOMAIN) def _local_path(url: str) -> Path: p = urlparse(url) rel = (p.netloc + p.path).strip("/") if not rel or rel.endswith("/"): rel += "index.html" return OUT_DIR / rel def crawl(max_depth: int, max_pages: int, delay: float) -> None: OUT_DIR.mkdir(exist_ok=True) visited: set[str] = set() prior = OUT_DIR / "_visited.txt" if prior.exists(): visited |= {ln.strip() for ln in prior.read_text().splitlines() if ln.strip()} print(f"preloaded {len(visited)} already-visited URLs, will skip re-fetching them") all_links: set[str] = set() # every link seen, html or not queue: deque[tuple[str, int]] = deque((s, 0) for s in SEEDS) prior_links = OUT_DIR / "_all_links.txt" if prior_links.exists(): all_links |= {ln.strip() for ln in prior_links.read_text().splitlines() if ln.strip()} for u in sorted(all_links): if _same_domain(u) and _is_html(u) and not _no_recurse(u) and u not in visited: queue.append((u, 1)) print(f"seeded queue with {len(queue)} unfetched html links from prior run") fetched = 0 while queue and fetched < max_pages: url, depth = queue.popleft() if url in visited or depth > max_depth: continue visited.add(url) req = urllib.request.Request(url, headers={"User-Agent": "tunie-research-crawler/1.0"}) try: with urllib.request.urlopen(req, timeout=20) as r: body = r.read() except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError) as exc: print(f" skip {url}: {exc}") continue fetched += 1 dest = _local_path(url) try: if dest.is_dir(): dest = dest / "index.html" dest.parent.mkdir(parents=True, exist_ok=True) dest.write_bytes(body) except (IsADirectoryError, NotADirectoryError, FileNotFoundError) as exc: print(f" write skip {url}: {exc}") print(f"[{fetched}/{max_pages}] depth={depth} {url} ({len(body)}B)") try: text = body.decode("utf-8", "replace") except Exception: text = "" for href in HREF_RE.findall(text): absu = urljoin(url, href).split("#")[0] if not absu.startswith("http"): continue all_links.add(absu) if ( _same_domain(absu) and _is_html(absu) and not _no_recurse(absu) and absu not in visited and depth + 1 <= max_depth ): queue.append((absu, depth + 1)) time.sleep(delay) (OUT_DIR / "_all_links.txt").write_text("\n".join(sorted(all_links))) (OUT_DIR / "_visited.txt").write_text("\n".join(sorted(visited))) print(f"\nFetched {fetched} pages, {len(visited)} visited, {len(all_links)} total links discovered.") print(f"Mirror in {OUT_DIR}") if __name__ == "__main__": ap = argparse.ArgumentParser() ap.add_argument("--max-depth", type=int, default=10) ap.add_argument("--max-pages", type=int, default=250) ap.add_argument("--delay", type=float, default=0.4) args = ap.parse_args() crawl(args.max_depth, args.max_pages, args.delay)