Removes the *.hex/maps_cache gitignore rule (explicit user call, reversing the earlier no-redistribution stance) so the official TuneECU catalogue maps, derived SAI/O2-delete composites, and the checksum/composition tooling are actually available to pull up on a phone browser when using the real TuneECU app. Also folds in tonight's KWP2000 fixes (TesterPresent keep-alive, connect-failure cleanup, slow-init StartCommunication fix) and the accumulated research docs. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01FP2GaxS9HkUdL5sLBnjKje
149 lines
5.3 KiB
Python
149 lines
5.3 KiB
Python
"""Polite same-domain BFS crawler for tuneecu.net -- mirrors the docs site
|
|
locally so we can grep/read full raw pages instead of relying on one-page-
|
|
at-a-time AI-summarized fetches (which have already been shown to drop
|
|
detail, e.g. the glossary needing a second full pass).
|
|
|
|
No robots.txt on tuneecu.net (404s). Still polite: single-threaded, delay
|
|
between requests, capped page count, HTML only (binaries/zips are recorded
|
|
as leaf links but not fetched/recursed into -- there's no reason to mirror
|
|
the whole map-download tree, just discover what's there).
|
|
|
|
Usage:
|
|
python3 tuneecu_crawl.py # default seeds, depth 10
|
|
python3 tuneecu_crawl.py --max-pages 300
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import argparse
|
|
import re
|
|
import time
|
|
import urllib.error
|
|
import urllib.request
|
|
from collections import deque
|
|
from pathlib import Path
|
|
from urllib.parse import urljoin, urlparse
|
|
|
|
DOMAIN = "tuneecu.net"
|
|
OUT_DIR = Path(__file__).parent / "tuneecu_site_mirror"
|
|
SEEDS = [
|
|
"https://tuneecu.net/TuneECU_En/index.html",
|
|
"https://tuneecu.net/TuneECU_En/links.html",
|
|
"https://tuneecu.net/TuneECU_En/glossary.html",
|
|
"https://tuneecu.net/TuneECU_En/mapedit.html",
|
|
"https://tuneecu.net/Map_Database.html",
|
|
]
|
|
HREF_RE = re.compile(r'href\s*=\s*["\']([^"\']+)["\']', re.I)
|
|
HTML_EXTS = (".html", ".htm", "") # "" covers dir-style URLs, treated as html
|
|
|
|
|
|
NON_HTML_EXTS = (
|
|
".zip", ".hex", ".pdf", ".mp4", ".jpg", ".jpeg", ".png", ".gif",
|
|
".dat", ".md5", ".exe", ".bin", ".rar", ".7z", ".doc", ".docx",
|
|
)
|
|
# Leaf file tree -- thousands of per-model tune files/checksums, not docs.
|
|
# Record links into it (so we know what's there) but never recurse.
|
|
NO_RECURSE_PREFIXES = ("https://tuneecu.net/tunes_in_hex_and_dat/",)
|
|
|
|
|
|
def _is_html(url: str) -> bool:
|
|
path = urlparse(url).path.lower()
|
|
if any(path.endswith(ext) for ext in NON_HTML_EXTS):
|
|
return False
|
|
return True
|
|
|
|
|
|
def _no_recurse(url: str) -> bool:
|
|
return url.lower().startswith(NO_RECURSE_PREFIXES)
|
|
|
|
|
|
def _same_domain(url: str) -> bool:
|
|
return urlparse(url).netloc.lower().endswith(DOMAIN)
|
|
|
|
|
|
def _local_path(url: str) -> Path:
|
|
p = urlparse(url)
|
|
rel = (p.netloc + p.path).strip("/")
|
|
if not rel or rel.endswith("/"):
|
|
rel += "index.html"
|
|
return OUT_DIR / rel
|
|
|
|
|
|
def crawl(max_depth: int, max_pages: int, delay: float) -> None:
|
|
OUT_DIR.mkdir(exist_ok=True)
|
|
visited: set[str] = set()
|
|
prior = OUT_DIR / "_visited.txt"
|
|
if prior.exists():
|
|
visited |= {ln.strip() for ln in prior.read_text().splitlines() if ln.strip()}
|
|
print(f"preloaded {len(visited)} already-visited URLs, will skip re-fetching them")
|
|
all_links: set[str] = set() # every link seen, html or not
|
|
queue: deque[tuple[str, int]] = deque((s, 0) for s in SEEDS)
|
|
prior_links = OUT_DIR / "_all_links.txt"
|
|
if prior_links.exists():
|
|
all_links |= {ln.strip() for ln in prior_links.read_text().splitlines() if ln.strip()}
|
|
for u in sorted(all_links):
|
|
if _same_domain(u) and _is_html(u) and not _no_recurse(u) and u not in visited:
|
|
queue.append((u, 1))
|
|
print(f"seeded queue with {len(queue)} unfetched html links from prior run")
|
|
fetched = 0
|
|
|
|
while queue and fetched < max_pages:
|
|
url, depth = queue.popleft()
|
|
if url in visited or depth > max_depth:
|
|
continue
|
|
visited.add(url)
|
|
|
|
req = urllib.request.Request(url, headers={"User-Agent": "tunie-research-crawler/1.0"})
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=20) as r:
|
|
body = r.read()
|
|
except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError) as exc:
|
|
print(f" skip {url}: {exc}")
|
|
continue
|
|
|
|
fetched += 1
|
|
dest = _local_path(url)
|
|
try:
|
|
if dest.is_dir():
|
|
dest = dest / "index.html"
|
|
dest.parent.mkdir(parents=True, exist_ok=True)
|
|
dest.write_bytes(body)
|
|
except (IsADirectoryError, NotADirectoryError, FileNotFoundError) as exc:
|
|
print(f" write skip {url}: {exc}")
|
|
print(f"[{fetched}/{max_pages}] depth={depth} {url} ({len(body)}B)")
|
|
|
|
try:
|
|
text = body.decode("utf-8", "replace")
|
|
except Exception:
|
|
text = ""
|
|
|
|
for href in HREF_RE.findall(text):
|
|
absu = urljoin(url, href).split("#")[0]
|
|
if not absu.startswith("http"):
|
|
continue
|
|
all_links.add(absu)
|
|
if (
|
|
_same_domain(absu)
|
|
and _is_html(absu)
|
|
and not _no_recurse(absu)
|
|
and absu not in visited
|
|
and depth + 1 <= max_depth
|
|
):
|
|
queue.append((absu, depth + 1))
|
|
|
|
time.sleep(delay)
|
|
|
|
(OUT_DIR / "_all_links.txt").write_text("\n".join(sorted(all_links)))
|
|
(OUT_DIR / "_visited.txt").write_text("\n".join(sorted(visited)))
|
|
print(f"\nFetched {fetched} pages, {len(visited)} visited, {len(all_links)} total links discovered.")
|
|
print(f"Mirror in {OUT_DIR}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--max-depth", type=int, default=10)
|
|
ap.add_argument("--max-pages", type=int, default=250)
|
|
ap.add_argument("--delay", type=float, default=0.4)
|
|
args = ap.parse_args()
|
|
crawl(args.max_depth, args.max_pages, args.delay)
|