Files
samplez/tunie/research/tuneecu_crawl.py
uhryniuk 4aa5da53d2 Commit tune maps and research so tunes are reachable from the phone
Removes the *.hex/maps_cache gitignore rule (explicit user call, reversing
the earlier no-redistribution stance) so the official TuneECU catalogue
maps, derived SAI/O2-delete composites, and the checksum/composition
tooling are actually available to pull up on a phone browser when using
the real TuneECU app. Also folds in tonight's KWP2000 fixes (TesterPresent
keep-alive, connect-failure cleanup, slow-init StartCommunication fix) and
the accumulated research docs.

Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01FP2GaxS9HkUdL5sLBnjKje
2026-08-27 01:34:05 -05:00

149 lines
5.3 KiB
Python

"""Polite same-domain BFS crawler for tuneecu.net -- mirrors the docs site
locally so we can grep/read full raw pages instead of relying on one-page-
at-a-time AI-summarized fetches (which have already been shown to drop
detail, e.g. the glossary needing a second full pass).
No robots.txt on tuneecu.net (404s). Still polite: single-threaded, delay
between requests, capped page count, HTML only (binaries/zips are recorded
as leaf links but not fetched/recursed into -- there's no reason to mirror
the whole map-download tree, just discover what's there).
Usage:
python3 tuneecu_crawl.py # default seeds, depth 10
python3 tuneecu_crawl.py --max-pages 300
"""
from __future__ import annotations
import argparse
import re
import time
import urllib.error
import urllib.request
from collections import deque
from pathlib import Path
from urllib.parse import urljoin, urlparse
DOMAIN = "tuneecu.net"
OUT_DIR = Path(__file__).parent / "tuneecu_site_mirror"
SEEDS = [
"https://tuneecu.net/TuneECU_En/index.html",
"https://tuneecu.net/TuneECU_En/links.html",
"https://tuneecu.net/TuneECU_En/glossary.html",
"https://tuneecu.net/TuneECU_En/mapedit.html",
"https://tuneecu.net/Map_Database.html",
]
HREF_RE = re.compile(r'href\s*=\s*["\']([^"\']+)["\']', re.I)
HTML_EXTS = (".html", ".htm", "") # "" covers dir-style URLs, treated as html
NON_HTML_EXTS = (
".zip", ".hex", ".pdf", ".mp4", ".jpg", ".jpeg", ".png", ".gif",
".dat", ".md5", ".exe", ".bin", ".rar", ".7z", ".doc", ".docx",
)
# Leaf file tree -- thousands of per-model tune files/checksums, not docs.
# Record links into it (so we know what's there) but never recurse.
NO_RECURSE_PREFIXES = ("https://tuneecu.net/tunes_in_hex_and_dat/",)
def _is_html(url: str) -> bool:
path = urlparse(url).path.lower()
if any(path.endswith(ext) for ext in NON_HTML_EXTS):
return False
return True
def _no_recurse(url: str) -> bool:
return url.lower().startswith(NO_RECURSE_PREFIXES)
def _same_domain(url: str) -> bool:
return urlparse(url).netloc.lower().endswith(DOMAIN)
def _local_path(url: str) -> Path:
p = urlparse(url)
rel = (p.netloc + p.path).strip("/")
if not rel or rel.endswith("/"):
rel += "index.html"
return OUT_DIR / rel
def crawl(max_depth: int, max_pages: int, delay: float) -> None:
OUT_DIR.mkdir(exist_ok=True)
visited: set[str] = set()
prior = OUT_DIR / "_visited.txt"
if prior.exists():
visited |= {ln.strip() for ln in prior.read_text().splitlines() if ln.strip()}
print(f"preloaded {len(visited)} already-visited URLs, will skip re-fetching them")
all_links: set[str] = set() # every link seen, html or not
queue: deque[tuple[str, int]] = deque((s, 0) for s in SEEDS)
prior_links = OUT_DIR / "_all_links.txt"
if prior_links.exists():
all_links |= {ln.strip() for ln in prior_links.read_text().splitlines() if ln.strip()}
for u in sorted(all_links):
if _same_domain(u) and _is_html(u) and not _no_recurse(u) and u not in visited:
queue.append((u, 1))
print(f"seeded queue with {len(queue)} unfetched html links from prior run")
fetched = 0
while queue and fetched < max_pages:
url, depth = queue.popleft()
if url in visited or depth > max_depth:
continue
visited.add(url)
req = urllib.request.Request(url, headers={"User-Agent": "tunie-research-crawler/1.0"})
try:
with urllib.request.urlopen(req, timeout=20) as r:
body = r.read()
except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError) as exc:
print(f" skip {url}: {exc}")
continue
fetched += 1
dest = _local_path(url)
try:
if dest.is_dir():
dest = dest / "index.html"
dest.parent.mkdir(parents=True, exist_ok=True)
dest.write_bytes(body)
except (IsADirectoryError, NotADirectoryError, FileNotFoundError) as exc:
print(f" write skip {url}: {exc}")
print(f"[{fetched}/{max_pages}] depth={depth} {url} ({len(body)}B)")
try:
text = body.decode("utf-8", "replace")
except Exception:
text = ""
for href in HREF_RE.findall(text):
absu = urljoin(url, href).split("#")[0]
if not absu.startswith("http"):
continue
all_links.add(absu)
if (
_same_domain(absu)
and _is_html(absu)
and not _no_recurse(absu)
and absu not in visited
and depth + 1 <= max_depth
):
queue.append((absu, depth + 1))
time.sleep(delay)
(OUT_DIR / "_all_links.txt").write_text("\n".join(sorted(all_links)))
(OUT_DIR / "_visited.txt").write_text("\n".join(sorted(visited)))
print(f"\nFetched {fetched} pages, {len(visited)} visited, {len(all_links)} total links discovered.")
print(f"Mirror in {OUT_DIR}")
if __name__ == "__main__":
ap = argparse.ArgumentParser()
ap.add_argument("--max-depth", type=int, default=10)
ap.add_argument("--max-pages", type=int, default=250)
ap.add_argument("--delay", type=float, default=0.4)
args = ap.parse_args()
crawl(args.max_depth, args.max_pages, args.delay)