Commit tune maps and research so tunes are reachable from the phone
Removes the *.hex/maps_cache gitignore rule (explicit user call, reversing the earlier no-redistribution stance) so the official TuneECU catalogue maps, derived SAI/O2-delete composites, and the checksum/composition tooling are actually available to pull up on a phone browser when using the real TuneECU app. Also folds in tonight's KWP2000 fixes (TesterPresent keep-alive, connect-failure cleanup, slow-init StartCommunication fix) and the accumulated research docs. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01FP2GaxS9HkUdL5sLBnjKje
This commit is contained in:
148
tunie/research/tuneecu_crawl.py
Normal file
148
tunie/research/tuneecu_crawl.py
Normal file
@@ -0,0 +1,148 @@
|
||||
"""Polite same-domain BFS crawler for tuneecu.net -- mirrors the docs site
|
||||
locally so we can grep/read full raw pages instead of relying on one-page-
|
||||
at-a-time AI-summarized fetches (which have already been shown to drop
|
||||
detail, e.g. the glossary needing a second full pass).
|
||||
|
||||
No robots.txt on tuneecu.net (404s). Still polite: single-threaded, delay
|
||||
between requests, capped page count, HTML only (binaries/zips are recorded
|
||||
as leaf links but not fetched/recursed into -- there's no reason to mirror
|
||||
the whole map-download tree, just discover what's there).
|
||||
|
||||
Usage:
|
||||
python3 tuneecu_crawl.py # default seeds, depth 10
|
||||
python3 tuneecu_crawl.py --max-pages 300
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import argparse
|
||||
import re
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from collections import deque
|
||||
from pathlib import Path
|
||||
from urllib.parse import urljoin, urlparse
|
||||
|
||||
DOMAIN = "tuneecu.net"
|
||||
OUT_DIR = Path(__file__).parent / "tuneecu_site_mirror"
|
||||
SEEDS = [
|
||||
"https://tuneecu.net/TuneECU_En/index.html",
|
||||
"https://tuneecu.net/TuneECU_En/links.html",
|
||||
"https://tuneecu.net/TuneECU_En/glossary.html",
|
||||
"https://tuneecu.net/TuneECU_En/mapedit.html",
|
||||
"https://tuneecu.net/Map_Database.html",
|
||||
]
|
||||
HREF_RE = re.compile(r'href\s*=\s*["\']([^"\']+)["\']', re.I)
|
||||
HTML_EXTS = (".html", ".htm", "") # "" covers dir-style URLs, treated as html
|
||||
|
||||
|
||||
NON_HTML_EXTS = (
|
||||
".zip", ".hex", ".pdf", ".mp4", ".jpg", ".jpeg", ".png", ".gif",
|
||||
".dat", ".md5", ".exe", ".bin", ".rar", ".7z", ".doc", ".docx",
|
||||
)
|
||||
# Leaf file tree -- thousands of per-model tune files/checksums, not docs.
|
||||
# Record links into it (so we know what's there) but never recurse.
|
||||
NO_RECURSE_PREFIXES = ("https://tuneecu.net/tunes_in_hex_and_dat/",)
|
||||
|
||||
|
||||
def _is_html(url: str) -> bool:
|
||||
path = urlparse(url).path.lower()
|
||||
if any(path.endswith(ext) for ext in NON_HTML_EXTS):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _no_recurse(url: str) -> bool:
|
||||
return url.lower().startswith(NO_RECURSE_PREFIXES)
|
||||
|
||||
|
||||
def _same_domain(url: str) -> bool:
|
||||
return urlparse(url).netloc.lower().endswith(DOMAIN)
|
||||
|
||||
|
||||
def _local_path(url: str) -> Path:
|
||||
p = urlparse(url)
|
||||
rel = (p.netloc + p.path).strip("/")
|
||||
if not rel or rel.endswith("/"):
|
||||
rel += "index.html"
|
||||
return OUT_DIR / rel
|
||||
|
||||
|
||||
def crawl(max_depth: int, max_pages: int, delay: float) -> None:
|
||||
OUT_DIR.mkdir(exist_ok=True)
|
||||
visited: set[str] = set()
|
||||
prior = OUT_DIR / "_visited.txt"
|
||||
if prior.exists():
|
||||
visited |= {ln.strip() for ln in prior.read_text().splitlines() if ln.strip()}
|
||||
print(f"preloaded {len(visited)} already-visited URLs, will skip re-fetching them")
|
||||
all_links: set[str] = set() # every link seen, html or not
|
||||
queue: deque[tuple[str, int]] = deque((s, 0) for s in SEEDS)
|
||||
prior_links = OUT_DIR / "_all_links.txt"
|
||||
if prior_links.exists():
|
||||
all_links |= {ln.strip() for ln in prior_links.read_text().splitlines() if ln.strip()}
|
||||
for u in sorted(all_links):
|
||||
if _same_domain(u) and _is_html(u) and not _no_recurse(u) and u not in visited:
|
||||
queue.append((u, 1))
|
||||
print(f"seeded queue with {len(queue)} unfetched html links from prior run")
|
||||
fetched = 0
|
||||
|
||||
while queue and fetched < max_pages:
|
||||
url, depth = queue.popleft()
|
||||
if url in visited or depth > max_depth:
|
||||
continue
|
||||
visited.add(url)
|
||||
|
||||
req = urllib.request.Request(url, headers={"User-Agent": "tunie-research-crawler/1.0"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=20) as r:
|
||||
body = r.read()
|
||||
except (urllib.error.URLError, urllib.error.HTTPError, TimeoutError) as exc:
|
||||
print(f" skip {url}: {exc}")
|
||||
continue
|
||||
|
||||
fetched += 1
|
||||
dest = _local_path(url)
|
||||
try:
|
||||
if dest.is_dir():
|
||||
dest = dest / "index.html"
|
||||
dest.parent.mkdir(parents=True, exist_ok=True)
|
||||
dest.write_bytes(body)
|
||||
except (IsADirectoryError, NotADirectoryError, FileNotFoundError) as exc:
|
||||
print(f" write skip {url}: {exc}")
|
||||
print(f"[{fetched}/{max_pages}] depth={depth} {url} ({len(body)}B)")
|
||||
|
||||
try:
|
||||
text = body.decode("utf-8", "replace")
|
||||
except Exception:
|
||||
text = ""
|
||||
|
||||
for href in HREF_RE.findall(text):
|
||||
absu = urljoin(url, href).split("#")[0]
|
||||
if not absu.startswith("http"):
|
||||
continue
|
||||
all_links.add(absu)
|
||||
if (
|
||||
_same_domain(absu)
|
||||
and _is_html(absu)
|
||||
and not _no_recurse(absu)
|
||||
and absu not in visited
|
||||
and depth + 1 <= max_depth
|
||||
):
|
||||
queue.append((absu, depth + 1))
|
||||
|
||||
time.sleep(delay)
|
||||
|
||||
(OUT_DIR / "_all_links.txt").write_text("\n".join(sorted(all_links)))
|
||||
(OUT_DIR / "_visited.txt").write_text("\n".join(sorted(visited)))
|
||||
print(f"\nFetched {fetched} pages, {len(visited)} visited, {len(all_links)} total links discovered.")
|
||||
print(f"Mirror in {OUT_DIR}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--max-depth", type=int, default=10)
|
||||
ap.add_argument("--max-pages", type=int, default=250)
|
||||
ap.add_argument("--delay", type=float, default=0.4)
|
||||
args = ap.parse_args()
|
||||
crawl(args.max_depth, args.max_pages, args.delay)
|
||||
Reference in New Issue
Block a user