#!/usr/bin/env python3 """ Targeted download of pages that require specific older timestamps or www. subdomain. Run AFTER mirror.py to fill gaps. """ import os, re, subprocess, tempfile, urllib.parse, time from pathlib import Path OUT_DIR = Path(__file__).parent / "site" UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36" # (orig_path, wbm_full_url) — timestamps from CDX API query TARGETS = [ # Recipes with old/alt timestamps ("/arroces-tipicos-de-valencia/arroz-al-horno-2/", "https://web.archive.org/web/20120604015718/http://lapaella.net/arroces-tipicos-de-valencia/arroz-al-horno-2/"), ("/arroces-tipicos-de-valencia/arroz-al-horno-2/", # try www too "https://web.archive.org/web/20120604015718/http://www.lapaella.net/arroces-tipicos-de-valencia/arroz-al-horno-2/"), ("/arroces-tipicos-de-valencia/fideua-de-marisco/", "https://web.archive.org/web/20160505061552/https://lapaella.net/arroces-tipicos-de-valencia/fideua-de-marisco/"), ("/arroces-tipicos-de-valencia/arroz-con-pasas-y-garbanzos/", "https://web.archive.org/web/20210603053819/http://www.lapaella.net/arroces-tipicos-de-valencia/arroz-con-pasas-y-garbanzos/"), ("/coca-valenciana/", "https://web.archive.org/web/20221005043236/https://lapaella.net/coca-valenciana/"), ("/coca-valenciana/", "https://web.archive.org/web/20251013020851/https://lapaella.net/coca-valenciana/"), ("/otras-recetas-valencianas/all-i-oli/", "https://web.archive.org/web/20120529140656/http://www.lapaella.net/otras-recetas-valencianas/all-i-oli/"), ("/otras-recetas-valencianas/angulas-al-ajillo/", "https://web.archive.org/web/20120807085623/http://lapaella.net/otras-recetas-valencianas/angulas-al-ajillo/"), ("/otras-recetas-valencianas/angulas-al-ajillo/", "https://web.archive.org/web/20131025062816/http://www.lapaella.net/otras-recetas-valencianas/angulas-al-ajillo/"), ("/otras-recetas-valencianas/lubina-a-la-sal/", "https://web.archive.org/web/20210603053819/http://www.lapaella.net/otras-recetas-valencianas/lubina-a-la-sal/"), ("/receta-paella-de-verduras/", "https://web.archive.org/web/20220811013510/https://lapaella.net/receta-paella-de-verduras/"), ("/receta-paella-de-verduras/", "https://web.archive.org/web/20251013021316/https://lapaella.net/receta-paella-de-verduras/"), # Info pages with older snapshots ("/trucos-y-consejos/", "https://web.archive.org/web/20120504013448/http://www.lapaella.net/trucos-y-consejos/"), ("/la-cocina-de-la-albufera/", "https://web.archive.org/web/20120507161310/http://lapaella.net/la-cocina-de-la-albufera/"), ("/la-cocina-de-la-albufera/", "https://web.archive.org/web/20120504203901/http://www.lapaella.net/la-cocina-de-la-albufera/"), ("/840-2/", "https://web.archive.org/web/20120507153857/http://lapaella.net/840-2/"), ("/840-2/", "https://web.archive.org/web/20120505110823/http://www.lapaella.net/840-2/"), ("/contacto/", "https://web.archive.org/web/20160526093659/https://lapaella.net/contacto/"), ("/valencian-paella-introduction/", "https://web.archive.org/web/20210603053819/http://www.lapaella.net/valencian-paella-introduction/"), ("/talibanes-de-la-paella/", "https://web.archive.org/web/20241014051856/https://lapaella.net/sin-categoria/talibanes-de-la-paella/"), ("/arroces-tipicos-de-valencia/fideua-negreta/", "https://web.archive.org/web/20160105060427/https://lapaella.net/arroces-tipicos-de-valencia/fideua-negreta/"), ("/otras-recetas-valencianas/", "https://web.archive.org/web/20210603053819/http://www.lapaella.net/otras-recetas-valencianas/"), ("/otras-recetas-valencianas/", "https://web.archive.org/web/20160405160122/https://lapaella.net/otras-recetas-valencianas/"), ] def local_path_for(orig_path): decoded = urllib.parse.unquote(orig_path).lstrip("/") if not decoded or decoded.endswith("/"): return OUT_DIR / decoded / "index.html" p = Path(decoded) return OUT_DIR / p / "index.html" if not p.suffix else OUT_DIR / p def local_rel(from_file, to_file): return os.path.relpath(str(to_file), str(from_file.parent)) def orig_path_from_wbm(url): m = re.match( r'https?://web\.archive\.org/web/\d+(?:im_|cs_|js_|oe_|if_|mp_)?/' r'https?://(?:www\.)?lapaella\.net(/[^"\'<>\s#]*)?', url ) return (m.group(1) or "/") if m else None def curl_fetch(url): tmp = None try: tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".tmp") tmp.close() result = subprocess.run( ["/usr/bin/curl", "-s", "-L", "--max-time", "30", "-A", UA, "-w", "\n__STATUS__%{http_code}", "-o", tmp.name, url], capture_output=True, text=True, timeout=45 ) status = 0 for line in result.stdout.splitlines(): if line.startswith("__STATUS__"): try: status = int(line.replace("__STATUS__", "").strip()) except: pass if status in (404, 403, 410): return None with open(tmp.name, "rb") as f: data = f.read() return data if len(data) > 500 else None except Exception as e: print(f" curl error: {e}") return None finally: if tmp: try: os.unlink(tmp.name) except: pass def _rewrite_url(url, local_file): if "web.archive.org" not in url: return url orig = orig_path_from_wbm(url) return local_rel(local_file, local_path_for(orig)) if orig else "#" def rewrite_html(html, local_file): html = re.sub(r'.*?', '', html, flags=re.DOTALL) html = re.sub(r']*(?:archive\.org|wombat|athena|bundle-playback|ruffle)[^>]*>.*?', '', html, flags=re.DOTALL|re.IGNORECASE) html = re.sub(r']*(?:archive\.org|wombat|athena|bundle-playback|ruffle)[^>]*/?>', '', html, flags=re.IGNORECASE) html = re.sub(r']*(?:archive\.org|iconochive|banner-styles)[^>]*/?>', '', html, flags=re.IGNORECASE) html = re.sub(r']*>\s*__wm\b.*?', '', html, flags=re.DOTALL) html = re.sub(r']*>\s*window\.RufflePlayer.*?', '', html, flags=re.DOTALL) def _attr(m): attr, q, url, q2 = m.group(1), m.group(2), m.group(3), m.group(4) return f'{attr}={q}{_rewrite_url(url, local_file)}{q2}' html = re.sub(r'((?:src|href|action|data-src)=)(["\'])([^"\']+)(["\'])', _attr, html) def _css_url(m): url = m.group(1).strip().strip("'\"") return f"url('{_rewrite_url(url, local_file)}')" html = re.sub(r'url\(([^)]+)\)', _css_url, html) return html if __name__ == "__main__": fetched = set() success = [] failed = [] for orig_path, wbm_url in TARGETS: local = local_path_for(orig_path) if local.exists(): print(f" SKIP (exists) {orig_path}") fetched.add(orig_path) continue if orig_path in fetched: continue print(f" TRY {orig_path}") print(f" {wbm_url[:90]}") data = curl_fetch(wbm_url) if not data: print(f" ✗ no data") failed.append(orig_path) time.sleep(0.5) continue html = data.decode("utf-8", errors="replace") local.parent.mkdir(parents=True, exist_ok=True) html = rewrite_html(html, local) local.write_text(html, encoding="utf-8") print(f" ✓ saved {len(data)//1024}KB → {local.relative_to(OUT_DIR.parent)}") fetched.add(orig_path) success.append(orig_path) time.sleep(1.0) print(f"\nFetched: {len(success)} Failed: {len(set(failed) - fetched)}") for p in sorted(set(failed) - fetched): print(f" ✗ {p}")