374 lines
16 KiB
Python
374 lines
16 KiB
Python
#!/usr/bin/env python3
|
||
"""
|
||
Mirror lapaella.net from the Wayback Machine using curl for all HTTP.
|
||
Primary: 2024-11-30. Falls back to 2024-10-14, then 2021-06-03.
|
||
"""
|
||
|
||
import os, re, sys, time, subprocess, shutil, urllib.parse
|
||
from pathlib import Path
|
||
|
||
# ---------- config ----------
|
||
PRIMARY_TS = "20241130065929"
|
||
FALLBACK_TS = ["20241014051856", "20210603053819"]
|
||
BASE_NEW = "https://lapaella.net"
|
||
BASE_OLD = "http://www.lapaella.net"
|
||
OUT_DIR = Path(__file__).parent / "site"
|
||
UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36"
|
||
DELAY = 1.0 # seconds between page requests
|
||
ASSET_DELAY = 0.3
|
||
MISSING = []
|
||
downloaded = set()
|
||
asset_queue = [] # list of (wbm_url, orig_path)
|
||
|
||
# ---------- complete page list ----------
|
||
# (path, is_recipe)
|
||
PAGES = [
|
||
# Home
|
||
("/", False),
|
||
|
||
# Main paella recipes
|
||
("/receta-paella-valenciana/", True),
|
||
("/receta-paella-de-marisco/", True),
|
||
("/receta-paella-de-verduras/", True),
|
||
|
||
# Arroces típicos de Valencia
|
||
("/arroces-tipicos-de-valencia/", False),
|
||
("/arroces-tipicos-de-valencia/arroz-a-banda/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-al-horno-2/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-al-horno-de-anguilas-y-bajocas-2/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-allipebrat-2/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-amb-bledes-con-acelgas/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-amb-fesols-i-naps-con-alubias-y-nabos/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-con-pasas-y-garbanzos/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-meloso-de-pato-setas-y-trufa-2/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-meloso-semicaldoso-con-brocheta-de-calamarcitos-y-gambas-2/", True),
|
||
("/arroces-tipicos-de-valencia/arroz-negro/", True),
|
||
("/arroces-tipicos-de-valencia/fideua-de-marisco-2/", True),
|
||
("/arroces-tipicos-de-valencia/fideua-de-marisco/", True),
|
||
("/arroces-tipicos-de-valencia/fideua-negreta-2/", True),
|
||
("/arroces-tipicos-de-valencia/fideua-negreta/", True),
|
||
("/arroces-tipicos-de-valencia/paella-de-bogavante-2/", True),
|
||
("/arroces-tipicos-de-valencia/paella-de-langosta-2/", True),
|
||
("/arroces-tipicos-de-valencia/paella-de-marisco-2/", True),
|
||
("/arroces-tipicos-de-valencia/paella-de-pato-2/", True),
|
||
("/arroces-tipicos-de-valencia/paella-de-verduras-2/", True),
|
||
("/arroces-tipicos-de-valencia/paella-morena-2/", True),
|
||
|
||
# Otras recetas valencianas
|
||
("/otras-recetas-valencianas/", False),
|
||
("/otras-recetas-valencianas/all-i-oli/", True),
|
||
("/otras-recetas-valencianas/all-i-pebre/", True),
|
||
("/otras-recetas-valencianas/angulas-al-ajillo/", True),
|
||
("/otras-recetas-valencianas/clochinas-al-vapor/", True),
|
||
("/otras-recetas-valencianas/espardenya/", True),
|
||
("/otras-recetas-valencianas/lubina-a-la-sal/", True),
|
||
("/otras-recetas-valencianas/pato-al-pebre-picant/", True),
|
||
|
||
# Blog / new recipes (2024)
|
||
("/blog/", False),
|
||
("/arroz-de-frutos-secos-garbanzos-y-nisperos/", True),
|
||
("/clochina-o-mejillon/", True),
|
||
("/coca-valenciana/", True),
|
||
("/fideua-de-pato-boletus-y-puerro/", True),
|
||
("/paella-de-coliflor-y-bacalao/", True),
|
||
("/talibanes-de-la-paella/", False),
|
||
|
||
# Informational / technique
|
||
("/aspectos-tecnicos-sobre-el-arroz/", False),
|
||
("/valencia-y-la-cultura-del-arroz-origen-y-evolucion/", False),
|
||
("/la-cocina-de-la-albufera/", False),
|
||
("/trucos-y-consejos/", False),
|
||
("/recipiente/", False),
|
||
("/ingredientes/", False),
|
||
("/recetas/", False),
|
||
("/fotos/", False),
|
||
("/lapaella-tv/", False),
|
||
("/enlaces/", False),
|
||
("/840-2/", False),
|
||
|
||
# Other languages
|
||
("/valencian-paella-introduction/", False),
|
||
("/paella-valenciana-recipe-english/", False),
|
||
("/paella-de-fruits-de-mer/", False),
|
||
("/paella-di-fruti-di-mari/", False),
|
||
("/la-paella-valenciana/", False),
|
||
("/la-paella-valenciana-it/", False),
|
||
("/la-paella-valencianaintroduzione/", False),
|
||
("/la-paella-il-recipiente-trucchi-e-consigli/", False),
|
||
("/receta-de-arroz-a-banda/", True),
|
||
("/receta-de-arroz-negro/", True),
|
||
("/receta-de-fideua", True),
|
||
("/%E8%8F%9C%E9%A5%AD-%E7%AE%80%E4%BD%93%E4%B8%AD%E6%96%87/", False),
|
||
|
||
# Site pages
|
||
("/nosotros/", False),
|
||
("/consultoria/", False),
|
||
("/cursos/", False),
|
||
("/contacto/", False),
|
||
]
|
||
|
||
# ---------- curl fetch ----------
|
||
|
||
import tempfile
|
||
|
||
def curl_fetch(url):
|
||
"""Fetch URL with curl into a temp file; return (bytes_or_None, content_type)."""
|
||
tmp = None
|
||
try:
|
||
tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".tmp")
|
||
tmp.close()
|
||
result = subprocess.run(
|
||
["curl", "-s", "-L", "--max-time", "30",
|
||
"-A", UA,
|
||
"-w", "\n__HTTPSTATUS__%{http_code}\n__CTYPE__%{content_type}",
|
||
"-o", tmp.name,
|
||
url],
|
||
capture_output=True, text=True, timeout=45
|
||
)
|
||
# Parse status/ctype from stdout
|
||
out = result.stdout
|
||
status = 0
|
||
ctype = ""
|
||
for line in out.splitlines():
|
||
if line.startswith("__HTTPSTATUS__"):
|
||
try:
|
||
status = int(line.replace("__HTTPSTATUS__", "").strip())
|
||
except ValueError:
|
||
pass
|
||
elif line.startswith("__CTYPE__"):
|
||
ctype = line.replace("__CTYPE__", "").strip()
|
||
if status in (404, 403, 410):
|
||
return None, None
|
||
with open(tmp.name, "rb") as f:
|
||
data = f.read()
|
||
if len(data) < 200:
|
||
return None, None
|
||
return data, ctype
|
||
except Exception as e:
|
||
print(f" curl error: {e}")
|
||
return None, None
|
||
finally:
|
||
if tmp:
|
||
try:
|
||
os.unlink(tmp.name)
|
||
except OSError:
|
||
pass
|
||
|
||
# ---------- path helpers ----------
|
||
|
||
def build_wbm_url(orig_path, ts, suffix=""):
|
||
base = BASE_OLD if ts == "20210603053819" else BASE_NEW
|
||
return f"https://web.archive.org/web/{ts}{suffix}/{base}{orig_path}"
|
||
|
||
def orig_path_from_wbm(url):
|
||
m = re.match(
|
||
r'https?://web\.archive\.org/web/\d+(?:im_|cs_|js_|oe_|if_|mp_)?/'
|
||
r'https?://(?:www\.)?lapaella\.net(/[^"\'<>\s#]*)?',
|
||
url
|
||
)
|
||
return (m.group(1) or "/") if m else None
|
||
|
||
def local_path_for(orig_path):
|
||
decoded = urllib.parse.unquote(orig_path).lstrip("/")
|
||
if not decoded or decoded.endswith("/"):
|
||
return OUT_DIR / decoded / "index.html"
|
||
p = Path(decoded)
|
||
return OUT_DIR / p / "index.html" if not p.suffix else OUT_DIR / p
|
||
|
||
def local_rel(from_file, to_file):
|
||
return os.path.relpath(str(to_file), str(from_file.parent))
|
||
|
||
# ---------- URL rewriting ----------
|
||
|
||
def _rewrite_url(url, local_file):
|
||
if "web.archive.org" not in url:
|
||
return url
|
||
orig = orig_path_from_wbm(url)
|
||
if orig is None:
|
||
return "#"
|
||
# Queue for download
|
||
key = orig
|
||
if key not in downloaded and (url, orig) not in asset_queue:
|
||
asset_queue.append((url, orig))
|
||
return local_rel(local_file, local_path_for(orig))
|
||
|
||
def rewrite_html(html, local_file):
|
||
# Strip Wayback Machine UI elements
|
||
html = re.sub(
|
||
r'<!-- BEGIN WAYBACK TOOLBAR INSERT -->.*?<!-- END WAYBACK TOOLBAR INSERT -->',
|
||
'', html, flags=re.DOTALL)
|
||
html = re.sub(
|
||
r'<script[^>]*(?:archive\.org|wombat\.js|athena\.js|bundle-playback|ruffle)[^>]*>.*?</script>',
|
||
'', html, flags=re.DOTALL | re.IGNORECASE)
|
||
html = re.sub(
|
||
r'<script[^>]*(?:archive\.org|wombat\.js|athena\.js|bundle-playback|ruffle)[^>]*/?>',
|
||
'', html, flags=re.IGNORECASE)
|
||
html = re.sub(r'<link[^>]*(?:archive\.org|iconochive|banner-styles)[^>]*/?>',
|
||
'', html, flags=re.IGNORECASE)
|
||
html = re.sub(r'<script[^>]*>\s*__wm\b.*?</script>', '', html, flags=re.DOTALL)
|
||
html = re.sub(r'<script[^>]*>\s*window\.RufflePlayer.*?</script>', '', html, flags=re.DOTALL)
|
||
|
||
def _attr(m):
|
||
attr, q, url, q2 = m.group(1), m.group(2), m.group(3), m.group(4)
|
||
return f'{attr}={q}{_rewrite_url(url, local_file)}{q2}'
|
||
|
||
html = re.sub(r'((?:src|href|action|data-src)=)(["\'])([^"\']+)(["\'])', _attr, html)
|
||
|
||
def _css_url(m):
|
||
url = m.group(1).strip().strip("'\"")
|
||
return f"url('{_rewrite_url(url, local_file)}')"
|
||
html = re.sub(r'url\(([^)]+)\)', _css_url, html)
|
||
return html
|
||
|
||
def rewrite_css(css, local_file):
|
||
def _r(m):
|
||
url = m.group(1).strip().strip("'\"")
|
||
return f"url('{_rewrite_url(url, local_file)}')"
|
||
return re.sub(r'url\(([^)]+)\)', _r, css)
|
||
|
||
# ---------- download ----------
|
||
|
||
def fetch_with_fallback(orig_path, suffixes=("",)):
|
||
"""Try all timestamps × suffixes, return first success."""
|
||
for ts in [PRIMARY_TS] + FALLBACK_TS:
|
||
for sfx in suffixes:
|
||
url = build_wbm_url(orig_path, ts, sfx)
|
||
data, ctype = curl_fetch(url)
|
||
if data:
|
||
print(f" ✓ {ts[:8]} {orig_path[:70]}")
|
||
return data, ctype, url
|
||
return None, None, None
|
||
|
||
def download_asset(wbm_url_str, orig_path):
|
||
if orig_path in downloaded:
|
||
return
|
||
downloaded.add(orig_path)
|
||
|
||
local = local_path_for(orig_path)
|
||
if local.exists():
|
||
return
|
||
|
||
ext = Path(orig_path).suffix.lower()
|
||
if ext in (".jpg", ".jpeg", ".png", ".gif", ".svg", ".webp", ".ico", ".bmp"):
|
||
suffixes = ("im_", "")
|
||
elif ext == ".css":
|
||
suffixes = ("cs_", "")
|
||
elif ext == ".js":
|
||
suffixes = ("js_", "")
|
||
else:
|
||
suffixes = ("", "im_", "cs_", "js_")
|
||
|
||
# Try the exact WBM URL first, then fallback
|
||
data, ctype = curl_fetch(wbm_url_str)
|
||
if not data:
|
||
data, ctype, _ = fetch_with_fallback(orig_path, suffixes)
|
||
|
||
if not data:
|
||
MISSING.append(("asset", orig_path, wbm_url_str))
|
||
return
|
||
|
||
local.parent.mkdir(parents=True, exist_ok=True)
|
||
is_css = (ctype and "css" in ctype) or ext == ".css"
|
||
if is_css:
|
||
text = data.decode("utf-8", errors="replace")
|
||
for sub in re.findall(r'url\(([^)]+)\)', text):
|
||
sub = sub.strip().strip("'\"")
|
||
if "web.archive.org" in sub:
|
||
sub_orig = orig_path_from_wbm(sub)
|
||
if sub_orig and sub_orig not in downloaded:
|
||
asset_queue.append((sub, sub_orig))
|
||
local.write_text(rewrite_css(text, local), encoding="utf-8")
|
||
else:
|
||
local.write_bytes(data)
|
||
time.sleep(ASSET_DELAY)
|
||
|
||
def download_page(orig_path, is_recipe):
|
||
if orig_path in downloaded:
|
||
return
|
||
downloaded.add(orig_path)
|
||
# Skip if already saved from a previous run
|
||
if local_path_for(orig_path).exists():
|
||
print(f" skip (exists) {orig_path}")
|
||
return
|
||
|
||
label = "RECIPE" if is_recipe else "PAGE "
|
||
print(f"\n{label} {orig_path}")
|
||
|
||
data, ctype, url_used = fetch_with_fallback(orig_path, ("",))
|
||
if not data:
|
||
print(f" ✗ not found in any snapshot")
|
||
MISSING.append(("page", orig_path, "—"))
|
||
return
|
||
|
||
html = data.decode("utf-8", errors="replace")
|
||
local = local_path_for(orig_path)
|
||
local.parent.mkdir(parents=True, exist_ok=True)
|
||
|
||
# Collect all WBM asset URLs before rewriting
|
||
for url in re.findall(r'(?:src|href|data-src)=["\']([^"\']+)["\']', html):
|
||
if "web.archive.org" in url:
|
||
sub = orig_path_from_wbm(url)
|
||
if sub and sub not in downloaded:
|
||
asset_queue.append((url, sub))
|
||
for url in re.findall(r'url\(([^)]+)\)', html):
|
||
url = url.strip().strip("'\"")
|
||
if "web.archive.org" in url:
|
||
sub = orig_path_from_wbm(url)
|
||
if sub and sub not in downloaded:
|
||
asset_queue.append((url, sub))
|
||
|
||
html = rewrite_html(html, local)
|
||
local.write_text(html, encoding="utf-8")
|
||
print(f" saved {len(data)//1024}KB → {local.relative_to(OUT_DIR.parent)}")
|
||
|
||
while asset_queue:
|
||
a_url, a_orig = asset_queue.pop(0)
|
||
download_asset(a_url, a_orig)
|
||
|
||
time.sleep(DELAY)
|
||
|
||
# ---------- main ----------
|
||
|
||
if __name__ == "__main__":
|
||
OUT_DIR.mkdir(parents=True, exist_ok=True)
|
||
print(f"Output : {OUT_DIR}")
|
||
print(f"Pages : {len(PAGES)} ({sum(1 for _,r in PAGES if r)} recipes)\n")
|
||
|
||
for path, is_recipe in sorted(PAGES, key=lambda x: (0 if x[1] else 1, x[0])):
|
||
download_page(path, is_recipe)
|
||
|
||
# ── summary ──────────────────────────────────────────────────────────────
|
||
print("\n" + "="*65)
|
||
print(f"Resources downloaded : {len(downloaded)}")
|
||
print(f"Missing : {len(MISSING)}")
|
||
for k, p, u in MISSING:
|
||
print(f" [{k}] {p}")
|
||
|
||
# ── inventory ─────────────────────────────────────────────────────────────
|
||
inv = OUT_DIR.parent / "inventory.md"
|
||
rf = [(p,r) for p,r in PAGES if r and local_path_for(p).exists()]
|
||
rm = [(p,r) for p,r in PAGES if r and not local_path_for(p).exists()]
|
||
pf = [(p,r) for p,r in PAGES if not r and local_path_for(p).exists()]
|
||
pm = [(p,r) for p,r in PAGES if not r and not local_path_for(p).exists()]
|
||
|
||
with open(inv, "w") as f:
|
||
f.write("# lapaella.net – local mirror\n\n")
|
||
f.write(f"Primary : `{PRIMARY_TS}` (2024-11-30)\n")
|
||
f.write(f"Fallback : `{FALLBACK_TS[0]}` · `{FALLBACK_TS[1]}`\n\n")
|
||
f.write(f"## Recipes saved ({len(rf)})\n\n")
|
||
for p,_ in rf: f.write(f"- ✓ `{p}`\n")
|
||
if rm:
|
||
f.write(f"\n## Recipes MISSING ({len(rm)})\n\n")
|
||
for p,_ in rm: f.write(f"- ✗ `{p}`\n")
|
||
f.write(f"\n## Other pages saved ({len(pf)})\n\n")
|
||
for p,_ in pf: f.write(f"- ✓ `{p}`\n")
|
||
if pm:
|
||
f.write(f"\n## Other pages MISSING ({len(pm)})\n\n")
|
||
for p,_ in pm: f.write(f"- ✗ `{p}`\n")
|
||
if MISSING:
|
||
f.write(f"\n## Failed assets ({len(MISSING)})\n\n")
|
||
for k,p,u in MISSING: f.write(f"- [{k}] `{p}`\n")
|
||
|
||
print(f"\nInventory → {inv}")
|
||
print(f"Open → open {OUT_DIR}/index.html")
|