Files
robertoandClaude Sonnet 4.6 21ea551196 Initial commit
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-01 09:27:56 +02:00

374 lines
16 KiB
Python
Raw Permalink Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python3
"""
Mirror lapaella.net from the Wayback Machine using curl for all HTTP.
Primary: 2024-11-30. Falls back to 2024-10-14, then 2021-06-03.
"""
import os, re, sys, time, subprocess, shutil, urllib.parse
from pathlib import Path
# ---------- config ----------
PRIMARY_TS = "20241130065929"
FALLBACK_TS = ["20241014051856", "20210603053819"]
BASE_NEW = "https://lapaella.net"
BASE_OLD = "http://www.lapaella.net"
OUT_DIR = Path(__file__).parent / "site"
UA = "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36"
DELAY = 1.0 # seconds between page requests
ASSET_DELAY = 0.3
MISSING = []
downloaded = set()
asset_queue = [] # list of (wbm_url, orig_path)
# ---------- complete page list ----------
# (path, is_recipe)
PAGES = [
# Home
("/", False),
# Main paella recipes
("/receta-paella-valenciana/", True),
("/receta-paella-de-marisco/", True),
("/receta-paella-de-verduras/", True),
# Arroces típicos de Valencia
("/arroces-tipicos-de-valencia/", False),
("/arroces-tipicos-de-valencia/arroz-a-banda/", True),
("/arroces-tipicos-de-valencia/arroz-al-horno-2/", True),
("/arroces-tipicos-de-valencia/arroz-al-horno-de-anguilas-y-bajocas-2/", True),
("/arroces-tipicos-de-valencia/arroz-allipebrat-2/", True),
("/arroces-tipicos-de-valencia/arroz-amb-bledes-con-acelgas/", True),
("/arroces-tipicos-de-valencia/arroz-amb-fesols-i-naps-con-alubias-y-nabos/", True),
("/arroces-tipicos-de-valencia/arroz-con-pasas-y-garbanzos/", True),
("/arroces-tipicos-de-valencia/arroz-meloso-de-pato-setas-y-trufa-2/", True),
("/arroces-tipicos-de-valencia/arroz-meloso-semicaldoso-con-brocheta-de-calamarcitos-y-gambas-2/", True),
("/arroces-tipicos-de-valencia/arroz-negro/", True),
("/arroces-tipicos-de-valencia/fideua-de-marisco-2/", True),
("/arroces-tipicos-de-valencia/fideua-de-marisco/", True),
("/arroces-tipicos-de-valencia/fideua-negreta-2/", True),
("/arroces-tipicos-de-valencia/fideua-negreta/", True),
("/arroces-tipicos-de-valencia/paella-de-bogavante-2/", True),
("/arroces-tipicos-de-valencia/paella-de-langosta-2/", True),
("/arroces-tipicos-de-valencia/paella-de-marisco-2/", True),
("/arroces-tipicos-de-valencia/paella-de-pato-2/", True),
("/arroces-tipicos-de-valencia/paella-de-verduras-2/", True),
("/arroces-tipicos-de-valencia/paella-morena-2/", True),
# Otras recetas valencianas
("/otras-recetas-valencianas/", False),
("/otras-recetas-valencianas/all-i-oli/", True),
("/otras-recetas-valencianas/all-i-pebre/", True),
("/otras-recetas-valencianas/angulas-al-ajillo/", True),
("/otras-recetas-valencianas/clochinas-al-vapor/", True),
("/otras-recetas-valencianas/espardenya/", True),
("/otras-recetas-valencianas/lubina-a-la-sal/", True),
("/otras-recetas-valencianas/pato-al-pebre-picant/", True),
# Blog / new recipes (2024)
("/blog/", False),
("/arroz-de-frutos-secos-garbanzos-y-nisperos/", True),
("/clochina-o-mejillon/", True),
("/coca-valenciana/", True),
("/fideua-de-pato-boletus-y-puerro/", True),
("/paella-de-coliflor-y-bacalao/", True),
("/talibanes-de-la-paella/", False),
# Informational / technique
("/aspectos-tecnicos-sobre-el-arroz/", False),
("/valencia-y-la-cultura-del-arroz-origen-y-evolucion/", False),
("/la-cocina-de-la-albufera/", False),
("/trucos-y-consejos/", False),
("/recipiente/", False),
("/ingredientes/", False),
("/recetas/", False),
("/fotos/", False),
("/lapaella-tv/", False),
("/enlaces/", False),
("/840-2/", False),
# Other languages
("/valencian-paella-introduction/", False),
("/paella-valenciana-recipe-english/", False),
("/paella-de-fruits-de-mer/", False),
("/paella-di-fruti-di-mari/", False),
("/la-paella-valenciana/", False),
("/la-paella-valenciana-it/", False),
("/la-paella-valencianaintroduzione/", False),
("/la-paella-il-recipiente-trucchi-e-consigli/", False),
("/receta-de-arroz-a-banda/", True),
("/receta-de-arroz-negro/", True),
("/receta-de-fideua", True),
("/%E8%8F%9C%E9%A5%AD-%E7%AE%80%E4%BD%93%E4%B8%AD%E6%96%87/", False),
# Site pages
("/nosotros/", False),
("/consultoria/", False),
("/cursos/", False),
("/contacto/", False),
]
# ---------- curl fetch ----------
import tempfile
def curl_fetch(url):
"""Fetch URL with curl into a temp file; return (bytes_or_None, content_type)."""
tmp = None
try:
tmp = tempfile.NamedTemporaryFile(delete=False, suffix=".tmp")
tmp.close()
result = subprocess.run(
["curl", "-s", "-L", "--max-time", "30",
"-A", UA,
"-w", "\n__HTTPSTATUS__%{http_code}\n__CTYPE__%{content_type}",
"-o", tmp.name,
url],
capture_output=True, text=True, timeout=45
)
# Parse status/ctype from stdout
out = result.stdout
status = 0
ctype = ""
for line in out.splitlines():
if line.startswith("__HTTPSTATUS__"):
try:
status = int(line.replace("__HTTPSTATUS__", "").strip())
except ValueError:
pass
elif line.startswith("__CTYPE__"):
ctype = line.replace("__CTYPE__", "").strip()
if status in (404, 403, 410):
return None, None
with open(tmp.name, "rb") as f:
data = f.read()
if len(data) < 200:
return None, None
return data, ctype
except Exception as e:
print(f" curl error: {e}")
return None, None
finally:
if tmp:
try:
os.unlink(tmp.name)
except OSError:
pass
# ---------- path helpers ----------
def build_wbm_url(orig_path, ts, suffix=""):
base = BASE_OLD if ts == "20210603053819" else BASE_NEW
return f"https://web.archive.org/web/{ts}{suffix}/{base}{orig_path}"
def orig_path_from_wbm(url):
m = re.match(
r'https?://web\.archive\.org/web/\d+(?:im_|cs_|js_|oe_|if_|mp_)?/'
r'https?://(?:www\.)?lapaella\.net(/[^"\'<>\s#]*)?',
url
)
return (m.group(1) or "/") if m else None
def local_path_for(orig_path):
decoded = urllib.parse.unquote(orig_path).lstrip("/")
if not decoded or decoded.endswith("/"):
return OUT_DIR / decoded / "index.html"
p = Path(decoded)
return OUT_DIR / p / "index.html" if not p.suffix else OUT_DIR / p
def local_rel(from_file, to_file):
return os.path.relpath(str(to_file), str(from_file.parent))
# ---------- URL rewriting ----------
def _rewrite_url(url, local_file):
if "web.archive.org" not in url:
return url
orig = orig_path_from_wbm(url)
if orig is None:
return "#"
# Queue for download
key = orig
if key not in downloaded and (url, orig) not in asset_queue:
asset_queue.append((url, orig))
return local_rel(local_file, local_path_for(orig))
def rewrite_html(html, local_file):
# Strip Wayback Machine UI elements
html = re.sub(
r'<!-- BEGIN WAYBACK TOOLBAR INSERT -->.*?<!-- END WAYBACK TOOLBAR INSERT -->',
'', html, flags=re.DOTALL)
html = re.sub(
r'<script[^>]*(?:archive\.org|wombat\.js|athena\.js|bundle-playback|ruffle)[^>]*>.*?</script>',
'', html, flags=re.DOTALL | re.IGNORECASE)
html = re.sub(
r'<script[^>]*(?:archive\.org|wombat\.js|athena\.js|bundle-playback|ruffle)[^>]*/?>',
'', html, flags=re.IGNORECASE)
html = re.sub(r'<link[^>]*(?:archive\.org|iconochive|banner-styles)[^>]*/?>',
'', html, flags=re.IGNORECASE)
html = re.sub(r'<script[^>]*>\s*__wm\b.*?</script>', '', html, flags=re.DOTALL)
html = re.sub(r'<script[^>]*>\s*window\.RufflePlayer.*?</script>', '', html, flags=re.DOTALL)
def _attr(m):
attr, q, url, q2 = m.group(1), m.group(2), m.group(3), m.group(4)
return f'{attr}={q}{_rewrite_url(url, local_file)}{q2}'
html = re.sub(r'((?:src|href|action|data-src)=)(["\'])([^"\']+)(["\'])', _attr, html)
def _css_url(m):
url = m.group(1).strip().strip("'\"")
return f"url('{_rewrite_url(url, local_file)}')"
html = re.sub(r'url\(([^)]+)\)', _css_url, html)
return html
def rewrite_css(css, local_file):
def _r(m):
url = m.group(1).strip().strip("'\"")
return f"url('{_rewrite_url(url, local_file)}')"
return re.sub(r'url\(([^)]+)\)', _r, css)
# ---------- download ----------
def fetch_with_fallback(orig_path, suffixes=("",)):
"""Try all timestamps × suffixes, return first success."""
for ts in [PRIMARY_TS] + FALLBACK_TS:
for sfx in suffixes:
url = build_wbm_url(orig_path, ts, sfx)
data, ctype = curl_fetch(url)
if data:
print(f" ✓ {ts[:8]} {orig_path[:70]}")
return data, ctype, url
return None, None, None
def download_asset(wbm_url_str, orig_path):
if orig_path in downloaded:
return
downloaded.add(orig_path)
local = local_path_for(orig_path)
if local.exists():
return
ext = Path(orig_path).suffix.lower()
if ext in (".jpg", ".jpeg", ".png", ".gif", ".svg", ".webp", ".ico", ".bmp"):
suffixes = ("im_", "")
elif ext == ".css":
suffixes = ("cs_", "")
elif ext == ".js":
suffixes = ("js_", "")
else:
suffixes = ("", "im_", "cs_", "js_")
# Try the exact WBM URL first, then fallback
data, ctype = curl_fetch(wbm_url_str)
if not data:
data, ctype, _ = fetch_with_fallback(orig_path, suffixes)
if not data:
MISSING.append(("asset", orig_path, wbm_url_str))
return
local.parent.mkdir(parents=True, exist_ok=True)
is_css = (ctype and "css" in ctype) or ext == ".css"
if is_css:
text = data.decode("utf-8", errors="replace")
for sub in re.findall(r'url\(([^)]+)\)', text):
sub = sub.strip().strip("'\"")
if "web.archive.org" in sub:
sub_orig = orig_path_from_wbm(sub)
if sub_orig and sub_orig not in downloaded:
asset_queue.append((sub, sub_orig))
local.write_text(rewrite_css(text, local), encoding="utf-8")
else:
local.write_bytes(data)
time.sleep(ASSET_DELAY)
def download_page(orig_path, is_recipe):
if orig_path in downloaded:
return
downloaded.add(orig_path)
# Skip if already saved from a previous run
if local_path_for(orig_path).exists():
print(f" skip (exists) {orig_path}")
return
label = "RECIPE" if is_recipe else "PAGE "
print(f"\n{label} {orig_path}")
data, ctype, url_used = fetch_with_fallback(orig_path, ("",))
if not data:
print(f" ✗ not found in any snapshot")
MISSING.append(("page", orig_path, "—"))
return
html = data.decode("utf-8", errors="replace")
local = local_path_for(orig_path)
local.parent.mkdir(parents=True, exist_ok=True)
# Collect all WBM asset URLs before rewriting
for url in re.findall(r'(?:src|href|data-src)=["\']([^"\']+)["\']', html):
if "web.archive.org" in url:
sub = orig_path_from_wbm(url)
if sub and sub not in downloaded:
asset_queue.append((url, sub))
for url in re.findall(r'url\(([^)]+)\)', html):
url = url.strip().strip("'\"")
if "web.archive.org" in url:
sub = orig_path_from_wbm(url)
if sub and sub not in downloaded:
asset_queue.append((url, sub))
html = rewrite_html(html, local)
local.write_text(html, encoding="utf-8")
print(f" saved {len(data)//1024}KB → {local.relative_to(OUT_DIR.parent)}")
while asset_queue:
a_url, a_orig = asset_queue.pop(0)
download_asset(a_url, a_orig)
time.sleep(DELAY)
# ---------- main ----------
if __name__ == "__main__":
OUT_DIR.mkdir(parents=True, exist_ok=True)
print(f"Output : {OUT_DIR}")
print(f"Pages : {len(PAGES)} ({sum(1 for _,r in PAGES if r)} recipes)\n")
for path, is_recipe in sorted(PAGES, key=lambda x: (0 if x[1] else 1, x[0])):
download_page(path, is_recipe)
# ── summary ──────────────────────────────────────────────────────────────
print("\n" + "="*65)
print(f"Resources downloaded : {len(downloaded)}")
print(f"Missing : {len(MISSING)}")
for k, p, u in MISSING:
print(f" [{k}] {p}")
# ── inventory ─────────────────────────────────────────────────────────────
inv = OUT_DIR.parent / "inventory.md"
rf = [(p,r) for p,r in PAGES if r and local_path_for(p).exists()]
rm = [(p,r) for p,r in PAGES if r and not local_path_for(p).exists()]
pf = [(p,r) for p,r in PAGES if not r and local_path_for(p).exists()]
pm = [(p,r) for p,r in PAGES if not r and not local_path_for(p).exists()]
with open(inv, "w") as f:
f.write("# lapaella.net local mirror\n\n")
f.write(f"Primary : `{PRIMARY_TS}` (2024-11-30)\n")
f.write(f"Fallback : `{FALLBACK_TS[0]}` · `{FALLBACK_TS[1]}`\n\n")
f.write(f"## Recipes saved ({len(rf)})\n\n")
for p,_ in rf: f.write(f"- ✓ `{p}`\n")
if rm:
f.write(f"\n## Recipes MISSING ({len(rm)})\n\n")
for p,_ in rm: f.write(f"- ✗ `{p}`\n")
f.write(f"\n## Other pages saved ({len(pf)})\n\n")
for p,_ in pf: f.write(f"- ✓ `{p}`\n")
if pm:
f.write(f"\n## Other pages MISSING ({len(pm)})\n\n")
for p,_ in pm: f.write(f"- ✗ `{p}`\n")
if MISSING:
f.write(f"\n## Failed assets ({len(MISSING)})\n\n")
for k,p,u in MISSING: f.write(f"- [{k}] `{p}`\n")
print(f"\nInventory → {inv}")
print(f"Open → open {OUT_DIR}/index.html")