metalfrom.eu/apps/crawler/src/scraper_band.py
Nicolas Fryder ab2fee2f43 fix(crawler): supprimer chemin Windows du docstring (SyntaxError \U)
Co-Authored-By: Claude Sonnet 4.6 <noreply@anthropic.com>
2026-06-27 15:17:39 +02:00

97 lines
3.1 KiB
Python

"""
Parser d'une page individuelle de band Metal Archives.
"""
import hashlib
import re
from typing import Any, Dict, List, Optional
from bs4 import BeautifulSoup
def parse_band_page(html: str) -> Dict[str, Any]:
soup = BeautifulSoup(html, "lxml")
out: Dict[str, Any] = {}
title = soup.find("title")
out["title"] = _text(title)
h1 = (
soup.find("h1", class_=re.compile(r"band_name", re.I))
or soup.find("h1")
)
out["name"] = _text(h1)
# Bloc #band_stats : dt/dd pairs
stats = soup.find("div", id="band_stats") or soup.find("div", id="band_info") or soup
kv: Dict[str, str] = {}
for dl in stats.find_all("dl"):
for dt in dl.find_all("dt"):
dd = dt.find_next_sibling("dd")
k = _text(dt).rstrip(":")
v = _text(dd)
if k and v:
kv[k] = v
out["info"] = kv
out["status"] = _pick(kv, "Status")
out["formed_in"] = _pick(kv, "Formed in", "Formed")
out["genre"] = _pick(kv, "Genre")
out["themes"] = _pick(kv, "Lyrical themes", "Lyrical Themes")
out["label"] = _pick(kv, "Current label", "Label")
out["years_active"] = _pick(kv, "Years active")
# Dates "Added on" / "Last modified" (souvent dans #auditTrail ou en bas de page)
audit = soup.find(id="auditTrail") or soup.find("div", class_=re.compile(r"audit", re.I))
if audit:
trail_text = _text(audit)
m_added = re.search(r"Added on:\s*([^\n,]+)", trail_text, re.I)
m_modified = re.search(r"Last modified on:\s*([^\n,]+)", trail_text, re.I)
out["ma_created_at"] = m_added.group(1).strip() if m_added else None
out["ma_modified_at"] = m_modified.group(1).strip() if m_modified else None
# Membres
members: List[Dict] = []
for table in soup.find_all("table"):
cls = " ".join(table.get("class") or [])
if "lineupTable" not in cls:
continue
section_el = table.find_previous(["h2", "h3"])
section = _text(section_el) if section_el else None
headers = [_text(th) for th in table.find_all("th")]
for tr in table.find_all("tr")[1:]:
tds = tr.find_all(["td", "th"])
if not tds:
continue
cells = [_text(td) for td in tds]
row = dict(zip(headers, cells)) if headers and len(headers) == len(cells) else {"cells": cells}
if section:
row["section"] = section
members.append(row)
if members:
out["members"] = members
# Liens externes
links_div = soup.find("div", id="band_links")
if links_div:
out["links"] = [
{"text": _text(a), "url": a["href"]}
for a in links_div.find_all("a", href=True)
]
return out
def page_hash(html: str) -> str:
"""MD5 du HTML pour détecter les changements."""
return hashlib.md5(html.encode()).hexdigest()
def _text(el) -> str:
return (el.get_text(" ", strip=True) if el else "").strip()
def _pick(kv: Dict, *keys: str) -> Optional[str]:
for k in keys:
if k in kv:
return kv[k]
return None