#!/usr/bin/env python3 """Scrape sumo results from sumodb.sumogames.de and generate the Suma website.""" import re import json import urllib.request from datetime import datetime, timezone, timedelta from html import unescape JST = timezone(timedelta(hours=9)) PARIS = timezone(timedelta(hours=2)) # CEST (summer) BASHO_CODES = { "01": ("Hatsu", "初場所", "Janvier"), "03": ("Haru", "春場所", "Mars"), "05": ("Natsu", "夏場所", "Mai"), "07": ("Nagoya", "名古屋場所", "Juillet"), "09": ("Aki", "秋場所", "Septembre"), "11": ("Kyushu", "九州場所", "Novembre"), } # ─── Scraping ─── def fetch_results(basho, day): """Fetch and parse results for a given basho code (e.g. 202607) and day (1-15).""" url = f"https://sumodb.sumogames.de/Results.aspx?b={basho}&d={day}" req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (SumaBot)"}) with urllib.request.urlopen(req, timeout=30) as resp: html = resp.read().decode("utf-8", errors="replace") # Header info h1 = re.search(r"

([^<]+)

", html) h2 = re.search(r"

([^<]+)

", html) title = h1.group(1).strip() if h1 else f"Basho {basho} Day {day}" date_str = h2.group(1).strip() if h2 else "" # Parse tables by division tables = re.findall(r'', html, re.DOTALL) divisions = {} division_names = ["Makuuchi", "Juryo", "Makushita", "Sandanme", "Jonidan", "Jonokuchi"] for i, table in enumerate(tables): if i >= len(division_names): break div_name = division_names[i] matches = re.findall( r'' r'' r'', table, re.DOTALL) parsed = [] for m in matches: east_result_img = m[0] east_cell = m[1] kim_cell = m[2] west_cell = m[3] west_result_img = m[4] # Parse wrestler info east = parse_wrestler(east_cell) west = parse_wrestler(west_cell) technique = parse_technique(kim_cell) # Determine winner east_won = "shiro" in east_result_img west_won = "shiro" in west_result_img east_lost = "kuro" in east_result_img west_lost = "kuro" in west_result_img fusen = "fusen" in east_result_img or "fusen" in west_result_img if east_won: winner = "east" elif west_won: winner = "west" elif fusen: # Check who got the fusensho (win by default) winner = "east" if "fusensho" in east_result_img else "west" else: winner = None parsed.append({ "east": east, "west": west, "technique": technique, "winner": winner, "fusen": fusen, }) divisions[div_name] = parsed return { "title": title, "date_str": date_str, "basho": basho, "day": day, "divisions": divisions, } def parse_wrestler(cell): """Extract wrestler name, rank, record, and shikona from a cell.""" rank = "" rank_m = re.search(r'(.*?)', cell) if rank_m: rank = unescape(rank_m.group(1).strip()) # Japanese shikona from title shikona_ja = "" title_m = re.search(r"title='([^']+)'", cell) if title_m: parts = title_m.group(1).split(",") if parts: shikona_ja = parts[0].strip() # English shikona from link text name = "" link_m = re.search(r"href='Rikishi\.aspx\?r=\d+'>([^<]+)", cell) if link_m: name = link_m.group(1).strip() # Record record = "" rec_m = re.search(r'([^<]+)', cell) if rec_m: record = rec_m.group(1).strip() return {"rank": rank, "name": name, "shikona_ja": shikona_ja, "record": record} def parse_technique(cell): """Extract winning technique from kimari cell.""" # Remove HTML tags text = re.sub(r"<[^>]+>", "", cell) text = unescape(text).strip() # The technique is on its own line (after the first
) lines = [l.strip() for l in text.split("\n") if l.strip()] # Filter out H2H records (contain digits) tech = "" for line in lines: if not re.search(r"\d", line) and line not in ("", " "): tech = line break return tech # ─── HTML generation ─── def generate_html(data): now_utc = datetime.now(timezone.utc) now_jst = now_utc.astimezone(JST) now_paris = now_utc.astimezone(PARIS) basho_code = data["basho"] month_code = basho_code[4:6] year = basho_code[:4] basho_en, basho_ja, basho_fr = BASHO_CODES.get(month_code, ("Unknown", "不明", "Inconnu")) # Build division sections divisions_html = [] for div_name in ["Makuuchi", "Juryo", "Makushita", "Sandanme", "Jonidan", "Jonokuchi"]: matches = data["divisions"].get(div_name, []) if not matches: continue matches_html = [] for m in matches: east = m["east"] west = m["west"] winner = m["winner"] # Winner/loser styling east_class = "winner" if winner == "east" else ("loser" if winner == "west" else "") west_class = "winner" if winner == "west" else ("loser" if winner == "east" else "") tech_display = m["technique"] if m["technique"] and not m["fusen"] else "" if m["fusen"]: tech_display = "不戦勝" if winner == "east" else "不戦勝" match_html = f"""
{east['rank']} {east['name']} {east['shikona_ja']} {east['record']}
{tech_display}
{west['rank']} {west['name']} {west['shikona_ja']} {west['record']}
""" matches_html.append(match_html) matches_str = "\n".join(matches_html) div_html = f"""

{div_name}

{matches_str}
""" divisions_html.append(div_html) divisions_str = "\n".join(divisions_html) html = f"""Suma — 相撲結果

🤼 Suma

{year} {basho_fr} · {basho_ja}
Jour {data['day']}
{data['date_str']}
Mise à jour automatique toutes les 10 minutes
{divisions_str}
""" return html # ─── Auto-detect current basho and day ─── def detect_current_basho_day(): """Determine the current basho code and day based on the date.""" now_jst = datetime.now(JST) year = now_jst.year month = now_jst.month # Basho months: 1, 3, 5, 7, 9, 11 basho_months = {1: "01", 3: "03", 5: "05", 7: "07", 9: "09", 11: "11"} # Find the most recent basho basho_code = None for m in [1, 3, 5, 7, 9, 11]: if month >= m: basho_code = f"{year}{basho_months[m]}" elif month < 3: basho_code = f"{year-1}11" break # If no basho yet this year, use last November if basho_code is None: basho_code = f"{year-1}11" # Basho runs for 15 days starting on the 2nd Sunday of the month # For simplicity, try days 1-15 and use the latest one with data # First try day 15, then go down for day in range(15, 0, -1): try: data = fetch_results(basho_code, day) if data["divisions"]: return basho_code, day, data except Exception: continue return basho_code, 1, None if __name__ == "__main__": import sys if len(sys.argv) >= 3: basho = sys.argv[1] day = int(sys.argv[2]) else: basho, day, pre_data = detect_current_basho_day() if pre_data: data = pre_data else: data = fetch_results(basho, day) if len(sys.argv) < 3: # Auto mode if 'data' not in dir(): data = fetch_results(basho, day) else: data = fetch_results(basho, day) html = generate_html(data) output_path = "/home/jerome/hermes-sites/suma.hidrago.click/index.html" with open(output_path, "w", encoding="utf-8") as f: f.write(html) print(f"Suma page generated: basho={data['basho']} day={data['day']} -> {output_path}")
(.*?)(.*?)(.*?)(.*?)(.*?)