#!/usr/bin/env python3
"""Scrape sumo results from sumodb.sumogames.de and generate the Suma website."""
import re
import json
import urllib.request
from datetime import datetime, timezone, timedelta
from html import unescape
JST = timezone(timedelta(hours=9))
PARIS = timezone(timedelta(hours=2)) # CEST (summer)
BASHO_CODES = {
"01": ("Hatsu", "初場所", "Janvier"),
"03": ("Haru", "春場所", "Mars"),
"05": ("Natsu", "夏場所", "Mai"),
"07": ("Nagoya", "名古屋場所", "Juillet"),
"09": ("Aki", "秋場所", "Septembre"),
"11": ("Kyushu", "九州場所", "Novembre"),
}
# ─── Scraping ───
def fetch_results(basho, day):
"""Fetch and parse results for a given basho code (e.g. 202607) and day (1-15)."""
url = f"https://sumodb.sumogames.de/Results.aspx?b={basho}&d={day}"
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0 (SumaBot)"})
with urllib.request.urlopen(req, timeout=30) as resp:
html = resp.read().decode("utf-8", errors="replace")
# Header info
h1 = re.search(r"
([^<]+)
", html)
h2 = re.search(r"
([^<]+)
", html)
title = h1.group(1).strip() if h1 else f"Basho {basho} Day {day}"
date_str = h2.group(1).strip() if h2 else ""
# Parse tables by division
tables = re.findall(r'
', html, re.DOTALL)
divisions = {}
division_names = ["Makuuchi", "Juryo", "Makushita", "Sandanme", "Jonidan", "Jonokuchi"]
for i, table in enumerate(tables):
if i >= len(division_names):
break
div_name = division_names[i]
matches = re.findall(
r'
(.*?)
(.*?)
'
r'
(.*?)
(.*?)
'
r'
(.*?)
',
table, re.DOTALL)
parsed = []
for m in matches:
east_result_img = m[0]
east_cell = m[1]
kim_cell = m[2]
west_cell = m[3]
west_result_img = m[4]
# Parse wrestler info
east = parse_wrestler(east_cell)
west = parse_wrestler(west_cell)
technique = parse_technique(kim_cell)
# Determine winner
east_won = "shiro" in east_result_img
west_won = "shiro" in west_result_img
east_lost = "kuro" in east_result_img
west_lost = "kuro" in west_result_img
fusen = "fusen" in east_result_img or "fusen" in west_result_img
if east_won:
winner = "east"
elif west_won:
winner = "west"
elif fusen:
# Check who got the fusensho (win by default)
winner = "east" if "fusensho" in east_result_img else "west"
else:
winner = None
parsed.append({
"east": east,
"west": west,
"technique": technique,
"winner": winner,
"fusen": fusen,
})
divisions[div_name] = parsed
return {
"title": title,
"date_str": date_str,
"basho": basho,
"day": day,
"divisions": divisions,
}
def parse_wrestler(cell):
"""Extract wrestler name, rank, record, and shikona from a cell."""
rank = ""
rank_m = re.search(r'(.*?)', cell)
if rank_m:
rank = unescape(rank_m.group(1).strip())
# Japanese shikona from title
shikona_ja = ""
title_m = re.search(r"title='([^']+)'", cell)
if title_m:
parts = title_m.group(1).split(",")
if parts:
shikona_ja = parts[0].strip()
# English shikona from link text
name = ""
link_m = re.search(r"href='Rikishi\.aspx\?r=\d+'>([^<]+)", cell)
if link_m:
name = link_m.group(1).strip()
# Record
record = ""
rec_m = re.search(r'([^<]+)', cell)
if rec_m:
record = rec_m.group(1).strip()
return {"rank": rank, "name": name, "shikona_ja": shikona_ja, "record": record}
def parse_technique(cell):
"""Extract winning technique from kimari cell."""
# Remove HTML tags
text = re.sub(r"<[^>]+>", "", cell)
text = unescape(text).strip()
# The technique is on its own line (after the first )
lines = [l.strip() for l in text.split("\n") if l.strip()]
# Filter out H2H records (contain digits)
tech = ""
for line in lines:
if not re.search(r"\d", line) and line not in ("", " "):
tech = line
break
return tech
# ─── HTML generation ───
def generate_html(data):
now_utc = datetime.now(timezone.utc)
now_jst = now_utc.astimezone(JST)
now_paris = now_utc.astimezone(PARIS)
basho_code = data["basho"]
month_code = basho_code[4:6]
year = basho_code[:4]
basho_en, basho_ja, basho_fr = BASHO_CODES.get(month_code, ("Unknown", "不明", "Inconnu"))
# Build division sections
divisions_html = []
for div_name in ["Makuuchi", "Juryo", "Makushita", "Sandanme", "Jonidan", "Jonokuchi"]:
matches = data["divisions"].get(div_name, [])
if not matches:
continue
matches_html = []
for m in matches:
east = m["east"]
west = m["west"]
winner = m["winner"]
# Winner/loser styling
east_class = "winner" if winner == "east" else ("loser" if winner == "west" else "")
west_class = "winner" if winner == "west" else ("loser" if winner == "east" else "")
tech_display = m["technique"] if m["technique"] and not m["fusen"] else ""
if m["fusen"]:
tech_display = "不戦勝" if winner == "east" else "不戦勝"
match_html = f"""
"""
return html
# ─── Auto-detect current basho and day ───
def detect_current_basho_day():
"""Determine the current basho code and day based on the date."""
now_jst = datetime.now(JST)
year = now_jst.year
month = now_jst.month
# Basho months: 1, 3, 5, 7, 9, 11
basho_months = {1: "01", 3: "03", 5: "05", 7: "07", 9: "09", 11: "11"}
# Find the most recent basho
basho_code = None
for m in [1, 3, 5, 7, 9, 11]:
if month >= m:
basho_code = f"{year}{basho_months[m]}"
elif month < 3:
basho_code = f"{year-1}11"
break
# If no basho yet this year, use last November
if basho_code is None:
basho_code = f"{year-1}11"
# Basho runs for 15 days starting on the 2nd Sunday of the month
# For simplicity, try days 1-15 and use the latest one with data
# First try day 15, then go down
for day in range(15, 0, -1):
try:
data = fetch_results(basho_code, day)
if data["divisions"]:
return basho_code, day, data
except Exception:
continue
return basho_code, 1, None
if __name__ == "__main__":
import sys
if len(sys.argv) >= 3:
basho = sys.argv[1]
day = int(sys.argv[2])
else:
basho, day, pre_data = detect_current_basho_day()
if pre_data:
data = pre_data
else:
data = fetch_results(basho, day)
if len(sys.argv) < 3:
# Auto mode
if 'data' not in dir():
data = fetch_results(basho, day)
else:
data = fetch_results(basho, day)
html = generate_html(data)
output_path = "/home/jerome/hermes-sites/suma.hidrago.click/index.html"
with open(output_path, "w", encoding="utf-8") as f:
f.write(html)
print(f"Suma page generated: basho={data['basho']} day={data['day']} -> {output_path}")