ChessCalcNextTour/swiss_calc/parser.py
Roman Vrubel cf7fafd2ba chessCalc: парсер chess-results.com + FIDE Swiss + Docker
Принимает URL турнира, показывает пары следующего тура в Telegram-формате.
- parser.py: парсинг chess-results.com
- swiss.py: FIDE Dutch System
- display.py: форматирование для Telegram
- Docker: docker compose run --rm chess-calc 'URL'
2026-06-14 13:57:32 +00:00

499 lines
17 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

"""
Parser for chess-results.com tournament pages.
Fetches and parses:
- Starting list (players with ratings)
- Round pairings/results
- Standings with tiebreakers
"""
import re
import requests
from bs4 import BeautifulSoup
from typing import Optional
HEADERS = {
'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
}
RESULT_MAP = {
'1': 1.0,
'0': 0.0,
'½': 0.5,
'0,5': 0.5,
'+': 1.0, # win by forfeit
'-': 0.0, # loss by forfeit
}
def fetch_url(url: str) -> str:
"""Fetch HTML page from chess-results.com."""
resp = requests.get(url, headers=HEADERS, timeout=20)
resp.raise_for_status()
resp.encoding = 'utf-8'
return resp.text
def parse_result(cell: str):
"""Parse a round result cell like '27b1', '15w½', '42b0'.
Returns (opponent_sno: int, color: str, points: float) or None.
"""
cell = cell.strip()
if not cell:
return None
m = re.match(r'^(\d+)([bw])([10½]+|0,5)$', cell)
if m:
sno = int(m.group(1))
color = 'b' if m.group(2) == 'b' else 'w'
pts_str = m.group(3).replace(',', '.')
pts = float(pts_str) if '.' in pts_str else (0.5 if pts_str == '½' else float(pts_str))
return sno, color, pts
# Handle forfeit results like '27b+'
m = re.match(r'^(\d+)([bw])([+\-])$', cell)
if m:
sno = int(m.group(1))
color = 'b' if m.group(2) == 'b' else 'w'
pts = 1.0 if m.group(3) == '+' else 0.0
return sno, color, pts
return None
def parse_start_list(html: str) -> dict:
"""Parse art=5 page: returns dict of {sno: {'name': str, 'rating': int, 'fed': str}}"""
soup = BeautifulSoup(html, 'html.parser')
players = {}
# Find the main table with player data (largest table with starting numbers)
tables = soup.find_all('table')
for table in tables:
rows = table.find_all('tr')
if len(rows) < 5:
continue
for row in rows:
cells = row.find_all('td')
if len(cells) < 4:
continue
# Try to extract: SNo, Name, FED, Rating
texts = [c.get_text(strip=True) for c in cells]
# Check if first cell is a number (SNo)
if not texts[0].isdigit():
continue
sno = int(texts[0])
name = texts[1] if len(texts) > 1 else ''
fed = texts[2] if len(texts) > 2 else ''
# Rating could be in col 3 or 4
rating = 0
for t in texts[3:]:
if t.isdigit() and len(t) >= 3:
rating = int(t)
break
if name:
players[sno] = {
'name': name,
'rating': rating,
'fed': fed,
}
if players:
break
return players
def parse_standings(html: str, current_round: int) -> list:
"""Parse art=4 page (standings).
Returns list of dicts:
{
'sno': int,
'rank': int,
'name': str,
'fed': str,
'points': float,
'results': [(opponent_sno, color, score), ...], # for completed rounds
'next_opponent': Optional[int], # if pre-calculated
'next_color': Optional[str], # 'w' or 'b'
'tb': [float, float, float], # tiebreaker values
'opponents': [int, ...], # all opponents so far
}
"""
soup = BeautifulSoup(html, 'html.parser')
players = []
tables = soup.find_all('table')
for table in tables:
rows = table.find_all('tr')
if len(rows) < 10:
continue
for row in rows:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
# Filter: first cell should be a rank number
if not texts or not texts[0].isdigit():
continue
# Skip if not enough cols for a player row (at least rank + name + results)
if len(texts) < 8:
continue
rank = int(texts[0])
# Usually: rank, (empty), name, fed, rd1, rd2, ..., pts, tb1, tb2, tb3
# The empty column sometimes joins with rank
col_offset = 0
if rank == 0 or rank > 200:
continue
# Find name column
# Typical: rank | (empty) | name | fed | results...
name = ''
fed = ''
name_idx = 1
for ci in range(1, min(5, len(texts))):
t = texts[ci]
if t and not t.isdigit() and len(t) > 2 and not t.startswith('http'):
if ci > 1 or not texts[0].isdigit():
# Check if this name contains 2+ words in Russian or English
if re.search(r'[а-яА-Яa-zA-Z]', t):
name = t
name_idx = ci
# Next column after name is usually federation
if ci + 1 < len(texts):
fed = texts[ci + 1]
break
if ci == name_idx:
name = t
if ci + 1 < len(texts):
fed = texts[ci + 1]
break
if not name:
continue
# Results start after federation column
# fed is at name_idx+1, so results start at name_idx+2
result_start = name_idx + 2 if name_idx + 2 < len(texts) else name_idx + 1
results = []
opponents = []
next_opponent = None
next_color = None
pts_found = False
tb_values = []
# Process each column from result_start
result_cols = texts[result_start:]
pts_col_idx = -1
# Find result cells (format: XXX or XXb1, XXw½ etc)
for ci, col in enumerate(result_cols):
if not col:
continue
# Check for next opponent format (e.g., "4w", "12b")
m = re.match(r'^(\d+)([bw])$', col)
if m and not pts_found:
# This could be a result (if it has 1/0/½) or next opponent
# Check if there are more cells and the next one is numeric (points)
pass
# Try to parse as round result
parsed = parse_result(col)
if parsed:
opp, color, pts = parsed
results.append({'opponent': opp, 'color': color, 'score': pts})
opponents.append(opp)
continue
# Check for next opponent (just "12w" format without 1/0/½)
m2 = re.match(r'^(\d+)([bw])$', col)
if m2:
next_opponent = int(m2.group(1))
next_color = m2.group(2)
continue
# Check if it's the points column
if re.match(r'^\d+(?:[.,]\d)?$', col) and not pts_found:
pts_str = col.replace(',', '.')
points = float(pts_str)
pts_found = True
pts_col_idx = ci
continue
# Tiebreaker columns (after points)
if pts_found and re.match(r'^\d+(?:[.,]\d)?$', col):
tb_values.append(float(col.replace(',', '.')))
# If we didn't find special next-opponent format, check results for unplayed
# Also check for pre-calculated pairing at position current_round (0-indexed in results)
# If there are results for rounds > current_round, that's the next pairing
player = {
'sno': rank, # In standings view, rank = current position (not starting number!)
'rank': rank,
'name': name,
'fed': fed,
'points': float(texts[-len(tb_values)-1].replace(',', '.')) if texts else 0,
'results': results,
'next_opponent': next_opponent,
'next_color': next_color,
'tb': tb_values,
'opponents': opponents,
}
players.append(player)
if players:
break
return players
def parse_round_pairings(html: str, round_num: int) -> list:
"""Parse art=2 page for a specific round.
Returns list of dicts:
{
'board': int,
'white_sno': int,
'white_name': str,
'white_rating': int,
'white_pts': float,
'black_sno': int,
'black_name': str,
'black_rating': int,
'black_pts': float,
'result': Optional[str], # '1-0', '½-½', '0-1'
}
"""
soup = BeautifulSoup(html, 'html.parser')
pairings = []
tables = soup.find_all('table')
for table in tables:
rows = table.find_all('tr')
if len(rows) < 5:
continue
for row in rows:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
if len(texts) < 10:
continue
# Format: Board | WhiteSNo | | WhiteName | WhiteRating | WhitePts | Result | BlackPts | | BlackName | BlackRating | BlackSNo
# Or: Board | SNo | | Name | Rating | Pts | Result | Pts | | Name | Rating | SNo
# First cell should be a board number
if not texts[0].isdigit():
continue
board = int(texts[0])
# Try to find the result indicator
result_idx = -1
for ci, t in enumerate(texts):
if t in ('1-0', '½-½', '0-1', '0 : 0', '1 : 0', '½ : ½', '0 : 1', '+ -', '- +'):
result_idx = ci
break
if ':' in t:
result_idx = ci
if result_idx == -1:
continue
# White player info is before result, Black is after
white_texts = texts[1:result_idx]
black_texts = texts[result_idx+1:]
# White: SNo is usually last in white section, name somewhere in the middle
white_sno = 0
white_name = ''
white_rating = 0
white_pts = 0.0
for t in white_texts:
if t.isdigit() and len(t) <= 3:
white_sno = int(t)
if re.search(r'[а-яА-Яa-zA-Z]{3,}', t) and len(t) > 3:
white_name = t
if t.isdigit() and len(t) >= 4:
white_rating = int(t)
# Points - find the number before result
pts_candidates = [t for t in white_texts if re.match(r'^\d+(?:[.,]\d)?$', t) and len(t) <= 4]
if pts_candidates:
white_pts = float(pts_candidates[-1].replace(',', '.'))
for t in black_texts:
if t.isdigit() and len(t) <= 3:
black_sno = int(t)
if re.search(r'[а-яА-Яa-zA-Z]{3,}', t) and len(t) > 3:
black_name = t
if t.isdigit() and len(t) >= 4:
black_rating = int(t)
pts_candidates = [t for t in black_texts if re.match(r'^\d+(?:[.,]\d)?$', t) and len(t) <= 4]
if pts_candidates:
black_pts = float(pts_candidates[0].replace(',', '.'))
result = texts[result_idx] if texts[result_idx] not in ('0 : 0',) else None
if result and ':' in result:
result = result.replace(' : ', '-')
pairings.append({
'board': board,
'white_sno': white_sno,
'white_name': white_name,
'white_rating': white_rating,
'white_pts': white_pts,
'black_sno': black_sno,
'black_name': black_name,
'black_rating': black_rating,
'black_pts': black_pts,
'result': result,
})
if pairings:
break
return pairings
def extract_tournament_meta(html: str) -> dict:
"""Extract tournament metadata (name, number of rounds) from any page HTML."""
soup = BeautifulSoup(html, 'html.parser')
info = {'name': '', 'num_rounds': 0, 'current_round': 1}
h2 = soup.find('h2')
if h2:
info['name'] = h2.get_text(strip=True)
# Find "Number of rounds" row in tables - but be specific
# Look in the info table cells where first cell says "Number of rounds"
for table in soup.find_all('table'):
rows = table.find_all('tr')
found_rounds = False
for row in rows:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
for ci, t in enumerate(texts):
if t == 'Number of rounds' and ci + 1 < len(texts):
m = re.search(r'^(\d+)$', texts[ci + 1])
if m:
info['num_rounds'] = int(m.group(1))
found_rounds = True
break
if found_rounds:
break
if found_rounds:
break
# Detect current round from navigation: "Тур4/9" or "Round X/Y"
nav_text = soup.get_text()
m = re.search(r'Тур(\d+)/\d+|Round\s*(\d+)\s*/\s*\d+', nav_text)
if m:
info['current_round'] = int(m.group(1) or m.group(2))
return info
def detect_current_round(url: str) -> int:
"""Detect current round by checking which rd parameter has results."""
for rd in range(1, 12):
rd_url = url.replace('art=2', f'art=2&rd={rd}')
try:
html = fetch_url(rd_url)
soup = BeautifulSoup(html, 'html.parser')
# Find tables with pairings
tables = soup.find_all('table')
has_pairings = False
for table in tables:
rows = table.find_all('tr')
if len(rows) > 3:
texts = rows[0].find_all('td')
txt = ' '.join(t.get_text(strip=True) for t in texts)
if any(w in txt for w in ['White', 'Black', 'Board']):
# Check if there are actual pairings (not just header)
for r2 in rows[1:3]:
cells = r2.find_all('td')
if len(cells) > 5:
has_pairings = True
break
if has_pairings:
break
if not has_pairings:
return rd - 1
except Exception:
return rd - 1
return 1
def fetch_tournament(url: str) -> dict:
"""Full tournament data fetch.
Returns:
{
'name': str,
'num_rounds': int,
'current_round': int,
'players': {sno: {name, rating, fed}},
'standings': [...], # players sorted by rank
'pairings': {rd: [...]}, # completed round pairings
}
"""
# Normalize URL: strip art/rd params, keep only base URL with tnr
base_url = re.sub(r'[&?]art=\d+', '', url)
base_url = re.sub(r'[&?]rd=\d+', '', base_url)
base_url = re.sub(r'[&?]turdet=\w+', '', base_url)
base_url = re.sub(r'[&?]SNode=\w+', '', base_url)
# Standings page (art=4) - this page has ALL the info we need
standings_url = base_url + '&art=4&turdet=YES'
try:
html = fetch_url(standings_url)
except Exception as e:
raise RuntimeError(f'Не удалось загрузить турнир: {e}')
# Extract metadata from same HTML
meta = extract_tournament_meta(html)
num_rounds = meta['num_rounds']
# Parse standings
standings = parse_standings(html, 0)
if standings:
current_round = len(standings[0].get('results', []))
else:
current_round = 1
# Override current_round from navigation if available
# Navigation "Тур4/9" means round 4 is current/playing → 3 completed
if meta['current_round'] > 1:
current_round = meta['current_round'] # this is the current playing round
# But we want the last COMPLETED round for calculations
# If results show N entries, N rounds are completed
# Reconcile: completed rounds = len(results for first player)
if standings and len(standings[0].get('results', [])) < current_round:
current_round = len(standings[0].get('results', []))
# Fetch starting list (art=5) for ratings
start_url = base_url + '&art=5&turdet=YES'
try:
html_start = fetch_url(start_url)
players = parse_start_list(html_start)
except Exception:
players = {}
# Match ratings from start list into standings
for s in standings:
# Find by name match
for sno, p in players.items():
if p['name'].lower() == s['name'].lower():
s['rating'] = p['rating']
s['starting_sno'] = sno
break
else:
s['rating'] = 0
s['starting_sno'] = s['rank']
return {
'name': meta['name'],
'num_rounds': num_rounds if num_rounds else 9, # fallback
'current_round': current_round,
'players': players,
'standings': standings,
}