""" Parser for chess-results.com tournament pages. Fetches and parses: - Starting list (players with ratings) - Round pairings/results - Standings with tiebreakers """ import re import requests from bs4 import BeautifulSoup from typing import Optional HEADERS = { 'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36' } RESULT_MAP = { '1': 1.0, '0': 0.0, '½': 0.5, '0,5': 0.5, '+': 1.0, # win by forfeit '-': 0.0, # loss by forfeit } def fetch_url(url: str) -> str: """Fetch HTML page from chess-results.com.""" resp = requests.get(url, headers=HEADERS, timeout=20) resp.raise_for_status() resp.encoding = 'utf-8' return resp.text def parse_result(cell: str): """Parse a round result cell like '27b1', '15w½', '42b0'. Returns (opponent_sno: int, color: str, points: float) or None. """ cell = cell.strip() if not cell: return None m = re.match(r'^(\d+)([bw])([10½]+|0,5)$', cell) if m: sno = int(m.group(1)) color = 'b' if m.group(2) == 'b' else 'w' pts_str = m.group(3).replace(',', '.') pts = float(pts_str) if '.' in pts_str else (0.5 if pts_str == '½' else float(pts_str)) return sno, color, pts # Handle forfeit results like '27b+' m = re.match(r'^(\d+)([bw])([+\-])$', cell) if m: sno = int(m.group(1)) color = 'b' if m.group(2) == 'b' else 'w' pts = 1.0 if m.group(3) == '+' else 0.0 return sno, color, pts return None def parse_start_list(html: str) -> dict: """Parse art=5 page: returns dict of {sno: {'name': str, 'rating': int, 'fed': str}}""" soup = BeautifulSoup(html, 'html.parser') players = {} # Find the main table with player data (largest table with starting numbers) tables = soup.find_all('table') for table in tables: rows = table.find_all('tr') if len(rows) < 5: continue for row in rows: cells = row.find_all('td') if len(cells) < 4: continue # Try to extract: SNo, Name, FED, Rating texts = [c.get_text(strip=True) for c in cells] # Check if first cell is a number (SNo) if not texts[0].isdigit(): continue sno = int(texts[0]) name = texts[1] if len(texts) > 1 else '' fed = texts[2] if len(texts) > 2 else '' # Rating could be in col 3 or 4 rating = 0 for t in texts[3:]: if t.isdigit() and len(t) >= 3: rating = int(t) break if name: players[sno] = { 'name': name, 'rating': rating, 'fed': fed, } if players: break return players def parse_standings(html: str, current_round: int) -> list: """Parse art=4 page (standings). Returns list of dicts: { 'sno': int, 'rank': int, 'name': str, 'fed': str, 'points': float, 'results': [(opponent_sno, color, score), ...], # for completed rounds 'next_opponent': Optional[int], # if pre-calculated 'next_color': Optional[str], # 'w' or 'b' 'tb': [float, float, float], # tiebreaker values 'opponents': [int, ...], # all opponents so far } """ soup = BeautifulSoup(html, 'html.parser') players = [] tables = soup.find_all('table') for table in tables: rows = table.find_all('tr') if len(rows) < 10: continue for row in rows: cells = row.find_all('td') texts = [c.get_text(strip=True) for c in cells] # Filter: first cell should be a rank number if not texts or not texts[0].isdigit(): continue # Skip if not enough cols for a player row (at least rank + name + results) if len(texts) < 8: continue rank = int(texts[0]) # Usually: rank, (empty), name, fed, rd1, rd2, ..., pts, tb1, tb2, tb3 # The empty column sometimes joins with rank col_offset = 0 if rank == 0 or rank > 200: continue # Find name column # Typical: rank | (empty) | name | fed | results... name = '' fed = '' name_idx = 1 for ci in range(1, min(5, len(texts))): t = texts[ci] if t and not t.isdigit() and len(t) > 2 and not t.startswith('http'): if ci > 1 or not texts[0].isdigit(): # Check if this name contains 2+ words in Russian or English if re.search(r'[а-яА-Яa-zA-Z]', t): name = t name_idx = ci # Next column after name is usually federation if ci + 1 < len(texts): fed = texts[ci + 1] break if ci == name_idx: name = t if ci + 1 < len(texts): fed = texts[ci + 1] break if not name: continue # Results start after federation column # fed is at name_idx+1, so results start at name_idx+2 result_start = name_idx + 2 if name_idx + 2 < len(texts) else name_idx + 1 results = [] opponents = [] next_opponent = None next_color = None pts_found = False tb_values = [] # Process each column from result_start result_cols = texts[result_start:] pts_col_idx = -1 # Find result cells (format: XXX or XXb1, XXw½ etc) for ci, col in enumerate(result_cols): if not col: continue # Check for next opponent format (e.g., "4w", "12b") m = re.match(r'^(\d+)([bw])$', col) if m and not pts_found: # This could be a result (if it has 1/0/½) or next opponent # Check if there are more cells and the next one is numeric (points) pass # Try to parse as round result parsed = parse_result(col) if parsed: opp, color, pts = parsed results.append({'opponent': opp, 'color': color, 'score': pts}) opponents.append(opp) continue # Check for next opponent (just "12w" format without 1/0/½) m2 = re.match(r'^(\d+)([bw])$', col) if m2: next_opponent = int(m2.group(1)) next_color = m2.group(2) continue # Check if it's the points column if re.match(r'^\d+(?:[.,]\d)?$', col) and not pts_found: pts_str = col.replace(',', '.') points = float(pts_str) pts_found = True pts_col_idx = ci continue # Tiebreaker columns (after points) if pts_found and re.match(r'^\d+(?:[.,]\d)?$', col): tb_values.append(float(col.replace(',', '.'))) # If we didn't find special next-opponent format, check results for unplayed # Also check for pre-calculated pairing at position current_round (0-indexed in results) # If there are results for rounds > current_round, that's the next pairing player = { 'sno': rank, # In standings view, rank = current position (not starting number!) 'rank': rank, 'name': name, 'fed': fed, 'points': float(texts[-len(tb_values)-1].replace(',', '.')) if texts else 0, 'results': results, 'next_opponent': next_opponent, 'next_color': next_color, 'tb': tb_values, 'opponents': opponents, } players.append(player) if players: break return players def parse_round_pairings(html: str, round_num: int) -> list: """Parse art=2 page for a specific round. Returns list of dicts: { 'board': int, 'white_sno': int, 'white_name': str, 'white_rating': int, 'white_pts': float, 'black_sno': int, 'black_name': str, 'black_rating': int, 'black_pts': float, 'result': Optional[str], # '1-0', '½-½', '0-1' } """ soup = BeautifulSoup(html, 'html.parser') pairings = [] tables = soup.find_all('table') for table in tables: rows = table.find_all('tr') if len(rows) < 5: continue for row in rows: cells = row.find_all('td') texts = [c.get_text(strip=True) for c in cells] if len(texts) < 10: continue # Format: Board | WhiteSNo | | WhiteName | WhiteRating | WhitePts | Result | BlackPts | | BlackName | BlackRating | BlackSNo # Or: Board | SNo | | Name | Rating | Pts | Result | Pts | | Name | Rating | SNo # First cell should be a board number if not texts[0].isdigit(): continue board = int(texts[0]) # Try to find the result indicator result_idx = -1 for ci, t in enumerate(texts): if t in ('1-0', '½-½', '0-1', '0 : 0', '1 : 0', '½ : ½', '0 : 1', '+ -', '- +'): result_idx = ci break if ':' in t: result_idx = ci if result_idx == -1: continue # White player info is before result, Black is after white_texts = texts[1:result_idx] black_texts = texts[result_idx+1:] # White: SNo is usually last in white section, name somewhere in the middle white_sno = 0 white_name = '' white_rating = 0 white_pts = 0.0 for t in white_texts: if t.isdigit() and len(t) <= 3: white_sno = int(t) if re.search(r'[а-яА-Яa-zA-Z]{3,}', t) and len(t) > 3: white_name = t if t.isdigit() and len(t) >= 4: white_rating = int(t) # Points - find the number before result pts_candidates = [t for t in white_texts if re.match(r'^\d+(?:[.,]\d)?$', t) and len(t) <= 4] if pts_candidates: white_pts = float(pts_candidates[-1].replace(',', '.')) for t in black_texts: if t.isdigit() and len(t) <= 3: black_sno = int(t) if re.search(r'[а-яА-Яa-zA-Z]{3,}', t) and len(t) > 3: black_name = t if t.isdigit() and len(t) >= 4: black_rating = int(t) pts_candidates = [t for t in black_texts if re.match(r'^\d+(?:[.,]\d)?$', t) and len(t) <= 4] if pts_candidates: black_pts = float(pts_candidates[0].replace(',', '.')) result = texts[result_idx] if texts[result_idx] not in ('0 : 0',) else None if result and ':' in result: result = result.replace(' : ', '-') pairings.append({ 'board': board, 'white_sno': white_sno, 'white_name': white_name, 'white_rating': white_rating, 'white_pts': white_pts, 'black_sno': black_sno, 'black_name': black_name, 'black_rating': black_rating, 'black_pts': black_pts, 'result': result, }) if pairings: break return pairings def extract_tournament_meta(html: str) -> dict: """Extract tournament metadata (name, number of rounds) from any page HTML.""" soup = BeautifulSoup(html, 'html.parser') info = {'name': '', 'num_rounds': 0, 'current_round': 1} h2 = soup.find('h2') if h2: info['name'] = h2.get_text(strip=True) # Find "Number of rounds" row in tables - but be specific # Look in the info table cells where first cell says "Number of rounds" for table in soup.find_all('table'): rows = table.find_all('tr') found_rounds = False for row in rows: cells = row.find_all('td') texts = [c.get_text(strip=True) for c in cells] for ci, t in enumerate(texts): if t == 'Number of rounds' and ci + 1 < len(texts): m = re.search(r'^(\d+)$', texts[ci + 1]) if m: info['num_rounds'] = int(m.group(1)) found_rounds = True break if found_rounds: break if found_rounds: break # Detect current round from navigation: "Тур4/9" or "Round X/Y" nav_text = soup.get_text() m = re.search(r'Тур(\d+)/\d+|Round\s*(\d+)\s*/\s*\d+', nav_text) if m: info['current_round'] = int(m.group(1) or m.group(2)) return info def detect_current_round(url: str) -> int: """Detect current round by checking which rd parameter has results.""" for rd in range(1, 12): rd_url = url.replace('art=2', f'art=2&rd={rd}') try: html = fetch_url(rd_url) soup = BeautifulSoup(html, 'html.parser') # Find tables with pairings tables = soup.find_all('table') has_pairings = False for table in tables: rows = table.find_all('tr') if len(rows) > 3: texts = rows[0].find_all('td') txt = ' '.join(t.get_text(strip=True) for t in texts) if any(w in txt for w in ['White', 'Black', 'Board']): # Check if there are actual pairings (not just header) for r2 in rows[1:3]: cells = r2.find_all('td') if len(cells) > 5: has_pairings = True break if has_pairings: break if not has_pairings: return rd - 1 except Exception: return rd - 1 return 1 def fetch_tournament(url: str) -> dict: """Full tournament data fetch. Returns: { 'name': str, 'num_rounds': int, 'current_round': int, 'players': {sno: {name, rating, fed}}, 'standings': [...], # players sorted by rank 'pairings': {rd: [...]}, # completed round pairings } """ # Normalize URL: strip art/rd params, keep only base URL with tnr base_url = re.sub(r'[&?]art=\d+', '', url) base_url = re.sub(r'[&?]rd=\d+', '', base_url) base_url = re.sub(r'[&?]turdet=\w+', '', base_url) base_url = re.sub(r'[&?]SNode=\w+', '', base_url) # Standings page (art=4) - this page has ALL the info we need standings_url = base_url + '&art=4&turdet=YES' try: html = fetch_url(standings_url) except Exception as e: raise RuntimeError(f'Не удалось загрузить турнир: {e}') # Extract metadata from same HTML meta = extract_tournament_meta(html) num_rounds = meta['num_rounds'] # Parse standings standings = parse_standings(html, 0) if standings: current_round = len(standings[0].get('results', [])) else: current_round = 1 # Override current_round from navigation if available # Navigation "Тур4/9" means round 4 is current/playing → 3 completed if meta['current_round'] > 1: current_round = meta['current_round'] # this is the current playing round # But we want the last COMPLETED round for calculations # If results show N entries, N rounds are completed # Reconcile: completed rounds = len(results for first player) if standings and len(standings[0].get('results', [])) < current_round: current_round = len(standings[0].get('results', [])) # Fetch starting list (art=5) for ratings start_url = base_url + '&art=5&turdet=YES' try: html_start = fetch_url(start_url) players = parse_start_list(html_start) except Exception: players = {} # Match ratings from start list into standings for s in standings: # Find by name match for sno, p in players.items(): if p['name'].lower() == s['name'].lower(): s['rating'] = p['rating'] s['starting_sno'] = sno break else: s['rating'] = 0 s['starting_sno'] = s['rank'] return { 'name': meta['name'], 'num_rounds': num_rounds if num_rounds else 9, # fallback 'current_round': current_round, 'players': players, 'standings': standings, }