""" Parser for chess-results.com tournament pages. Handles all page states: - Before pairings: standings with completed rounds - After pairings published: standings + next-round column - After games played: updated standings - art=2 pairings pages in compact format """ import re import requests from bs4 import BeautifulSoup from typing import Optional, List, Dict, Tuple HEADERS = { 'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36' } def fetch_url(url: str) -> str: resp = requests.get(url, headers=HEADERS, timeout=20) resp.raise_for_status() resp.encoding = 'utf-8' return resp.text # ── Result parsing ──────────────────────────────────────────── def parse_result(cell: str) -> Optional[Tuple[int, str, float]]: """Parse a round result cell. Returns (opponent_sno, color, score) or None. sno=0 means bye/forfeit; color='-' means no color. """ cell = cell.strip() if not cell: return None # Standard: '27b1', '15w½' m = re.match(r'^(\d+)([bw])([10½=]+|0[,.]5)$', cell) if m: sno = int(m.group(1)) color = 'w' if m.group(2) == 'w' else 'b' pts_str = m.group(3).replace(',', '.').replace('=', '0.5').replace('½', '0.5') pts = float(pts_str) return sno, color, pts # Forfeit with opponent: '27b+' m = re.match(r'^(\d+)([bw])([+\-])$', cell) if m: sno = int(m.group(1)) color = 'w' if m.group(2) == 'w' else 'b' pts = 1.0 if m.group(3) == '+' else 0.0 return sno, color, pts # Bye/forfeit: '-0', '-1', '-' m = re.match(r'^-([01½]?)$', cell) if m: pts_str = m.group(1).replace('½', '0.5') pts = float(pts_str) if pts_str else 0.0 return 0, '-', pts return None def parse_next_opponent(cell: str) -> Optional[Tuple[int, str]]: """Parse '4w' or '12b' — next opponent and color.""" m = re.match(r'^(\d+)([bw])$', cell.strip()) if m: return int(m.group(1)), m.group(2) return None # ── Standings parsing ───────────────────────────────────────── def _normalize_name(name: str) -> str: """Normalize name for matching: remove commas, collapse whitespace.""" return re.sub(r'\s+', ' ', name.replace(',', '')).strip() def _has_cyrillic(s: str) -> bool: return bool(re.search(r'[а-яА-ЯёЁ]', s)) def _is_player_name(t: str) -> bool: """Check if text looks like a player name (Latin or Cyrillic).""" if len(t) <= 5: return False if re.match(r'^\d+$', t): return False if re.match(r'^[A-Z]{3}$', t): return False if re.match(r'^\d+[bw][½\d]?$', t): return False return bool(re.search(r'[A-Za-zА-Яа-я]', t)) def parse_standings(html: str) -> List[Dict]: """Parse standings from art=4 or art=5 page. Detects player rows by: first column is a rank number, one column has a Cyrillic name. """ soup = BeautifulSoup(html, 'html.parser') players = [] tables = soup.find_all('table') for table in tables: rows = table.find_all('tr') if len(rows) < 10: continue for row in rows: cells = row.find_all('td') texts = [c.get_text(strip=True) for c in cells] if len(texts) < 6: continue if not texts[0].isdigit(): continue sno = int(texts[0]) if sno < 1 or sno > 5000: continue # Find name: look for player name (Latin or Cyrillic) name = '' name_idx = -1 for ci, t in enumerate(texts): if _is_player_name(t): name = t name_idx = ci break if not name: continue # Federation is usually the next column, if it looks like a code fed = '' if name_idx + 1 < len(texts): t = texts[name_idx + 1] if re.match(r'^[A-Z]{3}$', t): fed = t # Process columns after name+federation for results and points data_cols = texts[name_idx + (2 if fed else 1):] results = [] next_opponent = None next_color = None points = 0.0 tb_values = [] found_points = False for col in data_cols: if not col: continue # Try result res = parse_result(col) if res: opp, color, pts = res results.append({'opponent': opp, 'color': color, 'score': pts}) continue # Try next opponent no = parse_next_opponent(col) if no: next_opponent, next_color = no continue # Points: first numeric with optional decimal after results if re.match(r'^\d+([,.]\d)?$', col) and not found_points: points = float(col.replace(',', '.')) found_points = True continue # Tiebreakers (after points) if found_points and re.match(r'^\d+([,.]\d+)?$', col): tb_values.append(float(col.replace(',', '.'))) players.append({ 'rank': len(players) + 1, # position in list = current rank 'name': name, 'fed': fed, 'points': points, 'starting_sno': sno, # from first column 'results': results, 'next_opponent': next_opponent, 'next_color': next_color, 'tb': tb_values, }) if players: break return players # ── Round pairings (art=2 compact format) ───────────────────── def parse_game_row(texts: List[str], board: int) -> Optional[Dict]: """Parse one game row from art=2 page. Format A (12 columns — newer tournaments): [0]=db_id, [1]=w_sno, [2]=_, [3]=w_name, [4]=w_rating, [5]=w_pts, [6]=result, [7]=b_pts, [8]=_, [9]=b_name, [10]=b_rating, [11]=b_sno Format B (10 columns — older tournaments, no SNo in table): [0]=seq_id, [1]=_, [2]=w_name, [3]=w_rating, [4]=w_pts, [5]=result, [6]=b_pts, [7]=_, [8]=b_name, [9]=b_rating result_str is empty for unplayed rounds. """ if len(texts) < 10: return None if not texts[0].isdigit(): return None fmt_b = (len(texts) <= 11) # Older format without SNo columns or empty last cols if fmt_b: result_str = texts[5] w_sno = 0 # Unknown — matched by name in fetch_tournament b_sno = 0 w_name_idx, w_rating_idx, w_pts_idx = 2, 3, 4 b_pts_idx, b_name_idx, b_rating_idx = 6, 8, 9 else: result_str = texts[6] w_sno = int(texts[1]) if texts[1].isdigit() else 0 try: b_sno = int(texts[11]) if texts[11].strip().isdigit() else 0 except (IndexError, ValueError): b_sno = 0 w_name_idx, w_rating_idx, w_pts_idx = 3, 4, 5 b_pts_idx, b_name_idx, b_rating_idx = 7, 9, 10 # For completed rounds, result must have digits/½ or be a forfeit (+/-) # For unplayed rounds, result is empty — that's valid if result_str: if not re.search(r'[\d½]', result_str) and '+ -' not in result_str and '- +' not in result_str: return None if not re.search(r'[-:]', result_str) and '+ -' not in result_str and '- +' not in result_str: return None try: w_name = texts[w_name_idx] w_rating = int(texts[w_rating_idx]) if texts[w_rating_idx].isdigit() else 0 w_pts = float(texts[w_pts_idx].replace(',', '.').replace('½', '.5')) b_pts = float(texts[b_pts_idx].replace(',', '.').replace('½', '.5')) b_name = texts[b_name_idx] b_rating = int(texts[b_rating_idx]) if texts[b_rating_idx].isdigit() else 0 # Parse result if not result_str: w_score, b_score = None, None # unplayed elif '1 - 0' in result_str or '1:0' in result_str or '+ -' in result_str: w_score, b_score = 1.0, 0.0 elif '0 - 1' in result_str or '0:1' in result_str or '- +' in result_str: w_score, b_score = 0.0, 1.0 else: w_score, b_score = 0.5, 0.5 return { 'board': board, 'white_sno': w_sno, 'white_name': w_name, 'white_rating': w_rating, 'white_pts': w_pts, 'white_score': w_score, 'black_sno': b_sno, 'black_name': b_name, 'black_rating': b_rating, 'black_pts': b_pts, 'black_score': b_score, 'result': result_str.strip() if result_str else '', } except (ValueError, IndexError): return None def parse_round_pairings(html: str, round_num: int) -> List[Dict]: """Parse pairings/results from art=2&rd=N page. Board number is derived from row position in the table. """ soup = BeautifulSoup(html, 'html.parser') games = [] tables = soup.find_all('table') for table in tables: rows = table.find_all('tr') board = 0 for row in rows: cells = row.find_all('td') texts = [c.get_text(strip=True) for c in cells] if len(texts) >= 10 and texts[0].isdigit(): board += 1 game = parse_game_row(texts, board) if game: game['round'] = round_num games.append(game) if games: break return games # ── Tournament meta ─────────────────────────────────────────── def extract_tournament_meta(html: str) -> dict: """Extract name, num_rounds, current_round.""" soup = BeautifulSoup(html, 'html.parser') info = {'name': '', 'num_rounds': 9, 'current_round': 1} h2 = soup.find('h2') if h2: info['name'] = h2.get_text(strip=True) text = soup.get_text() # Number of rounds m = re.search(r'Number of rounds\s*(\d+)', text) if m: info['num_rounds'] = int(m.group(1)) # Current round: "Положение после тура N" m = re.search(r'Положение после тура\s*(\d+)', text) if m: info['current_round'] = int(m.group(1)) else: # Try "Round X/Y" navigation m = re.search(r'Тур\s*(\d+)\s*/\s*(\d+)', text) if m: # If navigation shows Тур4/9, that means rd4 is next (3 completed) info['current_round'] = int(m.group(1)) - 1 return info # ── Full tournament fetch ───────────────────────────────────── def _normalize_url(url: str) -> str: base = re.sub(r'[&?]art=\d+', '', url) base = re.sub(r'[&?]rd=\d+', '', base) base = re.sub(r'[&?]turdet=\w+', '', base) base = re.sub(r'[&?]SNode=\w+', '', base) return base def fetch_tournament_quick(url: str) -> dict: """Lightweight change-detection fetch: only art=4 (1 HTTP request). Returns standings from the page directly (no art=2 cross-reference). Use to check if results_count or current_round changed before doing a full fetch_tournament() call. """ base_url = _normalize_url(url) try: html = fetch_url(base_url + '&art=4&turdet=YES') except Exception as e: raise RuntimeError(f'Не удалось загрузить турнир: {e}') meta = extract_tournament_meta(html) standings = parse_standings(html) current_round = meta['current_round'] if standings: max_results = max(len(s['results']) for s in standings) if max_results > current_round: current_round = max_results return { 'name': meta['name'], 'num_rounds': meta['num_rounds'], 'current_round': current_round, 'standings': standings, } def fetch_tournament(url: str) -> dict: """Fetch full tournament data from any chess-results.com URL. Detects current state automatically (any number of completed rounds). """ base_url = _normalize_url(url) # Fetch standings standings_url = base_url + '&art=4&turdet=YES' try: html = fetch_url(standings_url) except Exception as e: raise RuntimeError(f'Не удалось загрузить турнир: {e}') meta = extract_tournament_meta(html) standings = parse_standings(html) if not standings: raise RuntimeError('Не удалось распарсить таблицу положения') # Completed rounds: use page metadata ("Положение после тура N") current_round = meta['current_round'] # Sanity check against actual results max_results = max(len(s['results']) for s in standings) if current_round < 2 and max_results > current_round: current_round = max_results elif max_results < current_round: current_round = max_results # Starting list for ratings try: start_html = fetch_url(base_url + '&art=5&turdet=YES') start_players = parse_start_list(start_html) except Exception: start_players = {} # Match real SNo and rating from start list # Build name→sno lookup (normalized) name_to_sno = {} for sno, info in start_players.items(): name_to_sno[_normalize_name(info['name'])] = sno for s in standings: sname = _normalize_name(s['name']) # Exact name match: safe to trust both SNo and rating from start list. real_sno = name_to_sno.get(sname) if real_sno: s['starting_sno'] = real_sno s['rating'] = start_players[real_sno].get('rating', s.get('rating', 0)) else: # Fuzzy match is used ONLY for rating lookup — never overrides starting_sno. # (starting_sno from column 0 of art=4 standings is already the correct # starting number; overwriting it with a fuzzy-matched SNo corrupts # the opponent history and leads to wrong pairings.) sname_parts = set(sname.split()) best_sno, best_overlap = None, 0 for sl_name, sl_sno in name_to_sno.items(): sl_parts = set(sl_name.split()) overlap = len(sname_parts & sl_parts) if overlap > best_overlap: best_overlap, best_sno = overlap, sl_sno if best_sno and best_sno in start_players: s['rating'] = start_players[best_sno].get('rating', s.get('rating', 0)) if 'rating' not in s: s['rating'] = 0 # Keep original results from standings as backup (for byes/forfeits) for s in standings: s['_orig_results'] = list(s.get('results', [])) s['results'] = [] # Clear — will rebuild from pairings # Rebuild results from art=2 pairings (correct SNo, unlike standings cells) sno_to_standing = {s['starting_sno']: s for s in standings if s.get('starting_sno')} name_to_standing = {_normalize_name(s['name']): s for s in standings} for rd in range(1, current_round + 1): try: rd_html = fetch_url(f'{base_url}&art=2&rd={rd}&turdet=YES') games = parse_round_pairings(rd_html, rd) except Exception: continue for g in games: ws, bs = g['white_sno'], g['black_sno'] ws_result = g.get('white_score') bs_result = g.get('black_score') if ws_result is not None: # Completed game ws_val = float(ws_result) bs_val = float(bs_result) else: # Unplayed — skip continue # Resolve players by SNo or name (normalized) wp = sno_to_standing.get(ws) or name_to_standing.get(_normalize_name(g.get('white_name', ''))) bp = sno_to_standing.get(bs) or name_to_standing.get(_normalize_name(g.get('black_name', ''))) if wp: opp_sno = bs if bs != 0 else (bp.get('starting_sno') if bp else 0) existing = [r for r in wp.get('results', []) if r.get('round') != rd] existing.append({'opponent': opp_sno, 'color': 'w', 'score': ws_val, 'round': rd}) wp['results'] = existing if bp: opp_sno = ws if ws != 0 else (wp.get('starting_sno') if wp else 0) existing = [r for r in bp.get('results', []) if r.get('round') != rd] existing.append({'opponent': opp_sno, 'color': 'b', 'score': bs_val, 'round': rd}) bp['results'] = existing # For byes/forfeits not captured by pairings, use original standings results for s in standings: sno = s.get('starting_sno') if sno and len(s.get('results', [])) < rd: orig_results = s.get('_orig_results', []) if len(orig_results) >= rd: res = orig_results[rd - 1] res['round'] = rd s['results'] = list(s.get('results', [])) s['results'].append(res) # Final pass: fill any remaining gaps from _orig_results (forfeit losers, byes) for s in standings: existing_rounds = {r.get('round', 0) for r in s.get('results', [])} orig = s.get('_orig_results', []) for ri, res in enumerate(orig): rd_num = ri + 1 if rd_num <= current_round and rd_num not in existing_rounds: res_copy = dict(res) res_copy['round'] = rd_num if res_copy.get('opponent') and res_copy['opponent'] > 0: opp_sno = res_copy['opponent'] if opp_sno not in sno_to_standing: for s2 in standings: if s2['rank'] == opp_sno and s2.get('starting_sno'): res_copy['opponent'] = s2['starting_sno'] break s['results'].append(res_copy) for s in standings: s['points'] = sum(r.get('score', 0) for r in s.get('results', [])) return { 'name': meta['name'], 'num_rounds': meta['num_rounds'], 'current_round': current_round, 'players': start_players, 'standings': standings, } def parse_start_list(html: str) -> dict: """Parse starting list from art=0 or art=5 page. Two formats supported: 1. art=0 text: SNo Name FIDE_ID National_ID FED Rating Region 2. art=5 table: SNo, Name, FED, Rating columns """ soup = BeautifulSoup(html, 'html.parser') # Try art=0 text format first text = soup.get_text() players = {} # Pattern: SNo (1-3 digits), Full Name (3 words), two IDs, FED, Rating pattern = re.compile( r'(?:^|\s)(\d{1,3})\s+' r'([А-ЯЁ][а-яё]+\s+[А-ЯЁ][а-яё]+\s+[А-ЯЁ][а-яё]+)' r'\s+\d+\s+\d+\s+' r'([A-Z]{3})\s+' r'(\d{3,4})' ) for m in pattern.finditer(text): sno = int(m.group(1)) name = m.group(2).strip() fed = m.group(3) rating = int(m.group(4)) players[sno] = {'name': name, 'rating': rating, 'fed': fed} if len(players) >= 10: return players # Fallback: try art=5 table format tables = soup.find_all('table') for table in tables: rows = table.find_all('tr') data_rows = 0 for row in rows: cells = row.find_all('td') texts = [c.get_text(strip=True) for c in cells] if len(texts) >= 4 and texts[0].isdigit() and any(_is_player_name(t) for t in texts): data_rows += 1 if data_rows < 5: continue for row in rows: cells = row.find_all('td') texts = [c.get_text(strip=True) for c in cells] if len(texts) < 4 or not texts[0].isdigit(): continue sno = int(texts[0]) name = '' fed = '' rating = 0 for ci, t in enumerate(texts[1:], 1): if _is_player_name(t) and not name: name = t if ci + 1 < len(texts) and re.match(r'^[A-Z]{3}$', texts[ci + 1]): fed = texts[ci + 1] continue if t.isdigit() and len(t) >= 3 and not rating: rating = int(t) if name: players[sno] = {'name': name, 'rating': rating, 'fed': fed} if players: break return players