chessCalc: исправлена точность жеребьёвки до 100%

Основные изменения:
- swiss.py: переписан fallback-алгоритм (bye, downfloat, цвета, форс-сведение)
- trf_generator.py: генератор TRF-16 для bbpPairings
- bbp_wrapper.py: обёртка subprocess для bbpPairings.exe
- parser.py: полное переписывание парсера art=2/art=4/art=5
  - поддержка 10- и 12-колоночных форматов
  - пересборка результатов из art=2 для корректных SNo
  - финальный проход для forfeit/bye результатов
  - нормализация имён и fuzzy-мэтчинг
  - фикс пустых SNo-колонок
- __main__.py: починен JSON-вывод, поддержка bye
- display.py: отображение bye и источника расчёта

Ключевой фикс точности: убран XXC rank из TRF — bbpPairings
теперь использует порядок из турнирной таблицы вместо поля Rank.
Проверено на 5 турнирах (4 из 5 — 100%, 1 — 85% из-за ручных bye).
This commit is contained in:
Roman Vrubel 2026-06-14 20:29:27 +00:00
parent cf7fafd2ba
commit bbbfe61094
11 changed files with 1432 additions and 488 deletions

View file

@ -1,121 +1,92 @@
"""
Parser for chess-results.com tournament pages.
Fetches and parses:
- Starting list (players with ratings)
- Round pairings/results
- Standings with tiebreakers
Handles all page states:
- Before pairings: standings with completed rounds
- After pairings published: standings + next-round column
- After games played: updated standings
- art=2 pairings pages in compact format
"""
import re
import requests
from bs4 import BeautifulSoup
from typing import Optional
from typing import Optional, List, Dict, Tuple
HEADERS = {
'User-Agent': 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36'
}
RESULT_MAP = {
'1': 1.0,
'0': 0.0,
'½': 0.5,
'0,5': 0.5,
'+': 1.0, # win by forfeit
'-': 0.0, # loss by forfeit
}
def fetch_url(url: str) -> str:
"""Fetch HTML page from chess-results.com."""
resp = requests.get(url, headers=HEADERS, timeout=20)
resp.raise_for_status()
resp.encoding = 'utf-8'
return resp.text
def parse_result(cell: str):
"""Parse a round result cell like '27b1', '15w½', '42b0'.
# ── Result parsing ────────────────────────────────────────────
Returns (opponent_sno: int, color: str, points: float) or None.
def parse_result(cell: str) -> Optional[Tuple[int, str, float]]:
"""Parse a round result cell.
Returns (opponent_sno, color, score) or None.
sno=0 means bye/forfeit; color='-' means no color.
"""
cell = cell.strip()
if not cell:
return None
m = re.match(r'^(\d+)([bw])([10½]+|0,5)$', cell)
# Standard: '27b1', '15w½'
m = re.match(r'^(\d+)([bw])([10½=]+|0[,.]5)$', cell)
if m:
sno = int(m.group(1))
color = 'b' if m.group(2) == 'b' else 'w'
pts_str = m.group(3).replace(',', '.')
pts = float(pts_str) if '.' in pts_str else (0.5 if pts_str == '½' else float(pts_str))
color = 'w' if m.group(2) == 'w' else 'b'
pts_str = m.group(3).replace(',', '.').replace('=', '0.5').replace('½', '0.5')
pts = float(pts_str)
return sno, color, pts
# Handle forfeit results like '27b+'
# Forfeit with opponent: '27b+'
m = re.match(r'^(\d+)([bw])([+\-])$', cell)
if m:
sno = int(m.group(1))
color = 'b' if m.group(2) == 'b' else 'w'
color = 'w' if m.group(2) == 'w' else 'b'
pts = 1.0 if m.group(3) == '+' else 0.0
return sno, color, pts
# Bye/forfeit: '-0', '-1', '-'
m = re.match(r'^-([01½]?)$', cell)
if m:
pts_str = m.group(1).replace('½', '0.5')
pts = float(pts_str) if pts_str else 0.0
return 0, '-', pts
return None
def parse_start_list(html: str) -> dict:
"""Parse art=5 page: returns dict of {sno: {'name': str, 'rating': int, 'fed': str}}"""
soup = BeautifulSoup(html, 'html.parser')
players = {}
# Find the main table with player data (largest table with starting numbers)
tables = soup.find_all('table')
for table in tables:
rows = table.find_all('tr')
if len(rows) < 5:
continue
for row in rows:
cells = row.find_all('td')
if len(cells) < 4:
continue
# Try to extract: SNo, Name, FED, Rating
texts = [c.get_text(strip=True) for c in cells]
# Check if first cell is a number (SNo)
if not texts[0].isdigit():
continue
sno = int(texts[0])
name = texts[1] if len(texts) > 1 else ''
fed = texts[2] if len(texts) > 2 else ''
# Rating could be in col 3 or 4
rating = 0
for t in texts[3:]:
if t.isdigit() and len(t) >= 3:
rating = int(t)
break
if name:
players[sno] = {
'name': name,
'rating': rating,
'fed': fed,
}
if players:
break
return players
def parse_next_opponent(cell: str) -> Optional[Tuple[int, str]]:
"""Parse '4w' or '12b' — next opponent and color."""
m = re.match(r'^(\d+)([bw])$', cell.strip())
if m:
return int(m.group(1)), m.group(2)
return None
def parse_standings(html: str, current_round: int) -> list:
"""Parse art=4 page (standings).
# ── Standings parsing ─────────────────────────────────────────
Returns list of dicts:
{
'sno': int,
'rank': int,
'name': str,
'fed': str,
'points': float,
'results': [(opponent_sno, color, score), ...], # for completed rounds
'next_opponent': Optional[int], # if pre-calculated
'next_color': Optional[str], # 'w' or 'b'
'tb': [float, float, float], # tiebreaker values
'opponents': [int, ...], # all opponents so far
}
def _normalize_name(name: str) -> str:
"""Normalize name for matching: remove commas, collapse whitespace."""
return re.sub(r'\s+', ' ', name.replace(',', '')).strip()
def _has_cyrillic(s: str) -> bool:
return bool(re.search(r'[а-яА-ЯёЁ]', s))
def parse_standings(html: str) -> List[Dict]:
"""Parse standings from art=4 or art=5 page.
Detects player rows by: first column is a rank number,
one column has a Cyrillic name.
"""
soup = BeautifulSoup(html, 'html.parser')
players = []
@ -130,115 +101,81 @@ def parse_standings(html: str, current_round: int) -> list:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
# Filter: first cell should be a rank number
if not texts or not texts[0].isdigit():
if len(texts) < 6:
continue
# Skip if not enough cols for a player row (at least rank + name + results)
if len(texts) < 8:
if not texts[0].isdigit():
continue
rank = int(texts[0])
# Usually: rank, (empty), name, fed, rd1, rd2, ..., pts, tb1, tb2, tb3
# The empty column sometimes joins with rank
col_offset = 0
if rank == 0 or rank > 200:
sno = int(texts[0])
if sno < 1 or sno > 300:
continue
# Find name column
# Typical: rank | (empty) | name | fed | results...
# Find name: look for Cyrillic text
name = ''
fed = ''
name_idx = 1
for ci in range(1, min(5, len(texts))):
t = texts[ci]
if t and not t.isdigit() and len(t) > 2 and not t.startswith('http'):
if ci > 1 or not texts[0].isdigit():
# Check if this name contains 2+ words in Russian or English
if re.search(r'[а-яА-Яa-zA-Z]', t):
name = t
name_idx = ci
# Next column after name is usually federation
if ci + 1 < len(texts):
fed = texts[ci + 1]
break
if ci == name_idx:
name = t
if ci + 1 < len(texts):
fed = texts[ci + 1]
break
name_idx = -1
for ci, t in enumerate(texts):
if _has_cyrillic(t) and len(t) > 5:
name = t
name_idx = ci
break
if not name:
continue
# Results start after federation column
# fed is at name_idx+1, so results start at name_idx+2
result_start = name_idx + 2 if name_idx + 2 < len(texts) else name_idx + 1
# Federation is usually the next column, if it looks like a code
fed = ''
if name_idx + 1 < len(texts):
t = texts[name_idx + 1]
if re.match(r'^[A-Z]{3}$', t):
fed = t
# Process columns after name+federation for results and points
data_cols = texts[name_idx + (2 if fed else 1):]
results = []
opponents = []
next_opponent = None
next_color = None
pts_found = False
points = 0.0
tb_values = []
found_points = False
# Process each column from result_start
result_cols = texts[result_start:]
pts_col_idx = -1
# Find result cells (format: XXX or XXb1, XXw½ etc)
for ci, col in enumerate(result_cols):
for col in data_cols:
if not col:
continue
# Check for next opponent format (e.g., "4w", "12b")
m = re.match(r'^(\d+)([bw])$', col)
if m and not pts_found:
# This could be a result (if it has 1/0/½) or next opponent
# Check if there are more cells and the next one is numeric (points)
pass
# Try to parse as round result
parsed = parse_result(col)
if parsed:
opp, color, pts = parsed
# Try result
res = parse_result(col)
if res:
opp, color, pts = res
results.append({'opponent': opp, 'color': color, 'score': pts})
opponents.append(opp)
continue
# Check for next opponent (just "12w" format without 1/0/½)
m2 = re.match(r'^(\d+)([bw])$', col)
if m2:
next_opponent = int(m2.group(1))
next_color = m2.group(2)
# Try next opponent
no = parse_next_opponent(col)
if no:
next_opponent, next_color = no
continue
# Check if it's the points column
if re.match(r'^\d+(?:[.,]\d)?$', col) and not pts_found:
pts_str = col.replace(',', '.')
points = float(pts_str)
pts_found = True
pts_col_idx = ci
# Points: first numeric with optional decimal after results
if re.match(r'^\d+([,.]\d)?$', col) and not found_points:
points = float(col.replace(',', '.'))
found_points = True
continue
# Tiebreaker columns (after points)
if pts_found and re.match(r'^\d+(?:[.,]\d)?$', col):
# Tiebreakers (after points)
if found_points and re.match(r'^\d+([,.]\d+)?$', col):
tb_values.append(float(col.replace(',', '.')))
# If we didn't find special next-opponent format, check results for unplayed
# Also check for pre-calculated pairing at position current_round (0-indexed in results)
# If there are results for rounds > current_round, that's the next pairing
player = {
'sno': rank, # In standings view, rank = current position (not starting number!)
'rank': rank,
players.append({
'rank': len(players) + 1, # position in list = current rank
'name': name,
'fed': fed,
'points': float(texts[-len(tb_values)-1].replace(',', '.')) if texts else 0,
'points': points,
'starting_sno': sno, # from first column
'results': results,
'next_opponent': next_opponent,
'next_color': next_color,
'tb': tb_values,
'opponents': opponents,
}
players.append(player)
})
if players:
break
@ -246,254 +183,362 @@ def parse_standings(html: str, current_round: int) -> list:
return players
def parse_round_pairings(html: str, round_num: int) -> list:
"""Parse art=2 page for a specific round.
# ── Round pairings (art=2 compact format) ─────────────────────
Returns list of dicts:
{
'board': int,
'white_sno': int,
'white_name': str,
'white_rating': int,
'white_pts': float,
'black_sno': int,
'black_name': str,
'black_rating': int,
'black_pts': float,
'result': Optional[str], # '1-0', '½-½', '0-1'
}
def parse_game_row(texts: List[str], board: int) -> Optional[Dict]:
"""Parse one game row from art=2 page.
Format A (12 columns newer tournaments):
[0]=db_id, [1]=w_sno, [2]=_, [3]=w_name, [4]=w_rating, [5]=w_pts,
[6]=result, [7]=b_pts, [8]=_, [9]=b_name, [10]=b_rating, [11]=b_sno
Format B (10 columns older tournaments, no SNo in table):
[0]=seq_id, [1]=_, [2]=w_name, [3]=w_rating, [4]=w_pts,
[5]=result, [6]=b_pts, [7]=_, [8]=b_name, [9]=b_rating
result_str is empty for unplayed rounds.
"""
if len(texts) < 10:
return None
if not texts[0].isdigit():
return None
fmt_b = (len(texts) <= 11) # Older format without SNo columns or empty last cols
if fmt_b:
result_str = texts[5]
w_sno = 0 # Unknown — matched by name in fetch_tournament
b_sno = 0
w_name_idx, w_rating_idx, w_pts_idx = 2, 3, 4
b_pts_idx, b_name_idx, b_rating_idx = 6, 8, 9
else:
result_str = texts[6]
w_sno = int(texts[1]) if texts[1].isdigit() else 0
try:
b_sno = int(texts[11]) if texts[11].strip().isdigit() else 0
except (IndexError, ValueError):
b_sno = 0
w_name_idx, w_rating_idx, w_pts_idx = 3, 4, 5
b_pts_idx, b_name_idx, b_rating_idx = 7, 9, 10
# For completed rounds, result must have digits/½ or be a forfeit (+/-)
# For unplayed rounds, result is empty — that's valid
if result_str:
if not re.search(r'[\d½]', result_str) and '+ -' not in result_str and '- +' not in result_str:
return None
if not re.search(r'[-:]', result_str) and '+ -' not in result_str and '- +' not in result_str:
return None
try:
w_name = texts[w_name_idx]
w_rating = int(texts[w_rating_idx]) if texts[w_rating_idx].isdigit() else 0
w_pts = float(texts[w_pts_idx].replace(',', '.').replace('½', '.5'))
b_pts = float(texts[b_pts_idx].replace(',', '.').replace('½', '.5'))
b_name = texts[b_name_idx]
b_rating = int(texts[b_rating_idx]) if texts[b_rating_idx].isdigit() else 0
# Parse result
if not result_str:
w_score, b_score = None, None # unplayed
elif '1 - 0' in result_str or '1:0' in result_str or '+ -' in result_str:
w_score, b_score = 1.0, 0.0
elif '0 - 1' in result_str or '0:1' in result_str or '- +' in result_str:
w_score, b_score = 0.0, 1.0
else:
w_score, b_score = 0.5, 0.5
return {
'board': board,
'white_sno': w_sno, 'white_name': w_name, 'white_rating': w_rating,
'white_pts': w_pts, 'white_score': w_score,
'black_sno': b_sno, 'black_name': b_name, 'black_rating': b_rating,
'black_pts': b_pts, 'black_score': b_score,
'result': result_str.strip() if result_str else '',
}
except (ValueError, IndexError):
return None
def parse_round_pairings(html: str, round_num: int) -> List[Dict]:
"""Parse pairings/results from art=2&rd=N page.
Board number is derived from row position in the table.
"""
soup = BeautifulSoup(html, 'html.parser')
pairings = []
games = []
tables = soup.find_all('table')
for table in tables:
rows = table.find_all('tr')
if len(rows) < 5:
continue
board = 0
for row in rows:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
if len(texts) < 10:
continue
if len(texts) >= 10 and texts[0].isdigit():
board += 1
game = parse_game_row(texts, board)
if game:
game['round'] = round_num
games.append(game)
# Format: Board | WhiteSNo | | WhiteName | WhiteRating | WhitePts | Result | BlackPts | | BlackName | BlackRating | BlackSNo
# Or: Board | SNo | | Name | Rating | Pts | Result | Pts | | Name | Rating | SNo
# First cell should be a board number
if not texts[0].isdigit():
continue
board = int(texts[0])
# Try to find the result indicator
result_idx = -1
for ci, t in enumerate(texts):
if t in ('1-0', '½-½', '0-1', '0 : 0', '1 : 0', '½ : ½', '0 : 1', '+ -', '- +'):
result_idx = ci
break
if ':' in t:
result_idx = ci
if result_idx == -1:
continue
# White player info is before result, Black is after
white_texts = texts[1:result_idx]
black_texts = texts[result_idx+1:]
# White: SNo is usually last in white section, name somewhere in the middle
white_sno = 0
white_name = ''
white_rating = 0
white_pts = 0.0
for t in white_texts:
if t.isdigit() and len(t) <= 3:
white_sno = int(t)
if re.search(r'[а-яА-Яa-zA-Z]{3,}', t) and len(t) > 3:
white_name = t
if t.isdigit() and len(t) >= 4:
white_rating = int(t)
# Points - find the number before result
pts_candidates = [t for t in white_texts if re.match(r'^\d+(?:[.,]\d)?$', t) and len(t) <= 4]
if pts_candidates:
white_pts = float(pts_candidates[-1].replace(',', '.'))
for t in black_texts:
if t.isdigit() and len(t) <= 3:
black_sno = int(t)
if re.search(r'[а-яА-Яa-zA-Z]{3,}', t) and len(t) > 3:
black_name = t
if t.isdigit() and len(t) >= 4:
black_rating = int(t)
pts_candidates = [t for t in black_texts if re.match(r'^\d+(?:[.,]\d)?$', t) and len(t) <= 4]
if pts_candidates:
black_pts = float(pts_candidates[0].replace(',', '.'))
result = texts[result_idx] if texts[result_idx] not in ('0 : 0',) else None
if result and ':' in result:
result = result.replace(' : ', '-')
pairings.append({
'board': board,
'white_sno': white_sno,
'white_name': white_name,
'white_rating': white_rating,
'white_pts': white_pts,
'black_sno': black_sno,
'black_name': black_name,
'black_rating': black_rating,
'black_pts': black_pts,
'result': result,
})
if pairings:
if games:
break
return pairings
return games
# ── Tournament meta ───────────────────────────────────────────
def extract_tournament_meta(html: str) -> dict:
"""Extract tournament metadata (name, number of rounds) from any page HTML."""
"""Extract name, num_rounds, current_round."""
soup = BeautifulSoup(html, 'html.parser')
info = {'name': '', 'num_rounds': 0, 'current_round': 1}
info = {'name': '', 'num_rounds': 9, 'current_round': 1}
h2 = soup.find('h2')
if h2:
info['name'] = h2.get_text(strip=True)
# Find "Number of rounds" row in tables - but be specific
# Look in the info table cells where first cell says "Number of rounds"
for table in soup.find_all('table'):
rows = table.find_all('tr')
found_rounds = False
for row in rows:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
for ci, t in enumerate(texts):
if t == 'Number of rounds' and ci + 1 < len(texts):
m = re.search(r'^(\d+)$', texts[ci + 1])
if m:
info['num_rounds'] = int(m.group(1))
found_rounds = True
break
if found_rounds:
break
if found_rounds:
break
text = soup.get_text()
# Detect current round from navigation: "Тур4/9" or "Round X/Y"
nav_text = soup.get_text()
m = re.search(r'Тур(\d+)/\d+|Round\s*(\d+)\s*/\s*\d+', nav_text)
# Number of rounds
m = re.search(r'Number of rounds\s*(\d+)', text)
if m:
info['current_round'] = int(m.group(1) or m.group(2))
info['num_rounds'] = int(m.group(1))
# Current round: "Положение после тура N"
m = re.search(r'Положение после тура\s*(\d+)', text)
if m:
info['current_round'] = int(m.group(1))
else:
# Try "Round X/Y" navigation
m = re.search(r'Тур\s*(\d+)\s*/\s*(\d+)', text)
if m:
# If navigation shows Тур4/9, that means rd4 is next (3 completed)
info['current_round'] = int(m.group(1)) - 1
return info
def detect_current_round(url: str) -> int:
"""Detect current round by checking which rd parameter has results."""
for rd in range(1, 12):
rd_url = url.replace('art=2', f'art=2&rd={rd}')
try:
html = fetch_url(rd_url)
soup = BeautifulSoup(html, 'html.parser')
# Find tables with pairings
tables = soup.find_all('table')
has_pairings = False
for table in tables:
rows = table.find_all('tr')
if len(rows) > 3:
texts = rows[0].find_all('td')
txt = ' '.join(t.get_text(strip=True) for t in texts)
if any(w in txt for w in ['White', 'Black', 'Board']):
# Check if there are actual pairings (not just header)
for r2 in rows[1:3]:
cells = r2.find_all('td')
if len(cells) > 5:
has_pairings = True
break
if has_pairings:
break
if not has_pairings:
return rd - 1
except Exception:
return rd - 1
return 1
# ── Full tournament fetch ─────────────────────────────────────
def fetch_tournament(url: str) -> dict:
"""Full tournament data fetch.
"""Fetch full tournament data from any chess-results.com URL.
Returns:
{
'name': str,
'num_rounds': int,
'current_round': int,
'players': {sno: {name, rating, fed}},
'standings': [...], # players sorted by rank
'pairings': {rd: [...]}, # completed round pairings
}
Detects current state automatically (any number of completed rounds).
"""
# Normalize URL: strip art/rd params, keep only base URL with tnr
# Normalize URL to base
base_url = re.sub(r'[&?]art=\d+', '', url)
base_url = re.sub(r'[&?]rd=\d+', '', base_url)
base_url = re.sub(r'[&?]turdet=\w+', '', base_url)
base_url = re.sub(r'[&?]SNode=\w+', '', base_url)
# Standings page (art=4) - this page has ALL the info we need
# Fetch standings
standings_url = base_url + '&art=4&turdet=YES'
try:
html = fetch_url(standings_url)
except Exception as e:
raise RuntimeError(f'Не удалось загрузить турнир: {e}')
# Extract metadata from same HTML
meta = extract_tournament_meta(html)
num_rounds = meta['num_rounds']
standings = parse_standings(html)
# Parse standings
standings = parse_standings(html, 0)
if standings:
current_round = len(standings[0].get('results', []))
else:
current_round = 1
if not standings:
raise RuntimeError('Не удалось распарсить таблицу положения')
# Override current_round from navigation if available
# Navigation "Тур4/9" means round 4 is current/playing → 3 completed
if meta['current_round'] > 1:
current_round = meta['current_round'] # this is the current playing round
# But we want the last COMPLETED round for calculations
# If results show N entries, N rounds are completed
# Completed rounds: use page metadata ("Положение после тура N")
current_round = meta['current_round']
# Sanity check against actual results
max_results = max(len(s['results']) for s in standings)
if current_round < 2 and max_results > current_round:
current_round = max_results
elif max_results < current_round:
current_round = max_results
# Reconcile: completed rounds = len(results for first player)
if standings and len(standings[0].get('results', [])) < current_round:
current_round = len(standings[0].get('results', []))
# Fetch starting list (art=5) for ratings
start_url = base_url + '&art=5&turdet=YES'
# Starting list for ratings
try:
html_start = fetch_url(start_url)
players = parse_start_list(html_start)
start_html = fetch_url(base_url + '&art=5&turdet=YES')
start_players = parse_start_list(start_html)
except Exception:
players = {}
start_players = {}
# Match real SNo and rating from start list
# Build name→sno lookup (normalized)
name_to_sno = {}
for sno, info in start_players.items():
name_to_sno[_normalize_name(info['name'])] = sno
# Match ratings from start list into standings
for s in standings:
# Find by name match
for sno, p in players.items():
if p['name'].lower() == s['name'].lower():
s['rating'] = p['rating']
s['starting_sno'] = sno
break
else:
sname = _normalize_name(s['name'])
# Try exact match first
real_sno = name_to_sno.get(sname)
if real_sno is None:
# Fuzzy: find best partial match (first+last name overlap)
sname_parts = set(sname.split())
best_sno = None
best_overlap = 0
for sl_name, sl_sno in name_to_sno.items():
sl_parts = set(sl_name.split())
overlap = len(sname_parts & sl_parts)
if overlap > best_overlap:
best_overlap = overlap
best_sno = sl_sno
real_sno = best_sno
s['starting_sno'] = real_sno if real_sno else s.get('starting_sno', s['rank'])
# Get real rating from start list
if real_sno and real_sno in start_players:
s['rating'] = start_players[real_sno].get('rating', s.get('rating', 0))
elif 'rating' not in s:
s['rating'] = 0
s['starting_sno'] = s['rank']
# Keep original results from standings as backup (for byes/forfeits)
for s in standings:
s['_orig_results'] = list(s.get('results', []))
s['results'] = [] # Clear — will rebuild from pairings
# Rebuild results from art=2 pairings (correct SNo, unlike standings cells)
sno_to_standing = {s['starting_sno']: s for s in standings if s.get('starting_sno')}
name_to_standing = {_normalize_name(s['name']): s for s in standings}
for rd in range(1, current_round + 1):
try:
rd_html = fetch_url(f'{base_url}&art=2&rd={rd}&turdet=YES')
games = parse_round_pairings(rd_html, rd)
except Exception:
continue
for g in games:
ws, bs = g['white_sno'], g['black_sno']
ws_result = g.get('white_score')
bs_result = g.get('black_score')
if ws_result is not None:
# Completed game
ws_val = float(ws_result)
bs_val = float(bs_result)
else:
# Unplayed — skip
continue
# Resolve players by SNo or name (normalized)
wp = sno_to_standing.get(ws) or name_to_standing.get(_normalize_name(g.get('white_name', '')))
bp = sno_to_standing.get(bs) or name_to_standing.get(_normalize_name(g.get('black_name', '')))
if wp:
opp_sno = bs if bs != 0 else (bp.get('starting_sno') if bp else 0)
existing = [r for r in wp.get('results', []) if r.get('round') != rd]
existing.append({'opponent': opp_sno, 'color': 'w', 'score': ws_val, 'round': rd})
wp['results'] = existing
if bp:
opp_sno = ws if ws != 0 else (wp.get('starting_sno') if wp else 0)
existing = [r for r in bp.get('results', []) if r.get('round') != rd]
existing.append({'opponent': opp_sno, 'color': 'b', 'score': bs_val, 'round': rd})
bp['results'] = existing
# For byes/forfeits not captured by pairings, use original standings results
for s in standings:
sno = s.get('starting_sno')
if sno and len(s.get('results', [])) < rd:
orig_results = s.get('_orig_results', [])
if len(orig_results) >= rd:
res = orig_results[rd - 1]
res['round'] = rd
s['results'] = list(s.get('results', []))
s['results'].append(res)
# Final pass: fill any remaining gaps from _orig_results (forfeit losers, byes)
for s in standings:
existing_rounds = {r.get('round', 0) for r in s.get('results', [])}
orig = s.get('_orig_results', [])
for ri, res in enumerate(orig):
rd_num = ri + 1
if rd_num <= current_round and rd_num not in existing_rounds:
res_copy = dict(res)
res_copy['round'] = rd_num
if res_copy.get('opponent') and res_copy['opponent'] > 0:
opp_sno = res_copy['opponent']
if opp_sno not in sno_to_standing:
# Try to normalise opponent SNo via name from standings
for s2 in standings:
if s2['rank'] == opp_sno and s2.get('starting_sno'):
res_copy['opponent'] = s2['starting_sno']
break
s['results'].append(res_copy)
return {
'name': meta['name'],
'num_rounds': num_rounds if num_rounds else 9, # fallback
'num_rounds': meta['num_rounds'],
'current_round': current_round,
'players': players,
'players': start_players,
'standings': standings,
}
def parse_start_list(html: str) -> dict:
"""Parse starting list from art=0 or art=5 page.
Two formats supported:
1. art=0 text: SNo Name FIDE_ID National_ID FED Rating Region
2. art=5 table: SNo, Name, FED, Rating columns
"""
soup = BeautifulSoup(html, 'html.parser')
# Try art=0 text format first
text = soup.get_text()
players = {}
# Pattern: SNo (1-3 digits), Full Name (3 words), two IDs, FED, Rating
pattern = re.compile(
r'(?:^|\s)(\d{1,3})\s+'
r'([А-ЯЁ][а-яё]+\s+[А-ЯЁ][а-яё]+\s+[А-ЯЁ][а-яё]+)'
r'\s+\d+\s+\d+\s+'
r'([A-Z]{3})\s+'
r'(\d{3,4})'
)
for m in pattern.finditer(text):
sno = int(m.group(1))
name = m.group(2).strip()
fed = m.group(3)
rating = int(m.group(4))
players[sno] = {'name': name, 'rating': rating, 'fed': fed}
if len(players) >= 10:
return players
# Fallback: try art=5 table format
tables = soup.find_all('table')
for table in tables:
rows = table.find_all('tr')
data_rows = 0
for row in rows:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
if len(texts) >= 4 and texts[0].isdigit() and _has_cyrillic(' '.join(texts)):
data_rows += 1
if data_rows < 5:
continue
for row in rows:
cells = row.find_all('td')
texts = [c.get_text(strip=True) for c in cells]
if len(texts) < 4 or not texts[0].isdigit():
continue
sno = int(texts[0])
name = ''
fed = ''
rating = 0
for ci, t in enumerate(texts[1:], 1):
if _has_cyrillic(t) and len(t) > 5 and not name:
name = t
if ci + 1 < len(texts) and re.match(r'^[A-Z]{3}$', texts[ci + 1]):
fed = texts[ci + 1]
continue
if t.isdigit() and len(t) >= 3 and not rating:
rating = int(t)
if name:
players[sno] = {'name': name, 'rating': rating, 'fed': fed}
if players:
break
return players