Channel features: subscribers, new-videos badges, activity-driven sync

- Store subscriber count from YouTube statistics and show it on channel page
- Sync 50 videos per channel with playlistItems pagination support
- Show per-channel and per-category new-videos counters (2-day window)
- Replace hourly videos sync with activity trigger (2h idle) and
  incremental backfill (hard cap 200 per channel)
- Clicking the sidebar new-videos count filters the category feed to
  recent videos only (new_only)
- Update agent-team docs: deploy after green checks
This commit is contained in:
vrubelroman 2026-09-17 21:07:33 +00:00
parent fde9a439df
commit e10df8dcbd
27 changed files with 1420 additions and 107 deletions

View file

@ -9,6 +9,12 @@ logger = logging.getLogger(__name__)
BATCH_SIZE = 50
# Hard safety cap for playlistItems pagination: at most this many pages
# (BATCH_SIZE items each, i.e. 500 ids) per playlist, so the pageToken loop
# can never run away. Raise both the cap and the caller's max_results
# together if more is ever needed.
MAX_PLAYLIST_PAGES = 10
class YouTubeQuotaExceeded(Exception):
pass
@ -102,26 +108,108 @@ def fetch_subscriptions(credentials: Credentials) -> list[dict]:
def fetch_playlist_video_ids(credentials: Credentials, playlist_id: str, max_results: int) -> list[str]:
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
params = {
"part": "contentDetails",
"playlistId": playlist_id,
"maxResults": min(max_results, 50),
}
response = client.get(f"{settings.youtube_api_base_url}/playlistItems", params=params, headers=_headers(credentials))
if response.status_code == 404:
return []
_raise_for_status(response)
data = response.json()
"""Low-level primitive: fetch up to `max_results` playlist item ids,
newest first, with no knowledge of what is already synced. The videos sync
uses fetch_playlist_video_ids_incremental instead; this stays as the plain
paginated helper (kept for tests and any future non-incremental callers)."""
video_ids: list[str] = []
page_token: str | None = None
pages_fetched = 0
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
while pages_fetched < MAX_PLAYLIST_PAGES and len(video_ids) < max_results:
params = {
"part": "contentDetails",
"playlistId": playlist_id,
"maxResults": min(max_results - len(video_ids), BATCH_SIZE),
}
if page_token:
params["pageToken"] = page_token
response = client.get(f"{settings.youtube_api_base_url}/playlistItems", params=params, headers=_headers(credentials))
if response.status_code == 404:
return []
_raise_for_status(response)
data = response.json()
for item in data.get("items", []):
video_id = item.get("contentDetails", {}).get("videoId")
if video_id:
video_ids.append(video_id)
pages_fetched += 1
page_token = data.get("nextPageToken")
if not page_token:
break
video_ids = []
for item in data.get("items", []):
video_id = item.get("contentDetails", {}).get("videoId")
if video_id:
video_ids.append(video_id)
return video_ids
def fetch_playlist_video_ids_incremental(
credentials: Credentials,
playlist_id: str,
known_ids: set[str],
max_results: int,
stop_threshold: int,
) -> list[str]:
"""Fetch up to `max_results` previously-unknown video ids from a playlist,
stopping pagination early once `stop_threshold` consecutive ids that are
already in `known_ids` are encountered. Uploads playlists are ordered
newest-first, so a long run of known ids means we reached history that was
already synced and there is nothing new further down.
`max_results` is a hard history-depth limit: at most the newest
`max_results` videos of the channel are ever considered, and everything
older than that is intentionally not backfilled (we do not mirror full
channel history). The early stop on known ids only saves pages on repeated
syncs within that window.
Returns only the unknown ids, in playlist order."""
new_ids: list[str] = []
seen_new: set[str] = set()
consecutive_known = 0
page_token: str | None = None
pages_fetched = 0
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
while pages_fetched < MAX_PLAYLIST_PAGES and len(new_ids) < max_results:
params = {
"part": "contentDetails",
"playlistId": playlist_id,
"maxResults": BATCH_SIZE,
}
if page_token:
params["pageToken"] = page_token
response = client.get(f"{settings.youtube_api_base_url}/playlistItems", params=params, headers=_headers(credentials))
if response.status_code == 404:
return new_ids
_raise_for_status(response)
data = response.json()
for item in data.get("items", []):
video_id = item.get("contentDetails", {}).get("videoId")
if not video_id:
continue
if video_id in known_ids or video_id in seen_new:
consecutive_known += 1
if consecutive_known >= stop_threshold:
return new_ids
else:
seen_new.add(video_id)
new_ids.append(video_id)
consecutive_known = 0
if len(new_ids) >= max_results:
return new_ids
pages_fetched += 1
page_token = data.get("nextPageToken")
if not page_token:
break
return new_ids
def fetch_videos_details(credentials: Credentials, video_ids: list[str]) -> list[dict]:
results: list[dict] = []
@ -171,14 +259,24 @@ def unsubscribe(credentials: Credentials, youtube_subscription_id: str) -> None:
_raise_for_status(response)
def fetch_uploads_playlists(credentials: Credentials, channel_ids: list[str]) -> dict[str, str]:
result: dict[str, str] = {}
def _parse_subscriber_count(statistics: dict) -> int | None:
raw = statistics.get("subscriberCount")
if raw is None:
return None
try:
return int(raw)
except (TypeError, ValueError):
return None
def fetch_uploads_playlists(credentials: Credentials, channel_ids: list[str]) -> dict[str, dict]:
result: dict[str, dict] = {}
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
for i in range(0, len(channel_ids), BATCH_SIZE):
batch = channel_ids[i : i + BATCH_SIZE]
params = {
"part": "snippet,contentDetails",
"part": "snippet,contentDetails,statistics",
"id": ",".join(batch),
"maxResults": BATCH_SIZE,
}
@ -191,7 +289,10 @@ def fetch_uploads_playlists(credentials: Credentials, channel_ids: list[str]) ->
uploads = (
item.get("contentDetails", {}).get("relatedPlaylists", {}).get("uploads")
)
if channel_id and uploads:
result[channel_id] = uploads
if channel_id:
result[channel_id] = {
"uploads_playlist_id": uploads or None,
"subscriber_count": _parse_subscriber_count(item.get("statistics", {})),
}
return result