Implement Phases 1-5: skeleton, OAuth, categories, video sync/feed, playback

- FastAPI + PostgreSQL + Alembic + React/Vite skeleton, Docker Compose, healthcheck
- Google OAuth (single allowed account), encrypted refresh token storage
- Subscriptions sync with pagination, uploads playlist batch fetch
- Categories CRUD, many-to-many channel assignment, category filtering
- Video sync (playlistItems + videos.list batching), cached feed with cursor
  pagination, background scheduler (APScheduler)
- Video detail page with YouTube embed player
- SPA fallback routing, optimistic UI updates, client-side query caching

40 backend tests covering OAuth allow-list, sync idempotency, cascade deletes,
cursor pagination, and category filtering.
Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com>
This commit is contained in:
vrubelroman 2026-09-16 18:44:30 +00:00
commit 0ed20bb838
90 changed files with 8545 additions and 0 deletions

View file

View file

@ -0,0 +1,142 @@
import logging
from datetime import datetime, timezone
import httpx
from google.auth.transport.requests import Request as GoogleAuthRequest
from google.oauth2.credentials import Credentials
from google_auth_oauthlib.flow import Flow
from sqlalchemy.orm import Session
from app.config import settings
from app.core.crypto import decrypt_token, encrypt_token
from app.models.oauth_credentials import SINGLETON_ID, OAuthCredentials
logger = logging.getLogger(__name__)
SCOPES = [
"https://www.googleapis.com/auth/youtube.readonly",
"openid",
"https://www.googleapis.com/auth/userinfo.email",
"https://www.googleapis.com/auth/userinfo.profile",
]
AUTH_URI = "https://accounts.google.com/o/oauth2/auth"
TOKEN_URI = "https://oauth2.googleapis.com/token"
USERINFO_URI = "https://www.googleapis.com/oauth2/v3/userinfo"
REVOKE_URI = "https://oauth2.googleapis.com/revoke"
class OAuthNotConnected(Exception):
pass
def _client_config() -> dict:
return {
"web": {
"client_id": settings.google_client_id,
"client_secret": settings.google_client_secret,
"auth_uri": AUTH_URI,
"token_uri": TOKEN_URI,
"redirect_uris": [settings.google_redirect_uri],
}
}
def _build_flow(state: str | None = None) -> Flow:
return Flow.from_client_config(
_client_config(),
scopes=SCOPES,
state=state,
redirect_uri=settings.google_redirect_uri,
)
def build_authorization_url() -> tuple[str, str]:
flow = _build_flow()
auth_url, state = flow.authorization_url(
access_type="offline",
prompt="consent",
include_granted_scopes="true",
)
return auth_url, state
def exchange_code(code: str, state: str) -> Credentials:
flow = _build_flow(state=state)
flow.fetch_token(code=code)
return flow.credentials
def fetch_userinfo(access_token: str) -> dict:
response = httpx.get(
USERINFO_URI,
headers={"Authorization": f"Bearer {access_token}"},
timeout=settings.metube_request_timeout_seconds,
)
response.raise_for_status()
return response.json()
def revoke_token(token: str) -> None:
try:
httpx.post(REVOKE_URI, params={"token": token}, timeout=10)
except Exception:
logger.warning("Failed to revoke Google token", exc_info=True)
def store_credentials(db: Session, google_email: str, credentials: Credentials) -> None:
encrypted = encrypt_token(credentials.refresh_token)
expires_at = credentials.expiry
if expires_at is not None and expires_at.tzinfo is None:
expires_at = expires_at.replace(tzinfo=timezone.utc)
row = db.get(OAuthCredentials, SINGLETON_ID)
if row is None:
row = OAuthCredentials(id=SINGLETON_ID, google_email=google_email, encrypted_refresh_token=encrypted)
db.add(row)
else:
row.google_email = google_email
row.encrypted_refresh_token = encrypted
row.access_token_expires_at = expires_at
db.commit()
def clear_credentials(db: Session) -> None:
row = db.get(OAuthCredentials, SINGLETON_ID)
if row is not None:
db.delete(row)
db.commit()
def is_connected(db: Session) -> bool:
return db.get(OAuthCredentials, SINGLETON_ID) is not None
def get_connected_email(db: Session) -> str | None:
row = db.get(OAuthCredentials, SINGLETON_ID)
return row.google_email if row else None
def get_credentials(db: Session) -> Credentials:
row = db.get(OAuthCredentials, SINGLETON_ID)
if row is None:
raise OAuthNotConnected("Google account is not connected")
refresh_token = decrypt_token(row.encrypted_refresh_token)
credentials = Credentials(
token=None,
refresh_token=refresh_token,
token_uri=TOKEN_URI,
client_id=settings.google_client_id,
client_secret=settings.google_client_secret,
scopes=SCOPES,
)
credentials.refresh(GoogleAuthRequest())
expires_at = credentials.expiry
if expires_at is not None and expires_at.tzinfo is None:
expires_at = expires_at.replace(tzinfo=timezone.utc)
row.access_token_expires_at = expires_at
db.commit()
return credentials

View file

@ -0,0 +1,50 @@
import logging
from apscheduler.schedulers.background import BackgroundScheduler
from app.config import settings
from app.db import SessionLocal
from app.services import sync
logger = logging.getLogger(__name__)
def _run_subscriptions_sync() -> None:
db = SessionLocal()
try:
sync.sync_subscriptions(db)
except sync.SyncInProgress:
logger.info("Scheduled subscriptions sync skipped: already running")
except Exception:
logger.exception("Scheduled subscriptions sync failed")
finally:
db.close()
def _run_videos_sync() -> None:
db = SessionLocal()
try:
sync.sync_videos(db)
except sync.SyncInProgress:
logger.info("Scheduled videos sync skipped: already running")
except Exception:
logger.exception("Scheduled videos sync failed")
finally:
db.close()
def create_scheduler() -> BackgroundScheduler:
scheduler = BackgroundScheduler(timezone="UTC")
scheduler.add_job(
_run_subscriptions_sync,
"interval",
hours=settings.subscriptions_sync_interval_hours,
id="subscriptions_sync",
)
scheduler.add_job(
_run_videos_sync,
"interval",
minutes=settings.videos_sync_interval_minutes,
id="videos_sync",
)
return scheduler

View file

@ -0,0 +1,18 @@
from sqlalchemy.orm import Session
from app.models.app_settings import AppSetting
def get_setting(db: Session, key: str) -> str | None:
row = db.query(AppSetting).filter(AppSetting.key == key).one_or_none()
return row.value if row else None
def set_setting(db: Session, key: str, value: str) -> None:
row = db.query(AppSetting).filter(AppSetting.key == key).one_or_none()
if row is None:
row = AppSetting(key=key, value=value)
db.add(row)
else:
row.value = value
db.commit()

View file

@ -0,0 +1,251 @@
import json
import logging
import threading
from datetime import datetime, timezone
from sqlalchemy.orm import Session
from app.config import settings
from app.core.duration import parse_iso8601_duration
from app.models.channel import Channel
from app.models.video import Video
from app.services import google_oauth, youtube_client
from app.services.state import get_setting, set_setting
logger = logging.getLogger(__name__)
SUBSCRIPTIONS_SYNC_STATUS_KEY = "sync_subscriptions_status"
VIDEOS_SYNC_STATUS_KEY = "sync_videos_status"
_subscriptions_lock = threading.Lock()
_videos_lock = threading.Lock()
class SyncInProgress(Exception):
pass
def _now_iso() -> str:
return datetime.now(timezone.utc).isoformat()
def _parse_youtube_datetime(value: str) -> datetime:
return datetime.fromisoformat(value.replace("Z", "+00:00"))
def _get_status(db: Session, key: str, lock: threading.Lock) -> dict:
raw = get_setting(db, key)
status = json.loads(raw) if raw else {"status": "never_run"}
status["running"] = lock.locked()
return status
def _save_status(db: Session, key: str, status: dict) -> None:
set_setting(db, key, json.dumps(status))
def get_subscriptions_sync_status(db: Session) -> dict:
return _get_status(db, SUBSCRIPTIONS_SYNC_STATUS_KEY, _subscriptions_lock)
def get_videos_sync_status(db: Session) -> dict:
return _get_status(db, VIDEOS_SYNC_STATUS_KEY, _videos_lock)
def sync_subscriptions(db: Session) -> dict:
if not _subscriptions_lock.acquire(blocking=False):
raise SyncInProgress("Subscriptions sync already in progress")
started_at = _now_iso()
try:
_save_status(
db, SUBSCRIPTIONS_SYNC_STATUS_KEY,
{"status": "running", "started_at": started_at, "finished_at": None, "error": None},
)
credentials = google_oauth.get_credentials(db)
subscriptions = youtube_client.fetch_subscriptions(credentials)
seen_channel_ids = {sub["youtube_channel_id"] for sub in subscriptions}
existing = {c.youtube_channel_id: c for c in db.query(Channel).all()}
added = 0
updated = 0
now = datetime.now(timezone.utc)
for sub in subscriptions:
channel = existing.get(sub["youtube_channel_id"])
if channel is None:
channel = Channel(
youtube_channel_id=sub["youtube_channel_id"],
title=sub["title"],
description=sub["description"],
thumbnail_url=sub["thumbnail_url"],
subscribed=True,
last_synced_at=now,
)
db.add(channel)
existing[sub["youtube_channel_id"]] = channel
added += 1
else:
channel.title = sub["title"]
channel.description = sub["description"]
channel.thumbnail_url = sub["thumbnail_url"]
channel.subscribed = True
channel.last_synced_at = now
updated += 1
unsubscribed = 0
for channel_id, channel in existing.items():
if channel_id not in seen_channel_ids and channel.subscribed:
channel.subscribed = False
unsubscribed += 1
db.commit()
subscribed_ids = [c.youtube_channel_id for c in existing.values() if c.subscribed]
try:
uploads = youtube_client.fetch_uploads_playlists(credentials, subscribed_ids)
for channel_id, uploads_playlist_id in uploads.items():
channel = existing.get(channel_id)
if channel is not None:
channel.uploads_playlist_id = uploads_playlist_id
db.commit()
except Exception:
logger.exception("Failed to fetch uploads playlists during subscriptions sync")
result = {
"status": "completed",
"started_at": started_at,
"finished_at": _now_iso(),
"error": None,
"channels_added": added,
"channels_updated": updated,
"channels_unsubscribed": unsubscribed,
}
_save_status(db, SUBSCRIPTIONS_SYNC_STATUS_KEY, result)
logger.info(
"Subscriptions sync completed: added=%d updated=%d unsubscribed=%d", added, updated, unsubscribed
)
return result
except Exception as exc:
logger.exception("Subscriptions sync failed")
db.rollback()
result = {
"status": "failed",
"started_at": started_at,
"finished_at": _now_iso(),
"error": str(exc),
}
_save_status(db, SUBSCRIPTIONS_SYNC_STATUS_KEY, result)
raise
finally:
_subscriptions_lock.release()
def sync_videos(db: Session) -> dict:
if not _videos_lock.acquire(blocking=False):
raise SyncInProgress("Videos sync already in progress")
started_at = _now_iso()
try:
_save_status(
db, VIDEOS_SYNC_STATUS_KEY,
{"status": "running", "started_at": started_at, "finished_at": None, "error": None},
)
credentials = google_oauth.get_credentials(db)
channels = (
db.query(Channel)
.filter(Channel.subscribed.is_(True), Channel.uploads_playlist_id.isnot(None))
.all()
)
channel_by_youtube_id = {c.youtube_channel_id: c for c in channels}
candidate_video_ids: set[str] = set()
for channel in channels:
try:
video_ids = youtube_client.fetch_playlist_video_ids(
credentials, channel.uploads_playlist_id, settings.videos_per_channel_sync
)
candidate_video_ids.update(video_ids)
except Exception:
logger.exception("Failed to fetch playlist items for channel %s", channel.youtube_channel_id)
details = youtube_client.fetch_videos_details(credentials, list(candidate_video_ids))
existing = {v.youtube_video_id: v for v in db.query(Video).all()}
added = 0
updated = 0
skipped = 0
for item in details:
channel = channel_by_youtube_id.get(item["youtube_channel_id"])
if channel is None:
skipped += 1
continue
published_at = _parse_youtube_datetime(item["published_at"]) if item["published_at"] else None
if published_at is None:
skipped += 1
continue
duration_seconds = parse_iso8601_duration(item["duration_iso8601"])
youtube_url = f"https://www.youtube.com/watch?v={item['youtube_video_id']}"
video = existing.get(item["youtube_video_id"])
if video is None:
video = Video(
youtube_video_id=item["youtube_video_id"],
channel_id=channel.id,
title=item["title"],
description=item["description"],
thumbnail_url=item["thumbnail_url"],
published_at=published_at,
duration_seconds=duration_seconds,
youtube_url=youtube_url,
)
db.add(video)
existing[item["youtube_video_id"]] = video
added += 1
else:
video.channel_id = channel.id
video.title = item["title"]
video.description = item["description"]
video.thumbnail_url = item["thumbnail_url"]
video.published_at = published_at
video.duration_seconds = duration_seconds
video.youtube_url = youtube_url
updated += 1
db.commit()
result = {
"status": "completed",
"started_at": started_at,
"finished_at": _now_iso(),
"error": None,
"videos_added": added,
"videos_updated": updated,
"videos_skipped": skipped,
"channels_checked": len(channels),
}
_save_status(db, VIDEOS_SYNC_STATUS_KEY, result)
logger.info("Videos sync completed: added=%d updated=%d skipped=%d", added, updated, skipped)
return result
except Exception as exc:
logger.exception("Videos sync failed")
db.rollback()
result = {
"status": "failed",
"started_at": started_at,
"finished_at": _now_iso(),
"error": str(exc),
}
_save_status(db, VIDEOS_SYNC_STATUS_KEY, result)
raise
finally:
_videos_lock.release()

View file

@ -0,0 +1,46 @@
from sqlalchemy import select
from sqlalchemy.orm import Session
from app.models.category import Category
from app.models.channel import Channel
from app.models.channel_category import channel_categories
from app.models.video import Video
def channel_categories_map(db: Session, channel_ids: list[int]) -> dict[int, list[dict]]:
if not channel_ids:
return {}
rows = db.execute(
select(channel_categories.c.channel_id, Category.id, Category.name)
.join(Category, Category.id == channel_categories.c.category_id)
.where(channel_categories.c.channel_id.in_(channel_ids))
).all()
result: dict[int, list[dict]] = {}
for channel_id, category_id, category_name in rows:
result.setdefault(channel_id, []).append({"id": category_id, "name": category_name})
return result
def serialize_video(video: Video, channel: Channel, categories: list[dict]) -> dict:
return {
"youtube_video_id": video.youtube_video_id,
"title": video.title,
"description": video.description,
"channel": {
"id": channel.id,
"youtube_channel_id": channel.youtube_channel_id,
"title": channel.title,
"thumbnail_url": channel.thumbnail_url,
},
"thumbnail_url": video.thumbnail_url,
"published_at": video.published_at,
"duration_seconds": video.duration_seconds,
"youtube_url": video.youtube_url,
"categories": categories,
"local": {
"available": False,
"status": "not_downloaded",
"progress_percent": None,
"media_url": None,
},
}

View file

@ -0,0 +1,168 @@
import logging
import httpx
from google.oauth2.credentials import Credentials
from app.config import settings
logger = logging.getLogger(__name__)
API_BASE = "https://www.googleapis.com/youtube/v3"
BATCH_SIZE = 50
class YouTubeQuotaExceeded(Exception):
pass
class YouTubeAPIError(Exception):
pass
def _headers(credentials: Credentials) -> dict:
return {"Authorization": f"Bearer {credentials.token}"}
def _raise_for_status(response: httpx.Response) -> None:
if response.status_code == 200:
return
try:
payload = response.json()
reason = payload.get("error", {}).get("errors", [{}])[0].get("reason", "")
message = payload.get("error", {}).get("message", response.text)
except Exception:
reason = ""
message = response.text
if response.status_code == 403 and reason in ("quotaExceeded", "dailyLimitExceeded", "rateLimitExceeded"):
raise YouTubeQuotaExceeded(message)
logger.error("YouTube API error %s: %s", response.status_code, message)
raise YouTubeAPIError(f"{response.status_code}: {message}")
def fetch_subscriptions(credentials: Credentials) -> list[dict]:
subscriptions: list[dict] = []
page_token: str | None = None
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
while True:
params = {
"part": "snippet,contentDetails",
"mine": "true",
"maxResults": 50,
}
if page_token:
params["pageToken"] = page_token
response = client.get(f"{API_BASE}/subscriptions", params=params, headers=_headers(credentials))
_raise_for_status(response)
data = response.json()
for item in data.get("items", []):
snippet = item.get("snippet", {})
resource_id = snippet.get("resourceId", {})
channel_id = resource_id.get("channelId")
if not channel_id:
continue
thumbnails = snippet.get("thumbnails", {})
thumbnail = (
thumbnails.get("high") or thumbnails.get("medium") or thumbnails.get("default") or {}
).get("url")
subscriptions.append(
{
"youtube_channel_id": channel_id,
"title": snippet.get("title", ""),
"description": snippet.get("description", ""),
"thumbnail_url": thumbnail,
}
)
page_token = data.get("nextPageToken")
if not page_token:
break
return subscriptions
def fetch_playlist_video_ids(credentials: Credentials, playlist_id: str, max_results: int) -> list[str]:
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
params = {
"part": "contentDetails",
"playlistId": playlist_id,
"maxResults": min(max_results, 50),
}
response = client.get(f"{API_BASE}/playlistItems", params=params, headers=_headers(credentials))
if response.status_code == 404:
return []
_raise_for_status(response)
data = response.json()
video_ids = []
for item in data.get("items", []):
video_id = item.get("contentDetails", {}).get("videoId")
if video_id:
video_ids.append(video_id)
return video_ids
def fetch_videos_details(credentials: Credentials, video_ids: list[str]) -> list[dict]:
results: list[dict] = []
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
for i in range(0, len(video_ids), BATCH_SIZE):
batch = video_ids[i : i + BATCH_SIZE]
params = {
"part": "snippet,contentDetails,status",
"id": ",".join(batch),
"maxResults": BATCH_SIZE,
}
response = client.get(f"{API_BASE}/videos", params=params, headers=_headers(credentials))
_raise_for_status(response)
data = response.json()
for item in data.get("items", []):
snippet = item.get("snippet", {})
thumbnails = snippet.get("thumbnails", {})
thumbnail = (
thumbnails.get("high") or thumbnails.get("medium") or thumbnails.get("default") or {}
).get("url")
results.append(
{
"youtube_video_id": item.get("id"),
"youtube_channel_id": snippet.get("channelId"),
"title": snippet.get("title", ""),
"description": snippet.get("description", ""),
"thumbnail_url": thumbnail,
"published_at": snippet.get("publishedAt"),
"duration_iso8601": item.get("contentDetails", {}).get("duration"),
}
)
return results
def fetch_uploads_playlists(credentials: Credentials, channel_ids: list[str]) -> dict[str, str]:
result: dict[str, str] = {}
with httpx.Client(timeout=settings.metube_request_timeout_seconds) as client:
for i in range(0, len(channel_ids), BATCH_SIZE):
batch = channel_ids[i : i + BATCH_SIZE]
params = {
"part": "snippet,contentDetails",
"id": ",".join(batch),
"maxResults": BATCH_SIZE,
}
response = client.get(f"{API_BASE}/channels", params=params, headers=_headers(credentials))
_raise_for_status(response)
data = response.json()
for item in data.get("items", []):
channel_id = item.get("id")
uploads = (
item.get("contentDetails", {}).get("relatedPlaylists", {}).get("uploads")
)
if channel_id and uploads:
result[channel_id] = uploads
return result