myYouTube/tests/test_sync.py
vrubelroman e10df8dcbd Channel features: subscribers, new-videos badges, activity-driven sync
- Store subscriber count from YouTube statistics and show it on channel page
- Sync 50 videos per channel with playlistItems pagination support
- Show per-channel and per-category new-videos counters (2-day window)
- Replace hourly videos sync with activity trigger (2h idle) and
  incremental backfill (hard cap 200 per channel)
- Clicking the sidebar new-videos count filters the category feed to
  recent videos only (new_only)
- Update agent-team docs: deploy after green checks
2026-09-17 21:07:33 +00:00

569 lines
19 KiB
Python

from datetime import datetime, timezone
from app.config import settings
from app.models.channel import Channel
from app.models.video import Video
from app.services import sync
def _fake_credentials():
return object()
def test_sync_subscriptions_idempotent_and_unsubscribes(monkeypatch, db_session):
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _fake_credentials())
monkeypatch.setattr(
sync.youtube_client,
"fetch_subscriptions",
lambda creds: [
{
"youtube_channel_id": "chanA",
"youtube_subscription_id": "subA",
"title": "Channel A",
"description": "d",
"thumbnail_url": "t",
},
{
"youtube_channel_id": "chanB",
"youtube_subscription_id": "subB",
"title": "Channel B",
"description": "d",
"thumbnail_url": "t",
},
],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_uploads_playlists",
lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": 12345} for cid in ids},
)
result = sync.sync_subscriptions(db_session)
assert result["status"] == "completed"
assert result["channels_added"] == 2
assert result["channels_unsubscribed"] == 0
channels = db_session.query(Channel).order_by(Channel.youtube_channel_id).all()
assert [c.youtube_channel_id for c in channels] == ["chanA", "chanB"]
assert all(c.subscribed for c in channels)
assert channels[0].uploads_playlist_id == "UUchanA"
assert channels[0].subscriber_count == 12345
# Second sync: chanA disappears from subscriptions, chanC appears.
monkeypatch.setattr(
sync.youtube_client,
"fetch_subscriptions",
lambda creds: [
{
"youtube_channel_id": "chanB",
"youtube_subscription_id": "subB",
"title": "Channel B",
"description": "d",
"thumbnail_url": "t",
},
{
"youtube_channel_id": "chanC",
"youtube_subscription_id": "subC",
"title": "Channel C",
"description": "d",
"thumbnail_url": "t",
},
],
)
result2 = sync.sync_subscriptions(db_session)
assert result2["channels_added"] == 1
assert result2["channels_updated"] == 1
assert result2["channels_unsubscribed"] == 1
channels = {c.youtube_channel_id: c for c in db_session.query(Channel).all()}
assert len(channels) == 3
assert channels["chanA"].subscribed is False
assert channels["chanB"].subscribed is True
assert channels["chanC"].subscribed is True
def test_sync_subscriptions_skips_none_subscriber_count(monkeypatch, db_session):
channel = Channel(
youtube_channel_id="chanA",
youtube_subscription_id="subA",
title="Channel A",
subscribed=True,
subscriber_count=12345,
)
db_session.add(channel)
db_session.commit()
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _fake_credentials())
monkeypatch.setattr(
sync.youtube_client,
"fetch_subscriptions",
lambda creds: [
{
"youtube_channel_id": "chanA",
"youtube_subscription_id": "subA",
"title": "Channel A",
"description": "d",
"thumbnail_url": "t",
},
],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_uploads_playlists",
lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": None} for cid in ids},
)
result = sync.sync_subscriptions(db_session)
assert result["status"] == "completed"
db_session.refresh(channel)
assert channel.uploads_playlist_id == "UUchanA"
# None in the API response must not overwrite the previously stored count.
assert channel.subscriber_count == 12345
def test_sync_in_progress_raises(monkeypatch, db_session):
sync._subscriptions_lock.acquire()
try:
try:
sync.sync_subscriptions(db_session)
assert False, "expected SyncInProgress"
except sync.SyncInProgress:
pass
finally:
sync._subscriptions_lock.release()
def _seed_channel(db_session, youtube_channel_id="chanA", uploads_playlist_id="UUchanA"):
channel = Channel(
youtube_channel_id=youtube_channel_id,
title="Channel",
subscribed=True,
uploads_playlist_id=uploads_playlist_id,
)
db_session.add(channel)
db_session.commit()
db_session.refresh(channel)
return channel
def test_sync_videos_adds_new_videos(monkeypatch, db_session):
channel = _seed_channel(db_session)
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT5M",
}
],
)
result = sync.sync_videos(db_session)
assert result["status"] == "completed"
assert result["videos_added"] == 1
assert result["channels_checked"] == 1
video = db_session.query(Video).filter_by(youtube_video_id="vid1").one()
assert video.title == "Video One"
assert video.duration_seconds == 300
assert video.youtube_url == "https://www.youtube.com/watch?v=vid1"
def test_sync_videos_second_run_is_idempotent_and_fetches_no_details(monkeypatch, db_session):
channel = _seed_channel(db_session)
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT5M",
}
],
)
first = sync.sync_videos(db_session)
assert first["videos_added"] == 1
# Second run: vid1 is now known, so the incremental fetch reports nothing
# new and no details are requested at all (existing videos are not
# metadata-refreshed by design).
details_calls = []
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: [],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids)) or [],
)
second = sync.sync_videos(db_session)
assert second["videos_added"] == 0
assert second["videos_updated"] == 0
assert details_calls == [[]]
assert db_session.query(Video).count() == 1
def test_sync_videos_passes_known_ids_and_backfill_settings(monkeypatch, db_session):
channel = _seed_channel(db_session)
db_session.add_all(
[
Video(
youtube_video_id="known1",
channel_id=channel.id,
title="Known 1",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url="https://www.youtube.com/watch?v=known1",
),
Video(
youtube_video_id="known2",
channel_id=channel.id,
title="Known 2",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url="https://www.youtube.com/watch?v=known2",
),
]
)
db_session.commit()
captured = {}
def fake_incremental(creds, playlist_id, known_ids, max_results, stop_threshold):
captured.update(known_ids=known_ids, max_results=max_results, stop_threshold=stop_threshold)
return ["vid1"]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(sync.youtube_client, "fetch_playlist_video_ids_incremental", fake_incremental)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT5M",
}
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == 1
assert captured["known_ids"] == {"known1", "known2"}
assert captured["max_results"] == settings.videos_backfill_cap
assert captured["stop_threshold"] == settings.videos_known_stop_threshold
def test_sync_videos_backfill_cap_ingests_all_candidates(monkeypatch, db_session):
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
new_ids = [f"vid{i}" for i in range(cap)]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: new_ids,
)
details_calls = []
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == cap
assert result["videos_updated"] == 0
assert set(details_calls[0]) == set(new_ids)
assert db_session.query(Video).count() == cap
def test_sync_videos_skips_unknown_channel(monkeypatch, db_session):
_seed_channel(db_session, "chanA")
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: [],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vidX",
"youtube_channel_id": "unknown-channel",
"title": "Orphan",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
],
)
result = sync.sync_videos(db_session)
assert result["videos_skipped"] == 1
assert db_session.query(Video).count() == 0
class _FakeResponse:
def __init__(self, payload, status_code=200):
self.status_code = status_code
self._payload = payload
self.text = str(payload)
def json(self):
return self._payload
class _FakeCredentials:
token = "fake-token"
class _FakeClient:
"""Serves canned playlistItems pages through the real incremental fetcher."""
def __init__(self, pages):
self._pages = list(pages)
self.requests = []
def __enter__(self):
return self
def __exit__(self, *args):
return False
def get(self, url, params=None, headers=None):
self.requests.append({"url": url, "params": params})
return self._pages.pop(0)
def _page(items, next_token=None):
return _FakeResponse(
{
"items": [{"contentDetails": {"videoId": vid}} for vid in items],
**({"nextPageToken": next_token} if next_token else {}),
}
)
def test_sync_videos_stops_pagination_on_known_run(monkeypatch, db_session):
"""A channel with many known videos in a row after the new ones: the
incremental fetch must stop paginating once the known-run threshold is hit
and only the new ids must get details."""
channel = _seed_channel(db_session)
known = [f"known{i}" for i in range(60)]
db_session.add_all(
[
Video(
youtube_video_id=vid,
channel_id=channel.id,
title=f"Known {vid}",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url=f"https://www.youtube.com/watch?v={vid}",
)
for vid in known
]
)
db_session.commit()
# Page 1: two new videos then 48 known; page 2 continues with known ids,
# so the run crosses 50 on the second page and pagination must stop there
# (a third page exists and must never be requested).
client = _FakeClient(
[
_page(["new1", "new2", *known[:48]], next_token="page2"),
_page(known[48:58], next_token="page3"),
_page(["never-seen"]),
]
)
details_calls = []
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == 2
assert len(client.requests) == 2 # stopped on page 2, page 3 never fetched
assert set(details_calls[0]) == {"new1", "new2"}
assert db_session.query(Video).count() == 62
def test_sync_videos_stops_at_backfill_cap(monkeypatch, db_session):
"""More new videos than the cap: pagination stops once the cap is reached
and details are fetched for exactly the capped candidates."""
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
pages = [_page([f"vid{page * 50 + i}" for i in range(50)], next_token=f"p{page + 2}") for page in range(6)]
client = _FakeClient(pages)
details_calls = []
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert len(client.requests) == 4 # 4 pages x 50 = cap reached
assert result["videos_added"] == cap
assert len(details_calls[0]) == cap
assert db_session.query(Video).count() == cap
def test_sync_videos_cap_is_hard_history_depth_limit(monkeypatch, db_session):
"""videos_backfill_cap is an intentional history-depth limit, not a
per-sync batch: a 250-video channel ingests only the newest 200 videos
ever; a later sync without new videos stops after one page of known ids
and adds nothing; a new video on top of the playlist is picked up while
the beyond-the-cap history never surfaces."""
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
def _details(creds, ids):
return [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client, "fetch_videos_details", _details)
# Uploads playlist, newest first: vid249 .. vid0.
playlist = [f"vid{i}" for i in range(249, -1, -1)]
def _run(ids):
client = _FakeClient(
[_page(ids[i : i + 50], next_token=f"page{i // 50}") for i in range(0, len(ids), 50)]
)
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
return sync.sync_videos(db_session), client
# First sync: 250 videos, only the newest 200 fit under the cap.
result1, client1 = _run(playlist)
assert result1["videos_added"] == cap
assert db_session.query(Video).count() == cap
assert len(client1.requests) == 4 # 4 pages x 50 = cap, older pages untouched
synced = {v.youtube_video_id for v in db_session.query(Video).all()}
assert synced == {f"vid{i}" for i in range(50, 250)}
assert "vid49" not in synced # older than the cap: never backfilled by design
# Second sync, nothing new: one page of 50 known ids in a row stops it.
result2, client2 = _run(playlist)
assert result2["videos_added"] == 0
assert len(client2.requests) == 1
assert db_session.query(Video).count() == cap
# Third sync, one new video on top: only that one is added, the
# beyond-the-cap history still does not surface.
result3, client3 = _run(["vid250", *playlist])
assert result3["videos_added"] == 1
assert len(client3.requests) == 2 # page 2 needed to confirm 50 known in a row
synced_after = {v.youtube_video_id for v in db_session.query(Video).all()}
assert synced_after == {f"vid{i}" for i in range(50, 251)}
assert "vid49" not in synced_after
def test_is_videos_sync_running_reflects_lock():
sync._videos_lock.acquire()
try:
assert sync.is_videos_sync_running() is True
finally:
sync._videos_lock.release()
assert sync.is_videos_sync_running() is False