from datetime import datetime, timezone from app.config import settings from app.models.channel import Channel from app.models.video import Video from app.services import sync def _fake_credentials(): return object() def test_sync_subscriptions_idempotent_and_unsubscribes(monkeypatch, db_session): monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _fake_credentials()) monkeypatch.setattr( sync.youtube_client, "fetch_subscriptions", lambda creds: [ { "youtube_channel_id": "chanA", "youtube_subscription_id": "subA", "title": "Channel A", "description": "d", "thumbnail_url": "t", }, { "youtube_channel_id": "chanB", "youtube_subscription_id": "subB", "title": "Channel B", "description": "d", "thumbnail_url": "t", }, ], ) monkeypatch.setattr( sync.youtube_client, "fetch_uploads_playlists", lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": 12345} for cid in ids}, ) result = sync.sync_subscriptions(db_session) assert result["status"] == "completed" assert result["channels_added"] == 2 assert result["channels_unsubscribed"] == 0 channels = db_session.query(Channel).order_by(Channel.youtube_channel_id).all() assert [c.youtube_channel_id for c in channels] == ["chanA", "chanB"] assert all(c.subscribed for c in channels) assert channels[0].uploads_playlist_id == "UUchanA" assert channels[0].subscriber_count == 12345 # Second sync: chanA disappears from subscriptions, chanC appears. monkeypatch.setattr( sync.youtube_client, "fetch_subscriptions", lambda creds: [ { "youtube_channel_id": "chanB", "youtube_subscription_id": "subB", "title": "Channel B", "description": "d", "thumbnail_url": "t", }, { "youtube_channel_id": "chanC", "youtube_subscription_id": "subC", "title": "Channel C", "description": "d", "thumbnail_url": "t", }, ], ) result2 = sync.sync_subscriptions(db_session) assert result2["channels_added"] == 1 assert result2["channels_updated"] == 1 assert result2["channels_unsubscribed"] == 1 channels = {c.youtube_channel_id: c for c in db_session.query(Channel).all()} assert len(channels) == 3 assert channels["chanA"].subscribed is False assert channels["chanB"].subscribed is True assert channels["chanC"].subscribed is True def test_sync_subscriptions_skips_none_subscriber_count(monkeypatch, db_session): channel = Channel( youtube_channel_id="chanA", youtube_subscription_id="subA", title="Channel A", subscribed=True, subscriber_count=12345, ) db_session.add(channel) db_session.commit() monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _fake_credentials()) monkeypatch.setattr( sync.youtube_client, "fetch_subscriptions", lambda creds: [ { "youtube_channel_id": "chanA", "youtube_subscription_id": "subA", "title": "Channel A", "description": "d", "thumbnail_url": "t", }, ], ) monkeypatch.setattr( sync.youtube_client, "fetch_uploads_playlists", lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": None} for cid in ids}, ) result = sync.sync_subscriptions(db_session) assert result["status"] == "completed" db_session.refresh(channel) assert channel.uploads_playlist_id == "UUchanA" # None in the API response must not overwrite the previously stored count. assert channel.subscriber_count == 12345 def test_sync_in_progress_raises(monkeypatch, db_session): sync._subscriptions_lock.acquire() try: try: sync.sync_subscriptions(db_session) assert False, "expected SyncInProgress" except sync.SyncInProgress: pass finally: sync._subscriptions_lock.release() def _seed_channel(db_session, youtube_channel_id="chanA", uploads_playlist_id="UUchanA"): channel = Channel( youtube_channel_id=youtube_channel_id, title="Channel", subscribed=True, uploads_playlist_id=uploads_playlist_id, ) db_session.add(channel) db_session.commit() db_session.refresh(channel) return channel def test_sync_videos_adds_new_videos(monkeypatch, db_session): channel = _seed_channel(db_session) monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object()) monkeypatch.setattr( sync.youtube_client, "fetch_playlist_video_ids_incremental", lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"], ) monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: [ { "youtube_video_id": "vid1", "youtube_channel_id": channel.youtube_channel_id, "title": "Video One", "description": "d", "thumbnail_url": "t", "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": "PT5M", } ], ) result = sync.sync_videos(db_session) assert result["status"] == "completed" assert result["videos_added"] == 1 assert result["channels_checked"] == 1 video = db_session.query(Video).filter_by(youtube_video_id="vid1").one() assert video.title == "Video One" assert video.duration_seconds == 300 assert video.youtube_url == "https://www.youtube.com/watch?v=vid1" def test_sync_videos_second_run_is_idempotent_and_fetches_no_details(monkeypatch, db_session): channel = _seed_channel(db_session) monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object()) monkeypatch.setattr( sync.youtube_client, "fetch_playlist_video_ids_incremental", lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"], ) monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: [ { "youtube_video_id": "vid1", "youtube_channel_id": channel.youtube_channel_id, "title": "Video One", "description": "d", "thumbnail_url": "t", "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": "PT5M", } ], ) first = sync.sync_videos(db_session) assert first["videos_added"] == 1 # Second run: vid1 is now known, so the incremental fetch reports nothing # new and no details are requested at all (existing videos are not # metadata-refreshed by design). details_calls = [] monkeypatch.setattr( sync.youtube_client, "fetch_playlist_video_ids_incremental", lambda creds, playlist_id, known_ids, max_results, stop_threshold: [], ) monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: details_calls.append(list(ids)) or [], ) second = sync.sync_videos(db_session) assert second["videos_added"] == 0 assert second["videos_updated"] == 0 assert details_calls == [[]] assert db_session.query(Video).count() == 1 def test_sync_videos_passes_known_ids_and_backfill_settings(monkeypatch, db_session): channel = _seed_channel(db_session) db_session.add_all( [ Video( youtube_video_id="known1", channel_id=channel.id, title="Known 1", published_at=datetime(2026, 9, 1, tzinfo=timezone.utc), youtube_url="https://www.youtube.com/watch?v=known1", ), Video( youtube_video_id="known2", channel_id=channel.id, title="Known 2", published_at=datetime(2026, 9, 1, tzinfo=timezone.utc), youtube_url="https://www.youtube.com/watch?v=known2", ), ] ) db_session.commit() captured = {} def fake_incremental(creds, playlist_id, known_ids, max_results, stop_threshold): captured.update(known_ids=known_ids, max_results=max_results, stop_threshold=stop_threshold) return ["vid1"] monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object()) monkeypatch.setattr(sync.youtube_client, "fetch_playlist_video_ids_incremental", fake_incremental) monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: [ { "youtube_video_id": "vid1", "youtube_channel_id": channel.youtube_channel_id, "title": "Video One", "description": "d", "thumbnail_url": "t", "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": "PT5M", } ], ) result = sync.sync_videos(db_session) assert result["videos_added"] == 1 assert captured["known_ids"] == {"known1", "known2"} assert captured["max_results"] == settings.videos_backfill_cap assert captured["stop_threshold"] == settings.videos_known_stop_threshold def test_sync_videos_backfill_cap_ingests_all_candidates(monkeypatch, db_session): channel = _seed_channel(db_session) cap = settings.videos_backfill_cap new_ids = [f"vid{i}" for i in range(cap)] monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object()) monkeypatch.setattr( sync.youtube_client, "fetch_playlist_video_ids_incremental", lambda creds, playlist_id, known_ids, max_results, stop_threshold: new_ids, ) details_calls = [] monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: details_calls.append(list(ids)) or [ { "youtube_video_id": vid, "youtube_channel_id": channel.youtube_channel_id, "title": f"Title {vid}", "description": "d", "thumbnail_url": "t", "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": None, } for vid in ids ], ) result = sync.sync_videos(db_session) assert result["videos_added"] == cap assert result["videos_updated"] == 0 assert set(details_calls[0]) == set(new_ids) assert db_session.query(Video).count() == cap def test_sync_videos_skips_unknown_channel(monkeypatch, db_session): _seed_channel(db_session, "chanA") monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object()) monkeypatch.setattr( sync.youtube_client, "fetch_playlist_video_ids_incremental", lambda creds, playlist_id, known_ids, max_results, stop_threshold: [], ) monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: [ { "youtube_video_id": "vidX", "youtube_channel_id": "unknown-channel", "title": "Orphan", "description": "", "thumbnail_url": None, "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": None, } ], ) result = sync.sync_videos(db_session) assert result["videos_skipped"] == 1 assert db_session.query(Video).count() == 0 class _FakeResponse: def __init__(self, payload, status_code=200): self.status_code = status_code self._payload = payload self.text = str(payload) def json(self): return self._payload class _FakeCredentials: token = "fake-token" class _FakeClient: """Serves canned playlistItems pages through the real incremental fetcher.""" def __init__(self, pages): self._pages = list(pages) self.requests = [] def __enter__(self): return self def __exit__(self, *args): return False def get(self, url, params=None, headers=None): self.requests.append({"url": url, "params": params}) return self._pages.pop(0) def _page(items, next_token=None): return _FakeResponse( { "items": [{"contentDetails": {"videoId": vid}} for vid in items], **({"nextPageToken": next_token} if next_token else {}), } ) def test_sync_videos_stops_pagination_on_known_run(monkeypatch, db_session): """A channel with many known videos in a row after the new ones: the incremental fetch must stop paginating once the known-run threshold is hit and only the new ids must get details.""" channel = _seed_channel(db_session) known = [f"known{i}" for i in range(60)] db_session.add_all( [ Video( youtube_video_id=vid, channel_id=channel.id, title=f"Known {vid}", published_at=datetime(2026, 9, 1, tzinfo=timezone.utc), youtube_url=f"https://www.youtube.com/watch?v={vid}", ) for vid in known ] ) db_session.commit() # Page 1: two new videos then 48 known; page 2 continues with known ids, # so the run crosses 50 on the second page and pagination must stop there # (a third page exists and must never be requested). client = _FakeClient( [ _page(["new1", "new2", *known[:48]], next_token="page2"), _page(known[48:58], next_token="page3"), _page(["never-seen"]), ] ) details_calls = [] monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials()) monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client) monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: details_calls.append(list(ids)) or [ { "youtube_video_id": vid, "youtube_channel_id": channel.youtube_channel_id, "title": f"Title {vid}", "description": "", "thumbnail_url": None, "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": None, } for vid in ids ], ) result = sync.sync_videos(db_session) assert result["videos_added"] == 2 assert len(client.requests) == 2 # stopped on page 2, page 3 never fetched assert set(details_calls[0]) == {"new1", "new2"} assert db_session.query(Video).count() == 62 def test_sync_videos_stops_at_backfill_cap(monkeypatch, db_session): """More new videos than the cap: pagination stops once the cap is reached and details are fetched for exactly the capped candidates.""" channel = _seed_channel(db_session) cap = settings.videos_backfill_cap pages = [_page([f"vid{page * 50 + i}" for i in range(50)], next_token=f"p{page + 2}") for page in range(6)] client = _FakeClient(pages) details_calls = [] monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials()) monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client) monkeypatch.setattr( sync.youtube_client, "fetch_videos_details", lambda creds, ids: details_calls.append(list(ids)) or [ { "youtube_video_id": vid, "youtube_channel_id": channel.youtube_channel_id, "title": f"Title {vid}", "description": "", "thumbnail_url": None, "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": None, } for vid in ids ], ) result = sync.sync_videos(db_session) assert len(client.requests) == 4 # 4 pages x 50 = cap reached assert result["videos_added"] == cap assert len(details_calls[0]) == cap assert db_session.query(Video).count() == cap def test_sync_videos_cap_is_hard_history_depth_limit(monkeypatch, db_session): """videos_backfill_cap is an intentional history-depth limit, not a per-sync batch: a 250-video channel ingests only the newest 200 videos ever; a later sync without new videos stops after one page of known ids and adds nothing; a new video on top of the playlist is picked up while the beyond-the-cap history never surfaces.""" channel = _seed_channel(db_session) cap = settings.videos_backfill_cap def _details(creds, ids): return [ { "youtube_video_id": vid, "youtube_channel_id": channel.youtube_channel_id, "title": f"Title {vid}", "description": "", "thumbnail_url": None, "published_at": "2026-09-10T12:00:00Z", "duration_iso8601": None, } for vid in ids ] monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials()) monkeypatch.setattr(sync.youtube_client, "fetch_videos_details", _details) # Uploads playlist, newest first: vid249 .. vid0. playlist = [f"vid{i}" for i in range(249, -1, -1)] def _run(ids): client = _FakeClient( [_page(ids[i : i + 50], next_token=f"page{i // 50}") for i in range(0, len(ids), 50)] ) monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client) return sync.sync_videos(db_session), client # First sync: 250 videos, only the newest 200 fit under the cap. result1, client1 = _run(playlist) assert result1["videos_added"] == cap assert db_session.query(Video).count() == cap assert len(client1.requests) == 4 # 4 pages x 50 = cap, older pages untouched synced = {v.youtube_video_id for v in db_session.query(Video).all()} assert synced == {f"vid{i}" for i in range(50, 250)} assert "vid49" not in synced # older than the cap: never backfilled by design # Second sync, nothing new: one page of 50 known ids in a row stops it. result2, client2 = _run(playlist) assert result2["videos_added"] == 0 assert len(client2.requests) == 1 assert db_session.query(Video).count() == cap # Third sync, one new video on top: only that one is added, the # beyond-the-cap history still does not surface. result3, client3 = _run(["vid250", *playlist]) assert result3["videos_added"] == 1 assert len(client3.requests) == 2 # page 2 needed to confirm 50 known in a row synced_after = {v.youtube_video_id for v in db_session.query(Video).all()} assert synced_after == {f"vid{i}" for i in range(50, 251)} assert "vid49" not in synced_after def test_is_videos_sync_running_reflects_lock(): sync._videos_lock.acquire() try: assert sync.is_videos_sync_running() is True finally: sync._videos_lock.release() assert sync.is_videos_sync_running() is False