myYouTube/tests/test_sync.py

570 lines
19 KiB
Python
Raw Normal View History

from datetime import datetime, timezone
from app.config import settings
from app.models.channel import Channel
from app.models.video import Video
from app.services import sync
def _fake_credentials():
return object()
def test_sync_subscriptions_idempotent_and_unsubscribes(monkeypatch, db_session):
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _fake_credentials())
monkeypatch.setattr(
sync.youtube_client,
"fetch_subscriptions",
lambda creds: [
{
"youtube_channel_id": "chanA",
"youtube_subscription_id": "subA",
"title": "Channel A",
"description": "d",
"thumbnail_url": "t",
},
{
"youtube_channel_id": "chanB",
"youtube_subscription_id": "subB",
"title": "Channel B",
"description": "d",
"thumbnail_url": "t",
},
],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_uploads_playlists",
lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": 12345} for cid in ids},
)
result = sync.sync_subscriptions(db_session)
assert result["status"] == "completed"
assert result["channels_added"] == 2
assert result["channels_unsubscribed"] == 0
channels = db_session.query(Channel).order_by(Channel.youtube_channel_id).all()
assert [c.youtube_channel_id for c in channels] == ["chanA", "chanB"]
assert all(c.subscribed for c in channels)
assert channels[0].uploads_playlist_id == "UUchanA"
assert channels[0].subscriber_count == 12345
# Second sync: chanA disappears from subscriptions, chanC appears.
monkeypatch.setattr(
sync.youtube_client,
"fetch_subscriptions",
lambda creds: [
{
"youtube_channel_id": "chanB",
"youtube_subscription_id": "subB",
"title": "Channel B",
"description": "d",
"thumbnail_url": "t",
},
{
"youtube_channel_id": "chanC",
"youtube_subscription_id": "subC",
"title": "Channel C",
"description": "d",
"thumbnail_url": "t",
},
],
)
result2 = sync.sync_subscriptions(db_session)
assert result2["channels_added"] == 1
assert result2["channels_updated"] == 1
assert result2["channels_unsubscribed"] == 1
channels = {c.youtube_channel_id: c for c in db_session.query(Channel).all()}
assert len(channels) == 3
assert channels["chanA"].subscribed is False
assert channels["chanB"].subscribed is True
assert channels["chanC"].subscribed is True
def test_sync_subscriptions_skips_none_subscriber_count(monkeypatch, db_session):
channel = Channel(
youtube_channel_id="chanA",
youtube_subscription_id="subA",
title="Channel A",
subscribed=True,
subscriber_count=12345,
)
db_session.add(channel)
db_session.commit()
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _fake_credentials())
monkeypatch.setattr(
sync.youtube_client,
"fetch_subscriptions",
lambda creds: [
{
"youtube_channel_id": "chanA",
"youtube_subscription_id": "subA",
"title": "Channel A",
"description": "d",
"thumbnail_url": "t",
},
],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_uploads_playlists",
lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": None} for cid in ids},
)
result = sync.sync_subscriptions(db_session)
assert result["status"] == "completed"
db_session.refresh(channel)
assert channel.uploads_playlist_id == "UUchanA"
# None in the API response must not overwrite the previously stored count.
assert channel.subscriber_count == 12345
def test_sync_in_progress_raises(monkeypatch, db_session):
sync._subscriptions_lock.acquire()
try:
try:
sync.sync_subscriptions(db_session)
assert False, "expected SyncInProgress"
except sync.SyncInProgress:
pass
finally:
sync._subscriptions_lock.release()
def _seed_channel(db_session, youtube_channel_id="chanA", uploads_playlist_id="UUchanA"):
channel = Channel(
youtube_channel_id=youtube_channel_id,
title="Channel",
subscribed=True,
uploads_playlist_id=uploads_playlist_id,
)
db_session.add(channel)
db_session.commit()
db_session.refresh(channel)
return channel
def test_sync_videos_adds_new_videos(monkeypatch, db_session):
channel = _seed_channel(db_session)
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT5M",
}
],
)
result = sync.sync_videos(db_session)
assert result["status"] == "completed"
assert result["videos_added"] == 1
assert result["channels_checked"] == 1
video = db_session.query(Video).filter_by(youtube_video_id="vid1").one()
assert video.title == "Video One"
assert video.duration_seconds == 300
assert video.youtube_url == "https://www.youtube.com/watch?v=vid1"
def test_sync_videos_second_run_is_idempotent_and_fetches_no_details(monkeypatch, db_session):
channel = _seed_channel(db_session)
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT5M",
}
],
)
first = sync.sync_videos(db_session)
assert first["videos_added"] == 1
# Second run: vid1 is now known, so the incremental fetch reports nothing
# new and no details are requested at all (existing videos are not
# metadata-refreshed by design).
details_calls = []
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: [],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids)) or [],
)
second = sync.sync_videos(db_session)
assert second["videos_added"] == 0
assert second["videos_updated"] == 0
assert details_calls == [[]]
assert db_session.query(Video).count() == 1
def test_sync_videos_passes_known_ids_and_backfill_settings(monkeypatch, db_session):
channel = _seed_channel(db_session)
db_session.add_all(
[
Video(
youtube_video_id="known1",
channel_id=channel.id,
title="Known 1",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url="https://www.youtube.com/watch?v=known1",
),
Video(
youtube_video_id="known2",
channel_id=channel.id,
title="Known 2",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url="https://www.youtube.com/watch?v=known2",
),
]
)
db_session.commit()
captured = {}
def fake_incremental(creds, playlist_id, known_ids, max_results, stop_threshold):
captured.update(known_ids=known_ids, max_results=max_results, stop_threshold=stop_threshold)
return ["vid1"]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(sync.youtube_client, "fetch_playlist_video_ids_incremental", fake_incremental)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT5M",
}
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == 1
assert captured["known_ids"] == {"known1", "known2"}
assert captured["max_results"] == settings.videos_backfill_cap
assert captured["stop_threshold"] == settings.videos_known_stop_threshold
def test_sync_videos_backfill_cap_ingests_all_candidates(monkeypatch, db_session):
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
new_ids = [f"vid{i}" for i in range(cap)]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: new_ids,
)
details_calls = []
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == cap
assert result["videos_updated"] == 0
assert set(details_calls[0]) == set(new_ids)
assert db_session.query(Video).count() == cap
def test_sync_videos_skips_unknown_channel(monkeypatch, db_session):
_seed_channel(db_session, "chanA")
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: [],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vidX",
"youtube_channel_id": "unknown-channel",
"title": "Orphan",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
],
)
result = sync.sync_videos(db_session)
assert result["videos_skipped"] == 1
assert db_session.query(Video).count() == 0
class _FakeResponse:
def __init__(self, payload, status_code=200):
self.status_code = status_code
self._payload = payload
self.text = str(payload)
def json(self):
return self._payload
class _FakeCredentials:
token = "fake-token"
class _FakeClient:
"""Serves canned playlistItems pages through the real incremental fetcher."""
def __init__(self, pages):
self._pages = list(pages)
self.requests = []
def __enter__(self):
return self
def __exit__(self, *args):
return False
def get(self, url, params=None, headers=None):
self.requests.append({"url": url, "params": params})
return self._pages.pop(0)
def _page(items, next_token=None):
return _FakeResponse(
{
"items": [{"contentDetails": {"videoId": vid}} for vid in items],
**({"nextPageToken": next_token} if next_token else {}),
}
)
def test_sync_videos_stops_pagination_on_known_run(monkeypatch, db_session):
"""A channel with many known videos in a row after the new ones: the
incremental fetch must stop paginating once the known-run threshold is hit
and only the new ids must get details."""
channel = _seed_channel(db_session)
known = [f"known{i}" for i in range(60)]
db_session.add_all(
[
Video(
youtube_video_id=vid,
channel_id=channel.id,
title=f"Known {vid}",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url=f"https://www.youtube.com/watch?v={vid}",
)
for vid in known
]
)
db_session.commit()
# Page 1: two new videos then 48 known; page 2 continues with known ids,
# so the run crosses 50 on the second page and pagination must stop there
# (a third page exists and must never be requested).
client = _FakeClient(
[
_page(["new1", "new2", *known[:48]], next_token="page2"),
_page(known[48:58], next_token="page3"),
_page(["never-seen"]),
]
)
details_calls = []
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == 2
assert len(client.requests) == 2 # stopped on page 2, page 3 never fetched
assert set(details_calls[0]) == {"new1", "new2"}
assert db_session.query(Video).count() == 62
def test_sync_videos_stops_at_backfill_cap(monkeypatch, db_session):
"""More new videos than the cap: pagination stops once the cap is reached
and details are fetched for exactly the capped candidates."""
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
pages = [_page([f"vid{page * 50 + i}" for i in range(50)], next_token=f"p{page + 2}") for page in range(6)]
client = _FakeClient(pages)
details_calls = []
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert len(client.requests) == 4 # 4 pages x 50 = cap reached
assert result["videos_added"] == cap
assert len(details_calls[0]) == cap
assert db_session.query(Video).count() == cap
def test_sync_videos_cap_is_hard_history_depth_limit(monkeypatch, db_session):
"""videos_backfill_cap is an intentional history-depth limit, not a
per-sync batch: a 250-video channel ingests only the newest 200 videos
ever; a later sync without new videos stops after one page of known ids
and adds nothing; a new video on top of the playlist is picked up while
the beyond-the-cap history never surfaces."""
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
def _details(creds, ids):
return [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client, "fetch_videos_details", _details)
# Uploads playlist, newest first: vid249 .. vid0.
playlist = [f"vid{i}" for i in range(249, -1, -1)]
def _run(ids):
client = _FakeClient(
[_page(ids[i : i + 50], next_token=f"page{i // 50}") for i in range(0, len(ids), 50)]
)
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
return sync.sync_videos(db_session), client
# First sync: 250 videos, only the newest 200 fit under the cap.
result1, client1 = _run(playlist)
assert result1["videos_added"] == cap
assert db_session.query(Video).count() == cap
assert len(client1.requests) == 4 # 4 pages x 50 = cap, older pages untouched
synced = {v.youtube_video_id for v in db_session.query(Video).all()}
assert synced == {f"vid{i}" for i in range(50, 250)}
assert "vid49" not in synced # older than the cap: never backfilled by design
# Second sync, nothing new: one page of 50 known ids in a row stops it.
result2, client2 = _run(playlist)
assert result2["videos_added"] == 0
assert len(client2.requests) == 1
assert db_session.query(Video).count() == cap
# Third sync, one new video on top: only that one is added, the
# beyond-the-cap history still does not surface.
result3, client3 = _run(["vid250", *playlist])
assert result3["videos_added"] == 1
assert len(client3.requests) == 2 # page 2 needed to confirm 50 known in a row
synced_after = {v.youtube_video_id for v in db_session.query(Video).all()}
assert synced_after == {f"vid{i}" for i in range(50, 251)}
assert "vid49" not in synced_after
def test_is_videos_sync_running_reflects_lock():
sync._videos_lock.acquire()
try:
assert sync.is_videos_sync_running() is True
finally:
sync._videos_lock.release()
assert sync.is_videos_sync_running() is False