Channel features: subscribers, new-videos badges, activity-driven sync

- Store subscriber count from YouTube statistics and show it on channel page
- Sync 50 videos per channel with playlistItems pagination support
- Show per-channel and per-category new-videos counters (2-day window)
- Replace hourly videos sync with activity trigger (2h idle) and
  incremental backfill (hard cap 200 per channel)
- Clicking the sidebar new-videos count filters the category feed to
  recent videos only (new_only)
- Update agent-team docs: deploy after green checks
This commit is contained in:
vrubelroman 2026-09-17 21:07:33 +00:00
parent fde9a439df
commit e10df8dcbd
27 changed files with 1420 additions and 107 deletions

View file

@ -1,10 +1,13 @@
import pytest
from datetime import datetime, timedelta, timezone
from fastapi.testclient import TestClient
from app.core.auth_dependency import require_session
from app.db import get_db
from app.main import app
from app.models.channel import Channel
from app.models.video import Video
@pytest.fixture
@ -29,6 +32,19 @@ def _create_channel(db_session, youtube_channel_id="chanA", title="Channel A"):
return channel
def _seed_video(db_session, channel, video_id, published_at):
video = Video(
youtube_video_id=video_id,
channel_id=channel.id,
title=f"Video {video_id}",
published_at=published_at,
youtube_url=f"https://www.youtube.com/watch?v={video_id}",
)
db_session.add(video)
db_session.commit()
return video
def test_category_crud(client):
created = client.post("/api/categories", json={"name": "Linux"}).json()
assert created["name"] == "Linux"
@ -118,3 +134,64 @@ def test_reorder_rejects_mismatched_ids(client):
client.post("/api/categories", json={"name": "A"})
resp = client.post("/api/categories/reorder", json={"category_ids": [9999]})
assert resp.status_code == 400
def test_category_new_videos_count_sums_recent_videos_of_subscribed_channels(client, db_session):
now = datetime.now(timezone.utc)
category = client.post("/api/categories", json={"name": "Linux"}).json()
assert category["new_videos_count"] == 0
channel_a = _create_channel(db_session, "chanA", "Channel A")
channel_b = _create_channel(db_session, "chanB", "Channel B")
channel_c = _create_channel(db_session, "chanC", "Channel C")
channel_c.subscribed = False
db_session.commit()
client.put(f"/api/channels/{channel_a.id}/categories", json={"category_ids": [category["id"]]})
client.put(f"/api/channels/{channel_b.id}/categories", json={"category_ids": [category["id"]]})
client.put(f"/api/channels/{channel_c.id}/categories", json={"category_ids": [category["id"]]})
_seed_video(db_session, channel_a, "vidARecent1", now - timedelta(days=1))
_seed_video(db_session, channel_a, "vidARecent2", now - timedelta(hours=2))
_seed_video(db_session, channel_a, "vidAOld", now - timedelta(days=30))
_seed_video(db_session, channel_b, "vidBRecent", now - timedelta(days=1))
# Unsubscribed channel: its videos must not count towards the category.
_seed_video(db_session, channel_c, "vidCUnsub", now - timedelta(days=1))
listed = client.get("/api/categories").json()
assert len(listed) == 1
assert listed[0]["channel_count"] == 3
assert listed[0]["new_videos_count"] == 3
def test_category_new_videos_count_excludes_other_categories(client, db_session):
now = datetime.now(timezone.utc)
cat1 = client.post("/api/categories", json={"name": "Linux"}).json()
cat2 = client.post("/api/categories", json={"name": "IT"}).json()
channel_a = _create_channel(db_session, "chanA", "Channel A")
channel_b = _create_channel(db_session, "chanB", "Channel B")
client.put(f"/api/channels/{channel_a.id}/categories", json={"category_ids": [cat1["id"]]})
client.put(f"/api/channels/{channel_b.id}/categories", json={"category_ids": [cat2["id"]]})
# Only an old video in cat1, a recent one in cat2: counts must not leak.
_seed_video(db_session, channel_a, "vidAOld", now - timedelta(days=30))
_seed_video(db_session, channel_b, "vidBRecent", now - timedelta(days=1))
listed = {c["id"]: c for c in client.get("/api/categories").json()}
assert listed[cat1["id"]]["new_videos_count"] == 0
assert listed[cat2["id"]]["new_videos_count"] == 1
def test_category_new_videos_count_zero_without_recent_videos(client, db_session):
now = datetime.now(timezone.utc)
category = client.post("/api/categories", json={"name": "Linux"}).json()
channel = _create_channel(db_session)
client.put(f"/api/channels/{channel.id}/categories", json={"category_ids": [category["id"]]})
_seed_video(db_session, channel, "vidOld", now - timedelta(days=30))
listed = client.get("/api/categories").json()
assert len(listed) == 1
assert listed[0]["channel_count"] == 1
assert listed[0]["new_videos_count"] == 0

View file

@ -1,3 +1,5 @@
from datetime import datetime, timedelta, timezone
import pytest
from fastapi.testclient import TestClient
@ -5,6 +7,7 @@ from app.core.auth_dependency import require_session
from app.db import get_db
from app.main import app
from app.models.channel import Channel
from app.models.video import Video
from app.services import sync
from app.services.youtube_client import YouTubeInsufficientScope
@ -86,3 +89,54 @@ def test_unsubscribe_insufficient_scope_returns_403(client, db_session, monkeypa
def test_unsubscribe_channel_not_found(client):
resp = client.post("/api/channels/9999/unsubscribe")
assert resp.status_code == 404
def _seed_video(db_session, channel, video_id, published_at):
video = Video(
youtube_video_id=video_id,
channel_id=channel.id,
title=f"Video {video_id}",
published_at=published_at,
youtube_url=f"https://www.youtube.com/watch?v={video_id}",
)
db_session.add(video)
db_session.commit()
return video
def test_channel_responses_include_subscriber_and_new_videos_counts(client, db_session):
channel = _seed_channel(db_session)
channel.subscriber_count = 1200000
db_session.commit()
now = datetime.now(timezone.utc)
_seed_video(db_session, channel, "vidRecent", now - timedelta(days=1))
_seed_video(db_session, channel, "vidOld", now - timedelta(days=30))
resp = client.get("/api/channels")
assert resp.status_code == 200
payload = resp.json()
assert len(payload) == 1
assert payload[0]["subscriber_count"] == 1200000
# Only the video from 1 day ago is inside the 7-day window.
assert payload[0]["new_videos_count"] == 1
resp_single = client.get(f"/api/channels/{channel.id}")
assert resp_single.status_code == 200
single = resp_single.json()
assert single["subscriber_count"] == 1200000
assert single["new_videos_count"] == 1
def test_channel_new_videos_count_zero_without_recent_videos(client, db_session):
channel = _seed_channel(db_session)
now = datetime.now(timezone.utc)
_seed_video(db_session, channel, "vidOld", now - timedelta(days=30))
resp = client.get("/api/channels")
assert resp.status_code == 200
payload = resp.json()
assert len(payload) == 1
assert payload[0]["subscriber_count"] is None
assert payload[0]["new_videos_count"] == 0

View file

@ -124,6 +124,133 @@ def test_feed_filters_uncategorized(client, db_session):
assert ids == {"vid1", "vid3"}
def _seed_new_only(db_session):
"""Two categories: channel A in a category (recent + old video),
channel B uncategorized (recent + old video). Published dates are
relative to now so the new_videos_window filter is exercised."""
now = datetime.now(timezone.utc)
channel_a = Channel(youtube_channel_id="chanNewA", title="Channel New A", subscribed=True)
channel_b = Channel(youtube_channel_id="chanNewB", title="Channel New B", subscribed=True)
db_session.add_all([channel_a, channel_b])
db_session.commit()
category = Category(name="Recent", slug="recent", sort_order=0)
db_session.add(category)
db_session.commit()
db_session.execute(channel_categories.insert().values(channel_id=channel_a.id, category_id=category.id))
db_session.commit()
videos = [
Video(
youtube_video_id="vidRecentA",
channel_id=channel_a.id,
title="Recent A",
published_at=now - timedelta(days=1),
youtube_url="https://www.youtube.com/watch?v=vidRecentA",
),
Video(
youtube_video_id="vidOldA",
channel_id=channel_a.id,
title="Old A",
published_at=now - timedelta(days=30),
youtube_url="https://www.youtube.com/watch?v=vidOldA",
),
Video(
youtube_video_id="vidRecentB",
channel_id=channel_b.id,
title="Recent B",
published_at=now - timedelta(hours=2),
youtube_url="https://www.youtube.com/watch?v=vidRecentB",
),
Video(
youtube_video_id="vidOldB",
channel_id=channel_b.id,
title="Old B",
published_at=now - timedelta(days=30),
youtube_url="https://www.youtube.com/watch?v=vidOldB",
),
]
db_session.add_all(videos)
db_session.commit()
return channel_a, channel_b, category, videos
def test_feed_new_only_includes_recent_and_excludes_old(client, db_session):
_seed_new_only(db_session)
resp = client.get("/api/feed?new_only=true").json()
ids = {i["youtube_video_id"] for i in resp["items"]}
assert ids == {"vidRecentA", "vidRecentB"}
# Without new_only nothing changes: old videos appear as usual.
all_ids = {i["youtube_video_id"] for i in client.get("/api/feed").json()["items"]}
assert all_ids == {"vidRecentA", "vidOldA", "vidRecentB", "vidOldB"}
def test_feed_new_only_combines_with_category(client, db_session):
_, _, category, _ = _seed_new_only(db_session)
resp = client.get(f"/api/feed?category_id={category.id}&new_only=true").json()
ids = {i["youtube_video_id"] for i in resp["items"]}
assert ids == {"vidRecentA"}
def test_feed_new_only_excludes_unsubscribed_channels(client, db_session):
now = datetime.now(timezone.utc)
channel = Channel(youtube_channel_id="chanUnsub", title="Channel Unsub", subscribed=False)
db_session.add(channel)
db_session.commit()
db_session.add(
Video(
youtube_video_id="vidFreshUnsub",
channel_id=channel.id,
title="Fresh Unsub",
published_at=now - timedelta(hours=2),
youtube_url="https://www.youtube.com/watch?v=vidFreshUnsub",
)
)
db_session.commit()
resp = client.get("/api/feed?new_only=true").json()
assert resp["items"] == []
# Regular feed is not filtered by subscription.
all_ids = {i["youtube_video_id"] for i in client.get("/api/feed").json()["items"]}
assert all_ids == {"vidFreshUnsub"}
def test_feed_new_only_pagination_cursor_works_in_filtered_set(client, db_session):
now = datetime.now(timezone.utc)
channel = Channel(youtube_channel_id="chanPage", title="Channel Page", subscribed=True)
db_session.add(channel)
db_session.commit()
db_session.add_all(
[
Video(
youtube_video_id=f"vidNew{i}",
channel_id=channel.id,
title=f"Video New {i}",
published_at=now - timedelta(hours=i),
youtube_url=f"https://www.youtube.com/watch?v=vidNew{i}",
)
for i in range(3)
]
)
db_session.commit()
page1 = client.get("/api/feed?new_only=true&limit=2").json()
assert [i["youtube_video_id"] for i in page1["items"]] == ["vidNew0", "vidNew1"]
assert page1["next_cursor"] is not None
page2 = client.get(f"/api/feed?new_only=true&limit=2&cursor={page1['next_cursor']}").json()
assert [i["youtube_video_id"] for i in page2["items"]] == ["vidNew2"]
assert page2["next_cursor"] is None
def test_feed_filters_downloaded(client, db_session):
_, _, _, videos = _seed(db_session)

View file

@ -1,3 +1,6 @@
from datetime import datetime, timezone
from app.config import settings
from app.models.channel import Channel
from app.models.video import Video
from app.services import sync
@ -32,7 +35,7 @@ def test_sync_subscriptions_idempotent_and_unsubscribes(monkeypatch, db_session)
monkeypatch.setattr(
sync.youtube_client,
"fetch_uploads_playlists",
lambda creds, ids: {cid: f"UU{cid}" for cid in ids},
lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": 12345} for cid in ids},
)
result = sync.sync_subscriptions(db_session)
@ -45,6 +48,7 @@ def test_sync_subscriptions_idempotent_and_unsubscribes(monkeypatch, db_session)
assert [c.youtube_channel_id for c in channels] == ["chanA", "chanB"]
assert all(c.subscribed for c in channels)
assert channels[0].uploads_playlist_id == "UUchanA"
assert channels[0].subscriber_count == 12345
# Second sync: chanA disappears from subscriptions, chanC appears.
monkeypatch.setattr(
@ -81,6 +85,46 @@ def test_sync_subscriptions_idempotent_and_unsubscribes(monkeypatch, db_session)
assert channels["chanC"].subscribed is True
def test_sync_subscriptions_skips_none_subscriber_count(monkeypatch, db_session):
channel = Channel(
youtube_channel_id="chanA",
youtube_subscription_id="subA",
title="Channel A",
subscribed=True,
subscriber_count=12345,
)
db_session.add(channel)
db_session.commit()
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _fake_credentials())
monkeypatch.setattr(
sync.youtube_client,
"fetch_subscriptions",
lambda creds: [
{
"youtube_channel_id": "chanA",
"youtube_subscription_id": "subA",
"title": "Channel A",
"description": "d",
"thumbnail_url": "t",
},
],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_uploads_playlists",
lambda creds, ids: {cid: {"uploads_playlist_id": f"UU{cid}", "subscriber_count": None} for cid in ids},
)
result = sync.sync_subscriptions(db_session)
assert result["status"] == "completed"
db_session.refresh(channel)
assert channel.uploads_playlist_id == "UUchanA"
# None in the API response must not overwrite the previously stored count.
assert channel.subscriber_count == 12345
def test_sync_in_progress_raises(monkeypatch, db_session):
sync._subscriptions_lock.acquire()
try:
@ -106,12 +150,14 @@ def _seed_channel(db_session, youtube_channel_id="chanA", uploads_playlist_id="U
return channel
def test_sync_videos_adds_and_updates(monkeypatch, db_session):
def test_sync_videos_adds_new_videos(monkeypatch, db_session):
channel = _seed_channel(db_session)
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client, "fetch_playlist_video_ids", lambda creds, playlist_id, max_results: ["vid1"]
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"],
)
monkeypatch.setattr(
sync.youtube_client,
@ -133,12 +179,23 @@ def test_sync_videos_adds_and_updates(monkeypatch, db_session):
assert result["status"] == "completed"
assert result["videos_added"] == 1
assert result["channels_checked"] == 1
video = db_session.query(Video).filter_by(youtube_video_id="vid1").one()
assert video.title == "Video One"
assert video.duration_seconds == 300
assert video.youtube_url == "https://www.youtube.com/watch?v=vid1"
def test_sync_videos_second_run_is_idempotent_and_fetches_no_details(monkeypatch, db_session):
channel = _seed_channel(db_session)
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: ["vid1"],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
@ -146,29 +203,142 @@ def test_sync_videos_adds_and_updates(monkeypatch, db_session):
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One Updated",
"description": "d2",
"thumbnail_url": "t2",
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT6M",
"duration_iso8601": "PT5M",
}
],
)
result2 = sync.sync_videos(db_session)
assert result2["videos_added"] == 0
assert result2["videos_updated"] == 1
first = sync.sync_videos(db_session)
assert first["videos_added"] == 1
videos = db_session.query(Video).all()
assert len(videos) == 1
assert videos[0].title == "Video One Updated"
# Second run: vid1 is now known, so the incremental fetch reports nothing
# new and no details are requested at all (existing videos are not
# metadata-refreshed by design).
details_calls = []
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: [],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids)) or [],
)
second = sync.sync_videos(db_session)
assert second["videos_added"] == 0
assert second["videos_updated"] == 0
assert details_calls == [[]]
assert db_session.query(Video).count() == 1
def test_sync_videos_passes_known_ids_and_backfill_settings(monkeypatch, db_session):
channel = _seed_channel(db_session)
db_session.add_all(
[
Video(
youtube_video_id="known1",
channel_id=channel.id,
title="Known 1",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url="https://www.youtube.com/watch?v=known1",
),
Video(
youtube_video_id="known2",
channel_id=channel.id,
title="Known 2",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url="https://www.youtube.com/watch?v=known2",
),
]
)
db_session.commit()
captured = {}
def fake_incremental(creds, playlist_id, known_ids, max_results, stop_threshold):
captured.update(known_ids=known_ids, max_results=max_results, stop_threshold=stop_threshold)
return ["vid1"]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(sync.youtube_client, "fetch_playlist_video_ids_incremental", fake_incremental)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: [
{
"youtube_video_id": "vid1",
"youtube_channel_id": channel.youtube_channel_id,
"title": "Video One",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": "PT5M",
}
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == 1
assert captured["known_ids"] == {"known1", "known2"}
assert captured["max_results"] == settings.videos_backfill_cap
assert captured["stop_threshold"] == settings.videos_known_stop_threshold
def test_sync_videos_backfill_cap_ingests_all_candidates(monkeypatch, db_session):
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
new_ids = [f"vid{i}" for i in range(cap)]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: new_ids,
)
details_calls = []
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "d",
"thumbnail_url": "t",
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == cap
assert result["videos_updated"] == 0
assert set(details_calls[0]) == set(new_ids)
assert db_session.query(Video).count() == cap
def test_sync_videos_skips_unknown_channel(monkeypatch, db_session):
_seed_channel(db_session, "chanA")
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: object())
monkeypatch.setattr(sync.youtube_client, "fetch_playlist_video_ids", lambda creds, playlist_id, max_results: [])
monkeypatch.setattr(
sync.youtube_client,
"fetch_playlist_video_ids_incremental",
lambda creds, playlist_id, known_ids, max_results, stop_threshold: [],
)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
@ -189,3 +359,211 @@ def test_sync_videos_skips_unknown_channel(monkeypatch, db_session):
assert result["videos_skipped"] == 1
assert db_session.query(Video).count() == 0
class _FakeResponse:
def __init__(self, payload, status_code=200):
self.status_code = status_code
self._payload = payload
self.text = str(payload)
def json(self):
return self._payload
class _FakeCredentials:
token = "fake-token"
class _FakeClient:
"""Serves canned playlistItems pages through the real incremental fetcher."""
def __init__(self, pages):
self._pages = list(pages)
self.requests = []
def __enter__(self):
return self
def __exit__(self, *args):
return False
def get(self, url, params=None, headers=None):
self.requests.append({"url": url, "params": params})
return self._pages.pop(0)
def _page(items, next_token=None):
return _FakeResponse(
{
"items": [{"contentDetails": {"videoId": vid}} for vid in items],
**({"nextPageToken": next_token} if next_token else {}),
}
)
def test_sync_videos_stops_pagination_on_known_run(monkeypatch, db_session):
"""A channel with many known videos in a row after the new ones: the
incremental fetch must stop paginating once the known-run threshold is hit
and only the new ids must get details."""
channel = _seed_channel(db_session)
known = [f"known{i}" for i in range(60)]
db_session.add_all(
[
Video(
youtube_video_id=vid,
channel_id=channel.id,
title=f"Known {vid}",
published_at=datetime(2026, 9, 1, tzinfo=timezone.utc),
youtube_url=f"https://www.youtube.com/watch?v={vid}",
)
for vid in known
]
)
db_session.commit()
# Page 1: two new videos then 48 known; page 2 continues with known ids,
# so the run crosses 50 on the second page and pagination must stop there
# (a third page exists and must never be requested).
client = _FakeClient(
[
_page(["new1", "new2", *known[:48]], next_token="page2"),
_page(known[48:58], next_token="page3"),
_page(["never-seen"]),
]
)
details_calls = []
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert result["videos_added"] == 2
assert len(client.requests) == 2 # stopped on page 2, page 3 never fetched
assert set(details_calls[0]) == {"new1", "new2"}
assert db_session.query(Video).count() == 62
def test_sync_videos_stops_at_backfill_cap(monkeypatch, db_session):
"""More new videos than the cap: pagination stops once the cap is reached
and details are fetched for exactly the capped candidates."""
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
pages = [_page([f"vid{page * 50 + i}" for i in range(50)], next_token=f"p{page + 2}") for page in range(6)]
client = _FakeClient(pages)
details_calls = []
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
monkeypatch.setattr(
sync.youtube_client,
"fetch_videos_details",
lambda creds, ids: details_calls.append(list(ids))
or [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
],
)
result = sync.sync_videos(db_session)
assert len(client.requests) == 4 # 4 pages x 50 = cap reached
assert result["videos_added"] == cap
assert len(details_calls[0]) == cap
assert db_session.query(Video).count() == cap
def test_sync_videos_cap_is_hard_history_depth_limit(monkeypatch, db_session):
"""videos_backfill_cap is an intentional history-depth limit, not a
per-sync batch: a 250-video channel ingests only the newest 200 videos
ever; a later sync without new videos stops after one page of known ids
and adds nothing; a new video on top of the playlist is picked up while
the beyond-the-cap history never surfaces."""
channel = _seed_channel(db_session)
cap = settings.videos_backfill_cap
def _details(creds, ids):
return [
{
"youtube_video_id": vid,
"youtube_channel_id": channel.youtube_channel_id,
"title": f"Title {vid}",
"description": "",
"thumbnail_url": None,
"published_at": "2026-09-10T12:00:00Z",
"duration_iso8601": None,
}
for vid in ids
]
monkeypatch.setattr(sync.google_oauth, "get_credentials", lambda db: _FakeCredentials())
monkeypatch.setattr(sync.youtube_client, "fetch_videos_details", _details)
# Uploads playlist, newest first: vid249 .. vid0.
playlist = [f"vid{i}" for i in range(249, -1, -1)]
def _run(ids):
client = _FakeClient(
[_page(ids[i : i + 50], next_token=f"page{i // 50}") for i in range(0, len(ids), 50)]
)
monkeypatch.setattr(sync.youtube_client.httpx, "Client", lambda timeout: client)
return sync.sync_videos(db_session), client
# First sync: 250 videos, only the newest 200 fit under the cap.
result1, client1 = _run(playlist)
assert result1["videos_added"] == cap
assert db_session.query(Video).count() == cap
assert len(client1.requests) == 4 # 4 pages x 50 = cap, older pages untouched
synced = {v.youtube_video_id for v in db_session.query(Video).all()}
assert synced == {f"vid{i}" for i in range(50, 250)}
assert "vid49" not in synced # older than the cap: never backfilled by design
# Second sync, nothing new: one page of 50 known ids in a row stops it.
result2, client2 = _run(playlist)
assert result2["videos_added"] == 0
assert len(client2.requests) == 1
assert db_session.query(Video).count() == cap
# Third sync, one new video on top: only that one is added, the
# beyond-the-cap history still does not surface.
result3, client3 = _run(["vid250", *playlist])
assert result3["videos_added"] == 1
assert len(client3.requests) == 2 # page 2 needed to confirm 50 known in a row
synced_after = {v.youtube_video_id for v in db_session.query(Video).all()}
assert synced_after == {f"vid{i}" for i in range(50, 251)}
assert "vid49" not in synced_after
def test_is_videos_sync_running_reflects_lock():
sync._videos_lock.acquire()
try:
assert sync.is_videos_sync_running() is True
finally:
sync._videos_lock.release()
assert sync.is_videos_sync_running() is False

159
tests/test_sync_trigger.py Normal file
View file

@ -0,0 +1,159 @@
import json
from datetime import datetime, timedelta, timezone
from types import SimpleNamespace
from app.services import sync_trigger
def _now():
return datetime.now(timezone.utc)
def test_is_videos_sync_due_without_finished_at():
assert sync_trigger.is_videos_sync_due(None, _now(), 2) is True
def test_is_videos_sync_due_after_idle():
finished = (_now() - timedelta(hours=3)).isoformat()
assert sync_trigger.is_videos_sync_due(finished, _now(), 2) is True
def test_is_videos_sync_due_at_exact_threshold():
finished = (_now() - timedelta(hours=2)).isoformat()
assert sync_trigger.is_videos_sync_due(finished, _now(), 2) is True
def test_is_videos_sync_due_before_idle():
finished = (_now() - timedelta(hours=1)).isoformat()
assert sync_trigger.is_videos_sync_due(finished, _now(), 2) is False
def test_is_videos_sync_due_unparseable_finished_at():
assert sync_trigger.is_videos_sync_due("not-a-date", _now(), 2) is True
assert sync_trigger.is_videos_sync_due(12345, _now(), 2) is True # non-str garbage
def test_is_videos_sync_due_naive_finished_at():
naive = datetime.now(timezone.utc).replace(tzinfo=None) - timedelta(hours=3)
assert sync_trigger.is_videos_sync_due(naive.isoformat(), _now(), 2) is True
def test_extract_finished_at():
assert sync_trigger._extract_finished_at(None) is None
assert sync_trigger._extract_finished_at("not json") is None
assert sync_trigger._extract_finished_at(json.dumps(["a"])) is None
assert sync_trigger._extract_finished_at(json.dumps({"status": "running", "finished_at": None})) is None
assert (
sync_trigger._extract_finished_at(json.dumps({"status": "completed", "finished_at": "2026-09-10T12:00:00+00:00"}))
== "2026-09-10T12:00:00+00:00"
)
class _FakeDB:
def __init__(self):
self.closed = False
def close(self):
self.closed = True
def _patch_threads(monkeypatch, started):
monkeypatch.setattr(
sync_trigger,
"threading",
SimpleNamespace(Thread=lambda target=None, daemon=None: started.append({"target": target, "daemon": daemon})),
)
def test_maybe_trigger_skips_when_sync_running(monkeypatch):
started = []
monkeypatch.setattr(sync_trigger.sync, "is_videos_sync_running", lambda: True)
monkeypatch.setattr(
sync_trigger,
"SessionLocal",
lambda: (_ for _ in ()).throw(AssertionError("DB must not be touched while running")),
)
_patch_threads(monkeypatch, started)
sync_trigger.maybe_trigger_videos_sync()
assert started == []
def test_maybe_trigger_skips_when_not_due(monkeypatch):
started = []
db = _FakeDB()
raw = json.dumps({"status": "completed", "finished_at": (_now() - timedelta(hours=1)).isoformat()})
monkeypatch.setattr(sync_trigger.sync, "is_videos_sync_running", lambda: False)
monkeypatch.setattr(sync_trigger, "SessionLocal", lambda: db)
monkeypatch.setattr(sync_trigger, "get_setting", lambda session, key: raw)
_patch_threads(monkeypatch, started)
sync_trigger.maybe_trigger_videos_sync()
assert started == []
assert db.closed is True
def test_maybe_trigger_starts_thread_when_due_never_synced(monkeypatch):
started = []
db = _FakeDB()
monkeypatch.setattr(sync_trigger.sync, "is_videos_sync_running", lambda: False)
monkeypatch.setattr(sync_trigger, "SessionLocal", lambda: db)
monkeypatch.setattr(sync_trigger, "get_setting", lambda session, key: None)
_patch_threads(monkeypatch, started)
sync_trigger.maybe_trigger_videos_sync()
assert len(started) == 1
assert started[0]["target"] == sync_trigger._run_videos_sync
assert started[0]["daemon"] is True
assert db.closed is True
def test_maybe_trigger_starts_thread_when_due_after_idle(monkeypatch):
started = []
db = _FakeDB()
raw = json.dumps({"status": "completed", "finished_at": (_now() - timedelta(hours=5)).isoformat()})
monkeypatch.setattr(sync_trigger.sync, "is_videos_sync_running", lambda: False)
monkeypatch.setattr(sync_trigger, "SessionLocal", lambda: db)
monkeypatch.setattr(sync_trigger, "get_setting", lambda session, key: raw)
_patch_threads(monkeypatch, started)
sync_trigger.maybe_trigger_videos_sync()
assert len(started) == 1
assert started[0]["daemon"] is True
def test_maybe_trigger_never_raises_when_db_unavailable(monkeypatch):
started = []
monkeypatch.setattr(sync_trigger.sync, "is_videos_sync_running", lambda: False)
monkeypatch.setattr(
sync_trigger,
"SessionLocal",
lambda: (_ for _ in ()).throw(RuntimeError("db down")),
)
_patch_threads(monkeypatch, started)
sync_trigger.maybe_trigger_videos_sync() # must not raise
assert started == []
def test_run_videos_sync_swallows_sync_in_progress(monkeypatch):
closed = []
monkeypatch.setattr(
sync_trigger,
"SessionLocal",
lambda: SimpleNamespace(close=lambda: closed.append(True)),
)
monkeypatch.setattr(
sync_trigger.sync,
"sync_videos",
lambda db: (_ for _ in ()).throw(sync_trigger.sync.SyncInProgress("busy")),
)
sync_trigger._run_videos_sync() # must not raise
assert closed == [True]

View file

@ -115,6 +115,160 @@ def test_fetch_playlist_video_ids_returns_empty_on_404(monkeypatch):
assert result == []
def test_fetch_playlist_video_ids_paginates(monkeypatch):
page1 = FakeResponse(
200,
{
"items": [{"contentDetails": {"videoId": f"vid{i}"}} for i in range(50)],
"nextPageToken": "page2",
},
)
page2 = FakeResponse(
200,
{"items": [{"contentDetails": {"videoId": f"vid{i}"}} for i in range(50, 80)]},
)
seen = []
class RecordingClient(FakeClient):
def get(self, url, params=None, headers=None):
seen.append({"url": url, "params": params})
return self._responses.pop(0)
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: RecordingClient([page1, page2]))
result = youtube_client.fetch_playlist_video_ids(_credentials(), "UUplaylist", 80)
assert result == [f"vid{i}" for i in range(80)]
assert len(seen) == 2
assert seen[0]["params"]["maxResults"] == 50
assert "pageToken" not in seen[0]["params"]
assert seen[1]["params"]["maxResults"] == 30
assert seen[1]["params"]["pageToken"] == "page2"
assert seen[0]["params"]["part"] == "contentDetails"
assert seen[1]["params"]["part"] == "contentDetails"
def test_fetch_playlist_video_ids_stops_at_safety_cap(monkeypatch):
page = FakeResponse(
200,
{"items": [{"contentDetails": {"videoId": "vid"}}], "nextPageToken": "next"},
)
seen = []
class RecordingClient(FakeClient):
def get(self, url, params=None, headers=None):
seen.append(params)
return self._responses.pop(0)
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: RecordingClient([page] * 100))
result = youtube_client.fetch_playlist_video_ids(_credentials(), "UUplaylist", 10000)
assert len(result) == youtube_client.MAX_PLAYLIST_PAGES
assert len(seen) == youtube_client.MAX_PLAYLIST_PAGES
assert "pageToken" not in seen[0]
assert seen[1]["pageToken"] == "next"
def _page_items(video_ids, next_token=None):
payload = {"items": [{"contentDetails": {"videoId": vid}} for vid in video_ids]}
if next_token:
payload["nextPageToken"] = next_token
return FakeResponse(200, payload)
def test_fetch_playlist_video_ids_incremental_returns_only_unknown(monkeypatch):
page = _page_items(["new1", "known1", "new2", "known2"])
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: FakeClient([page]))
result = youtube_client.fetch_playlist_video_ids_incremental(
_credentials(), "UUplaylist", known_ids={"known1", "known2"}, max_results=10, stop_threshold=50
)
assert result == ["new1", "new2"]
def test_fetch_playlist_video_ids_incremental_stops_on_known_run(monkeypatch):
page1 = _page_items([f"known{i}" for i in range(50)], next_token="page2")
page2 = _page_items(["new1", "new2"], next_token="page3")
seen = []
class RecordingClient(FakeClient):
def get(self, url, params=None, headers=None):
seen.append(params)
return self._responses.pop(0)
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: RecordingClient([page1, page2]))
result = youtube_client.fetch_playlist_video_ids_incremental(
_credentials(), "UUplaylist", known_ids={f"known{i}" for i in range(50)}, max_results=10, stop_threshold=50
)
assert result == []
assert len(seen) == 1 # page2 never requested
def test_fetch_playlist_video_ids_incremental_known_run_crosses_pages(monkeypatch):
page1 = _page_items(["new1"] + [f"known{i}" for i in range(49)], next_token="page2")
page2 = _page_items([f"known{i}" for i in range(49, 60)], next_token="page3")
seen = []
class RecordingClient(FakeClient):
def get(self, url, params=None, headers=None):
seen.append(params)
return self._responses.pop(0)
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: RecordingClient([page1, page2]))
result = youtube_client.fetch_playlist_video_ids_incremental(
_credentials(), "UUplaylist", known_ids={f"known{i}" for i in range(60)}, max_results=10, stop_threshold=50
)
assert result == ["new1"]
assert len(seen) == 2 # stopped mid-page-2, page3 never requested
def test_fetch_playlist_video_ids_incremental_stops_at_cap(monkeypatch):
pages = [_page_items([f"vid{page * 50 + i}" for i in range(50)], next_token=f"p{page + 2}") for page in range(6)]
seen = []
class RecordingClient(FakeClient):
def get(self, url, params=None, headers=None):
seen.append(params)
return self._responses.pop(0)
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: RecordingClient(pages))
result = youtube_client.fetch_playlist_video_ids_incremental(
_credentials(), "UUplaylist", known_ids=set(), max_results=200, stop_threshold=50
)
assert len(result) == 200
assert len(seen) == 4
def test_fetch_playlist_video_ids_incremental_empty_on_404(monkeypatch):
response = FakeResponse(404, {"error": {"message": "playlist not found"}})
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: FakeClient([response]))
result = youtube_client.fetch_playlist_video_ids_incremental(
_credentials(), "UUplaylist", known_ids=set(), max_results=10, stop_threshold=50
)
assert result == []
def test_fetch_playlist_video_ids_incremental_dedupes_new_ids(monkeypatch):
page = _page_items(["new1", "new1", "known1"])
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: FakeClient([page]))
result = youtube_client.fetch_playlist_video_ids_incremental(
_credentials(), "UUplaylist", known_ids={"known1"}, max_results=10, stop_threshold=50
)
assert result == ["new1"]
def test_fetch_videos_details(monkeypatch):
response = FakeResponse(
200,
@ -194,13 +348,48 @@ def test_fetch_uploads_playlists_batches(monkeypatch):
200,
{
"items": [
{"id": "chanA", "contentDetails": {"relatedPlaylists": {"uploads": "UUchanA"}}},
{"id": "chanB", "contentDetails": {"relatedPlaylists": {"uploads": "UUchanB"}}},
{
"id": "chanA",
"contentDetails": {"relatedPlaylists": {"uploads": "UUchanA"}},
"statistics": {"subscriberCount": "12345"},
},
{
"id": "chanB",
"contentDetails": {"relatedPlaylists": {"uploads": "UUchanB"}},
"statistics": {"hiddenSubscriberCount": True},
},
{"id": "chanC", "contentDetails": {"relatedPlaylists": {"uploads": "UUchanC"}}},
{"id": "chanD", "contentDetails": {"relatedPlaylists": {}}, "statistics": {"subscriberCount": "42"}},
]
},
)
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: FakeClient([response]))
result = youtube_client.fetch_uploads_playlists(_credentials(), ["chanA", "chanB"])
result = youtube_client.fetch_uploads_playlists(_credentials(), ["chanA", "chanB", "chanC", "chanD"])
assert result == {"chanA": "UUchanA", "chanB": "UUchanB"}
assert result == {
"chanA": {"uploads_playlist_id": "UUchanA", "subscriber_count": 12345},
"chanB": {"uploads_playlist_id": "UUchanB", "subscriber_count": None},
"chanC": {"uploads_playlist_id": "UUchanC", "subscriber_count": None},
"chanD": {"uploads_playlist_id": None, "subscriber_count": 42},
}
def test_fetch_uploads_playlists_subscriber_count_not_an_int_is_none(monkeypatch):
response = FakeResponse(
200,
{
"items": [
{
"id": "chanA",
"contentDetails": {"relatedPlaylists": {"uploads": "UUchanA"}},
"statistics": {"subscriberCount": "not-a-number"},
}
]
},
)
monkeypatch.setattr(youtube_client.httpx, "Client", lambda timeout: FakeClient([response]))
result = youtube_client.fetch_uploads_playlists(_credentials(), ["chanA"])
assert result == {"chanA": {"uploads_playlist_id": "UUchanA", "subscriber_count": None}}