fix(youtube): use url= param for ScrapeCreators comments/transcript + parse new response shape (#265)
PR #260 wired YouTube comment enrichment against `/v1/youtube/video/comments` with `id=<video_id>`, but the endpoint requires `url=https://www.youtube.com/watch?v=<video_id>`. Every enrich call was returning 400 "missing_parameter: you must provide a url", so no YouTube items ever carried `top_comments`. The SC transcript fallback (`_sc_fetch_transcript`) had the identical contract mistake. It was latent because `_fetch_transcript` prefers yt-dlp and the SC path only fires when yt-dlp is missing, but it would have failed the same way on hosts without yt-dlp installed. Switching both callers to `url=` surfaces a second issue in the response parser: SC returns `author` as `{"name": "@handle", ...}` and nests like counts under `engagement.likes`, not top-level. The parser was reading `author` as a string and missing the nested likes, so even after the param fix every comment would land with an object-shaped author and 0 likes. - `_fetch_video_comments`: send `url=` on both urllib and requests branches - `_sc_fetch_transcript`: same - Response parser: extract `author.name` when author is a dict, read `engagement.likes` when top-level `likes` is absent, prefer `publishedTime` / `publishedTimeText` for date. Legacy string-author and top-level-likes shapes still work, so existing mocks are unchanged. Verified live against api.scrapecreators.com: `_fetch_video_comments` now returns fully-populated comments with real @handles and like counts (e.g. "@JennyNicholson: ... (49000 likes, 2025-04-15)"). All tests in youtube_yt/normalize/signals/render pass. Plan: docs/plans/2026-04-15-002-fix-youtube-comments-scrapecreators-param-plan.md 🤖 Generated with Claude Opus 4.6 (1M context) via [Claude Code](https://claude.com/claude-code) + Compound Engineering v2.56.1 Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com> Co-authored-by: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -732,10 +732,11 @@ def _fetch_video_comments(
|
|||||||
Returns:
|
Returns:
|
||||||
List of comment dicts with author, text, likes, date.
|
List of comment dicts with author, text, likes, date.
|
||||||
"""
|
"""
|
||||||
|
video_url = f"https://www.youtube.com/watch?v={video_id}"
|
||||||
if not _requests:
|
if not _requests:
|
||||||
try:
|
try:
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
params = urlencode({"id": video_id})
|
params = urlencode({"url": video_url})
|
||||||
url = f"{SCRAPECREATORS_YT_BASE}/video/comments?{params}"
|
url = f"{SCRAPECREATORS_YT_BASE}/video/comments?{params}"
|
||||||
headers = http.scrapecreators_headers(token)
|
headers = http.scrapecreators_headers(token)
|
||||||
headers["User-Agent"] = http.USER_AGENT
|
headers["User-Agent"] = http.USER_AGENT
|
||||||
@@ -747,7 +748,7 @@ def _fetch_video_comments(
|
|||||||
try:
|
try:
|
||||||
resp = _requests.get(
|
resp = _requests.get(
|
||||||
f"{SCRAPECREATORS_YT_BASE}/video/comments",
|
f"{SCRAPECREATORS_YT_BASE}/video/comments",
|
||||||
params={"id": video_id},
|
params={"url": video_url},
|
||||||
headers=http.scrapecreators_headers(token),
|
headers=http.scrapecreators_headers(token),
|
||||||
timeout=30,
|
timeout=30,
|
||||||
)
|
)
|
||||||
@@ -763,11 +764,32 @@ def _fetch_video_comments(
|
|||||||
text = c.get("text") or c.get("body") or c.get("content", "")
|
text = c.get("text") or c.get("body") or c.get("content", "")
|
||||||
if not text:
|
if not text:
|
||||||
continue
|
continue
|
||||||
|
|
||||||
|
# SC returns author as {"name": "@handle", ...}; legacy mocks may pass a string.
|
||||||
|
author = c.get("author") or c.get("author_name", "")
|
||||||
|
if isinstance(author, dict):
|
||||||
|
author = author.get("name") or author.get("handle") or ""
|
||||||
|
|
||||||
|
# SC nests likes under engagement.likes; legacy shapes used top-level keys.
|
||||||
|
engagement = c.get("engagement") or {}
|
||||||
|
likes = c.get("likes")
|
||||||
|
if likes is None:
|
||||||
|
likes = engagement.get("likes", 0) if isinstance(engagement, dict) else 0
|
||||||
|
if not likes:
|
||||||
|
likes = c.get("vote_count", 0)
|
||||||
|
|
||||||
|
date = (
|
||||||
|
c.get("date")
|
||||||
|
or c.get("published_at")
|
||||||
|
or c.get("publishedTime")
|
||||||
|
or c.get("publishedTimeText", "")
|
||||||
|
)
|
||||||
|
|
||||||
comments.append({
|
comments.append({
|
||||||
"author": c.get("author") or c.get("author_name", ""),
|
"author": author,
|
||||||
"text": text[:400],
|
"text": text[:400],
|
||||||
"likes": c.get("likes") or c.get("vote_count", 0),
|
"likes": likes,
|
||||||
"date": c.get("date") or c.get("published_at", ""),
|
"date": date,
|
||||||
})
|
})
|
||||||
|
|
||||||
return comments
|
return comments
|
||||||
@@ -931,10 +953,11 @@ def _sc_fetch_transcript(video_id: str, token: str) -> Optional[str]:
|
|||||||
Returns:
|
Returns:
|
||||||
Plaintext transcript string, or None if unavailable.
|
Plaintext transcript string, or None if unavailable.
|
||||||
"""
|
"""
|
||||||
|
video_url = f"https://www.youtube.com/watch?v={video_id}"
|
||||||
if not _requests:
|
if not _requests:
|
||||||
try:
|
try:
|
||||||
from urllib.parse import urlencode
|
from urllib.parse import urlencode
|
||||||
params = urlencode({"id": video_id})
|
params = urlencode({"url": video_url})
|
||||||
url = f"{SCRAPECREATORS_YT_BASE}/video/transcript?{params}"
|
url = f"{SCRAPECREATORS_YT_BASE}/video/transcript?{params}"
|
||||||
headers = http.scrapecreators_headers(token)
|
headers = http.scrapecreators_headers(token)
|
||||||
headers["User-Agent"] = http.USER_AGENT
|
headers["User-Agent"] = http.USER_AGENT
|
||||||
@@ -946,7 +969,7 @@ def _sc_fetch_transcript(video_id: str, token: str) -> Optional[str]:
|
|||||||
try:
|
try:
|
||||||
resp = _requests.get(
|
resp = _requests.get(
|
||||||
f"{SCRAPECREATORS_YT_BASE}/video/transcript",
|
f"{SCRAPECREATORS_YT_BASE}/video/transcript",
|
||||||
params={"id": video_id},
|
params={"url": video_url},
|
||||||
headers=http.scrapecreators_headers(token),
|
headers=http.scrapecreators_headers(token),
|
||||||
timeout=30,
|
timeout=30,
|
||||||
)
|
)
|
||||||
|
|||||||
Reference in New Issue
Block a user