Merge pull request #340 from dzivkovi/fix/youtube-transcript-observability

fix(youtube): surface transcript-fetch ratio + add degraded nudge for stale yt-dlp
This commit is contained in:
Trevin Chow
2026-05-17 00:31:37 -07:00
committed by GitHub
5 changed files with 268 additions and 11 deletions
+14 -1
View File
@@ -880,7 +880,20 @@ def main() -> int:
# Show quality nudge if applicable # Show quality nudge if applicable
try: try:
from lib import quality_nudge from lib import quality_nudge
quality = quality_nudge.compute_quality_score(config, {}) # Populate transcript-fetch ratio so quality_nudge can detect the
# degraded-YouTube failure mode (videos returned but transcripts
# silently failed - typically a stale yt-dlp binary).
youtube_items = report.items_by_source.get("youtube") or []
research_results = {
"youtube_videos_count": len(youtube_items),
"youtube_transcripts_count": sum(
1 for it in youtube_items
if (it.metadata.get("transcript_highlights") or it.metadata.get("transcript_snippet"))
),
"youtube_error": report.errors_by_source.get("youtube"),
"x_error": report.errors_by_source.get("x"),
}
quality = quality_nudge.compute_quality_score(config, research_results)
if quality.get("nudge_text"): if quality.get("nudge_text"):
sys.stderr.write(f"\n{quality['nudge_text']}\n") sys.stderr.write(f"\n{quality['nudge_text']}\n")
sys.stderr.flush() sys.stderr.flush()
+70 -6
View File
@@ -45,26 +45,51 @@ def _is_youtube_active(config: dict, research_results: dict) -> bool:
return True return True
# Below this transcript-fetch ratio, YouTube is considered "degraded" rather
# than active. Picked at 50% so a single legitimate caption-disabled video in a
# multi-video result does not trip the nudge, but a stale-yt-dlp run that fails
# every transcript does. Tunable via DEGRADED_TRANSCRIPT_THRESHOLD env var if
# operators need to adjust without code changes.
DEFAULT_DEGRADED_TRANSCRIPT_THRESHOLD = 0.5
def _is_youtube_degraded(research_results: dict, threshold: float) -> bool:
"""YouTube is degraded when videos were returned but the transcript-fetch
ratio is below threshold. The canonical cause is a stale yt-dlp binary -
YouTube's caption format changes frequently and old binaries silently fail
every transcript while the search itself still succeeds.
"""
videos = int(research_results.get("youtube_videos_count") or 0)
transcripts = int(research_results.get("youtube_transcripts_count") or 0)
if videos <= 0:
return False
return (transcripts / videos) < threshold
def compute_quality_score(config: dict, research_results: dict) -> dict: def compute_quality_score(config: dict, research_results: dict) -> dict:
"""Compute research quality score based on 5 core sources. """Compute research quality score based on 5 core sources.
Args: Args:
config: Configuration dict from env.get_config() config: Configuration dict from env.get_config()
research_results: Dict with keys like x_error, youtube_error, research_results: Dict with keys like x_error, youtube_error,
reddit_error reflecting what happened this run. reddit_error reflecting what happened this run. Optional keys
``youtube_videos_count`` and ``youtube_transcripts_count`` enable
degraded-YouTube detection (transcript-fetch ratio below threshold).
Returns: Returns:
{ {
"score_pct": 40-100, "score_pct": 40-100,
"core_active": ["hn", "polymarket", ...], "core_active": ["hn", "polymarket", ...],
"core_missing": ["x", "youtube"], "core_missing": ["x", "youtube"],
"core_errored": [], # configured but errored "core_errored": [], # configured but errored at top level
"core_degraded": [], # configured and returned items but quality below threshold
"nudge_text": "..." or None if 100% "nudge_text": "..." or None if 100%
} }
""" """
core_active: List[str] = [] core_active: List[str] = []
core_missing: List[str] = [] core_missing: List[str] = []
core_errored: List[str] = [] core_errored: List[str] = []
core_degraded: List[str] = []
# HN, Polymarket, and Reddit are always active # HN, Polymarket, and Reddit are always active
core_active.append("hn") core_active.append("hn")
@@ -84,6 +109,13 @@ def compute_quality_score(config: dict, research_results: dict) -> dict:
yt_active = _is_youtube_active(config, research_results) yt_active = _is_youtube_active(config, research_results)
if yt_active: if yt_active:
core_active.append("youtube") core_active.append("youtube")
# Active means yt-dlp is installed and search did not error at the top
# level. But search-success + transcript-failure is the canonical
# stale-binary failure mode that the footer used to hide. Flag as
# degraded so the user gets an actionable nudge to update the binary.
threshold = float(config.get("DEGRADED_TRANSCRIPT_THRESHOLD") or DEFAULT_DEGRADED_TRANSCRIPT_THRESHOLD)
if _is_youtube_degraded(research_results, threshold):
core_degraded.append("youtube")
else: else:
core_missing.append("youtube") core_missing.append("youtube")
# Check if configured but errored (yt-dlp installed but failed this run) # Check if configured but errored (yt-dlp installed but failed this run)
@@ -99,24 +131,41 @@ def compute_quality_score(config: dict, research_results: dict) -> dict:
has_sc = bool(config.get("SCRAPECREATORS_API_KEY")) has_sc = bool(config.get("SCRAPECREATORS_API_KEY"))
active_sources = research_results.get("active_sources") or [] active_sources = research_results.get("active_sources") or []
nudge_text = _build_nudge_text(core_missing, core_errored, has_sc=has_sc, active_sources=active_sources) if core_missing else None nudge_text = _build_nudge_text(
core_missing,
core_errored,
core_degraded,
research_results,
has_sc=has_sc,
active_sources=active_sources,
) if (core_missing or core_degraded) else None
return { return {
"score_pct": score_pct, "score_pct": score_pct,
"core_active": core_active, "core_active": core_active,
"core_missing": core_missing, "core_missing": core_missing,
"core_errored": core_errored, "core_errored": core_errored,
"core_degraded": core_degraded,
"nudge_text": nudge_text, "nudge_text": nudge_text,
} }
def _build_nudge_text(core_missing: List[str], core_errored: List[str], has_sc: bool = False, active_sources: list = None) -> str: def _build_nudge_text(
"""Build human-readable nudge text describing what was missed. core_missing: List[str],
core_errored: List[str],
core_degraded: List[str] = None,
research_results: dict = None,
has_sc: bool = False,
active_sources: list = None,
) -> str:
"""Build human-readable nudge text describing what was missed or degraded.
Prioritizes free suggestions. Optionally mentions bonus sources Prioritizes free suggestions. Optionally mentions bonus sources
(TikTok, Instagram, Threads, Pinterest) if ScrapeCreators key is configured. (TikTok, Instagram, Threads, Pinterest) if ScrapeCreators key is configured.
""" """
lines: List[str] = [] lines: List[str] = []
core_degraded = core_degraded or []
research_results = research_results or {}
# Describe what was missed # Describe what was missed
missed_parts: List[str] = [] missed_parts: List[str] = []
@@ -129,7 +178,11 @@ def _build_nudge_text(core_missing: List[str], core_errored: List[str], has_sc:
active_count = 5 - len(core_missing) active_count = 5 - len(core_missing)
lines.append(f"Research quality: {active_count}/5 core sources.") lines.append(f"Research quality: {active_count}/5 core sources.")
lines.append(f"Missing: {', '.join(missed_parts)}.") if missed_parts:
lines.append(f"Missing: {', '.join(missed_parts)}.")
if core_degraded:
degraded_labels = ", ".join(SOURCE_LABELS[s] for s in core_degraded)
lines.append(f"Degraded: {degraded_labels}.")
lines.append("") lines.append("")
# Free suggestions # Free suggestions
@@ -159,6 +212,17 @@ def _build_nudge_text(core_missing: List[str], core_errored: List[str], has_sc:
"explanations on any topic. Install yt-dlp: brew install yt-dlp (free)" "explanations on any topic. Install yt-dlp: brew install yt-dlp (free)"
) )
if "youtube" in core_degraded:
videos = int(research_results.get("youtube_videos_count") or 0)
transcripts = int(research_results.get("youtube_transcripts_count") or 0)
free_suggestions.append(
f"YouTube returned {videos} videos but only {transcripts} transcripts "
"captured. The most common cause is a stale yt-dlp binary - YouTube's "
"caption format changes frequently and old binaries silently fail every "
"transcript. Update via your package manager: scoop update yt-dlp "
"(Windows), brew upgrade yt-dlp (macOS), or pip install -U yt-dlp."
)
# Mention bonus opt-in sources when SC key is present # Mention bonus opt-in sources when SC key is present
if has_sc: if has_sc:
bonus_hints = [] bonus_hints = []
+5 -4
View File
@@ -1285,15 +1285,16 @@ def _build_source_footer_lines(report: schema.Report) -> list[str]:
if total > 0: if total > 0:
total_str = f"{total:,}" if total >= 1000 else str(total) total_str = f"{total:,}" if total >= 1000 else str(total)
parts.append(f"{total_str} {word}") parts.append(f"{total_str} {word}")
# YouTube: append "N with transcripts" instead of a third likes-based column. # YouTube: always append "M/N with transcripts" so a zero-transcript run
# Transcripts are a more meaningful research-depth signal than likes. # (typically caused by a stale yt-dlp binary) is visible at the conclusion
# surface. Hiding zero converts a problem signal into an absence; the very
# case that needs to be loud is the one previously omitted from the footer.
if source_key == "youtube": if source_key == "youtube":
with_transcripts = sum( with_transcripts = sum(
1 for it in items 1 for it in items
if (it.metadata.get("transcript_highlights") or it.metadata.get("transcript_snippet")) if (it.metadata.get("transcript_highlights") or it.metadata.get("transcript_snippet"))
) )
if with_transcripts > 0: parts.append(f"{with_transcripts}/{len(items)} with transcripts")
parts.append(f"{with_transcripts} with transcripts")
stats = "".join(parts) stats = "".join(parts)
out.append(_footer_line_for_source(emoji, label, len(items), item_word, stats)) out.append(_footer_line_for_source(emoji, label, len(items), item_word, stats))
+94
View File
@@ -199,3 +199,97 @@ class TestRedditNeverInCoreErrored:
# Reddit is always-active in core (public path), error doesn't demote it # Reddit is always-active in core (public path), error doesn't demote it
assert "reddit" in q["core_active"] assert "reddit" in q["core_active"]
assert q["score_pct"] == 100 assert q["score_pct"] == 100
class TestYouTubeDegraded:
"""YouTube is `degraded` when videos returned but transcripts below threshold.
Canonical failure mode: a stale yt-dlp binary still finds videos via search
but silently fails every transcript fetch because YouTube's caption format
has moved on. Pre-fix the user got no signal of this; the footer hid zero,
and quality_nudge only checked top-level errors.
"""
def test_zero_of_six_transcripts_flags_degraded(self):
q = _compute(
ytdlp_installed=True,
result_overrides={
"youtube_videos_count": 6,
"youtube_transcripts_count": 0,
},
)
assert "youtube" in q["core_degraded"]
assert q["nudge_text"] is not None
# Counts surface in the message so the user sees the actual ratio
assert "6 videos" in q["nudge_text"]
assert "0 transcripts" in q["nudge_text"]
assert "stale yt-dlp" in q["nudge_text"].lower()
# Updates path mentions all three common package managers
assert "scoop" in q["nudge_text"].lower()
assert "brew" in q["nudge_text"].lower()
assert "pip install" in q["nudge_text"].lower()
def test_five_of_six_transcripts_does_not_flag_degraded(self):
# 83% transcript success - well above the 50% threshold
# X is also enabled so all 5 cores are active and no nudge should fire
q = _compute(
config_overrides={"AUTH_TOKEN": "tok123"},
ytdlp_installed=True,
result_overrides={
"youtube_videos_count": 6,
"youtube_transcripts_count": 5,
},
)
assert "youtube" not in q["core_degraded"]
assert q["nudge_text"] is None # All 5 core sources active, no degradation
def test_zero_videos_does_not_flag_degraded(self):
# No videos returned -> degraded check is meaningless and must not fire
q = _compute(
ytdlp_installed=True,
result_overrides={
"youtube_videos_count": 0,
"youtube_transcripts_count": 0,
},
)
assert "youtube" not in q["core_degraded"]
def test_one_of_three_transcripts_flags_degraded(self):
# 33% - below 50% threshold; the canonical "yt-dlp partially working" case
q = _compute(
ytdlp_installed=True,
result_overrides={
"youtube_videos_count": 3,
"youtube_transcripts_count": 1,
},
)
assert "youtube" in q["core_degraded"]
assert "Degraded: YouTube" in q["nudge_text"]
def test_threshold_tunable_via_config(self):
# Operator overrides threshold via env-style config to be more permissive
q = _compute(
config_overrides={"DEGRADED_TRANSCRIPT_THRESHOLD": "0.1"},
ytdlp_installed=True,
result_overrides={
"youtube_videos_count": 10,
"youtube_transcripts_count": 2, # 20%, below default 50% but above override 10%
},
)
assert "youtube" not in q["core_degraded"]
def test_degraded_does_not_affect_score(self):
# Degradation is informational, not score-affecting; YouTube still counts as active
q = _compute(
config_overrides={"AUTH_TOKEN": "tok123"},
ytdlp_installed=True,
result_overrides={
"youtube_videos_count": 6,
"youtube_transcripts_count": 0,
},
)
assert "youtube" in q["core_active"]
assert q["score_pct"] == 100 # Full active count regardless of degradation
# But nudge still fires
assert q["nudge_text"] is not None
assert "Degraded: YouTube" in q["nudge_text"]
+85
View File
@@ -552,5 +552,90 @@ class DegradedRunBannerTests(unittest.TestCase):
self.assertIn("--plan", text) self.assertIn("--plan", text)
class YoutubeFooterTranscriptRatioTests(unittest.TestCase):
"""The YouTube footer line must surface the transcript-fetch ratio in all
cases where videos were returned. Pre-fix the segment was suppressed when
transcripts == 0, which converted the canonical stale-yt-dlp failure mode
into a silent absence at the footer (the very surface users read for
'did this work?'). Always-render the ratio so zero is loud.
"""
def _build_youtube_report(self, transcript_flags: list[bool]) -> schema.Report:
"""Build a Report with one YouTube item per entry in transcript_flags.
True means the item has transcript data; False means it does not.
"""
items = []
for idx, has_transcript in enumerate(transcript_flags):
metadata = {"views": 1000}
if has_transcript:
metadata["transcript_highlights"] = ["Some pre-extracted quote."]
items.append(schema.SourceItem(
item_id=f"yt{idx}",
source="youtube",
title=f"Video {idx}",
body=f"Description for video {idx}.",
url=f"https://youtube.com/watch?v=v{idx}",
container="some-channel",
published_at="2026-04-15",
date_confidence="high",
engagement={"views": 1000, "likes": 100},
metadata=metadata,
))
return schema.Report(
topic="test topic",
range_from="2026-04-01",
range_to="2026-05-01",
generated_at="2026-05-01T00:00:00+00:00",
provider_runtime=schema.ProviderRuntime(
reasoning_provider="gemini",
planner_model="gemini",
rerank_model="gemini",
),
query_plan=schema.QueryPlan(
intent="general",
freshness_mode="balanced_recent",
cluster_mode="none",
raw_topic="test topic",
subqueries=[schema.SubQuery(
label="primary", search_query="test topic",
ranking_query="What about test topic?", sources=["youtube"],
)],
source_weights={"youtube": 1.0},
),
clusters=[],
ranked_candidates=[],
items_by_source={"youtube": items},
errors_by_source={},
)
def test_zero_transcripts_with_videos_present_renders_zero_over_total(self):
# The canonical stale-yt-dlp case: 6 videos found, 0 transcripts captured.
# Pre-fix the footer hid this entirely; post-fix it must say "0/6 with transcripts".
report = self._build_youtube_report([False] * 6)
text = render.render_compact(report)
self.assertIn("0/6 with transcripts", text)
def test_partial_transcripts_renders_ratio(self):
# 5 of 6 transcripts captured - shows ratio so user knows one was missed.
report = self._build_youtube_report([True] * 5 + [False])
text = render.render_compact(report)
self.assertIn("5/6 with transcripts", text)
def test_full_transcripts_renders_ratio(self):
# All 3 transcripts captured - still shows ratio for consistency.
report = self._build_youtube_report([True] * 3)
text = render.render_compact(report)
self.assertIn("3/3 with transcripts", text)
def test_no_videos_no_transcript_segment(self):
# When YouTube has no items at all, the YouTube footer line is
# suppressed entirely (existing behavior) - the transcript segment
# should not appear without a parent line.
report = self._build_youtube_report([])
text = render.render_compact(report)
# No YouTube footer line at all - so no transcript segment either
self.assertNotIn("with transcripts", text)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()