feat(quality): YouTube relevance scoring and cross-source linking

YouTube videos now get real relevance scores based on token overlap
between the search query and video title (was hardcoded at 0.7).
Uses ratio overlap with stopword removal, floored at 0.1.

Cross-source linking annotates items that discuss the same story
across different platforms (e.g., Reddit + HN + X). Items get
bidirectional cross_refs displayed as [xref: R3, HN5] in compact
output so Claude can triangulate multi-platform coverage.

Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
Matt Van Horn
2026-02-25 10:46:58 -08:00
parent f60a4359a0
commit 0591f55f0e
8 changed files with 642 additions and 14 deletions
+35 -3
View File
@@ -17,7 +17,7 @@ import sys
import tempfile
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
from typing import Any, Dict, List, Optional, Tuple
from typing import Any, Dict, List, Optional, Set, Tuple
# Depth configurations: how many videos to search / transcribe
DEPTH_CONFIG = {
@@ -35,6 +35,38 @@ TRANSCRIPT_LIMITS = {
# Max words to keep from each transcript
TRANSCRIPT_MAX_WORDS = 500
# Stopwords for relevance computation (common English words that dilute token overlap)
STOPWORDS = frozenset({
'the', 'a', 'an', 'to', 'for', 'how', 'is', 'in', 'of', 'on',
'and', 'with', 'from', 'by', 'at', 'this', 'that', 'it', 'my',
'your', 'i', 'me', 'we', 'you', 'what', 'are', 'do', 'can',
'its', 'be', 'or', 'not', 'no', 'so', 'if', 'but', 'about',
'all', 'just', 'get', 'has', 'have', 'was', 'will',
})
def _tokenize(text: str) -> Set[str]:
"""Lowercase, strip punctuation, remove stopwords, drop single-char tokens."""
words = re.sub(r'[^\w\s]', ' ', text.lower()).split()
return {w for w in words if w not in STOPWORDS and len(w) > 1}
def _compute_relevance(query: str, title: str) -> float:
"""Compute relevance as ratio of query tokens found in title.
Uses ratio overlap (intersection / query_length) so short queries
score higher when fully represented in the title. Floors at 0.1.
"""
q_tokens = _tokenize(query)
t_tokens = _tokenize(title)
if not q_tokens:
return 0.5 # Neutral fallback for empty/stopword-only queries
overlap = len(q_tokens & t_tokens)
ratio = overlap / len(q_tokens)
return max(0.1, min(1.0, ratio))
def _log(msg: str):
"""Log to stderr."""
@@ -183,8 +215,8 @@ def search_youtube(
"comments": comment_count,
},
"duration": video.get("duration"),
"relevance": 0.7, # Default; no LLM relevance scoring for YouTube
"why_relevant": f"YouTube video about {core_topic}",
"relevance": _compute_relevance(core_topic, video.get("title", "")),
"why_relevant": f"YouTube: {video.get('title', core_topic)[:60]}",
})
# Soft date filter: prefer recent items but fall back to all if too few