feat(youtube): extract transcript highlights like Reddit comment gems
Add extract_transcript_highlights() that scores sentences by specificity (numbers, proper nouns, topic relevance) and filters YouTube filler (subscribe, welcome back, etc). Top 5 highlights shown as structured bullets in compact output. Full transcript moved to collapsible <details> block so the LLM reads highlights first, full text on demand. SKILL.md updated to instruct the judge agent to quote highlights directly in synthesis, same as Reddit top comments. Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -210,6 +210,7 @@ class YouTubeItem:
|
||||
date_confidence: str = "high" # YouTube dates are always reliable
|
||||
engagement: Optional[Engagement] = None
|
||||
transcript_snippet: str = ""
|
||||
transcript_highlights: List[str] = field(default_factory=list)
|
||||
relevance: float = 0.7
|
||||
why_relevant: str = ""
|
||||
subs: SubScores = field(default_factory=SubScores)
|
||||
@@ -226,6 +227,7 @@ class YouTubeItem:
|
||||
'date_confidence': self.date_confidence,
|
||||
'engagement': self.engagement.to_dict() if self.engagement else None,
|
||||
'transcript_snippet': self.transcript_snippet,
|
||||
'transcript_highlights': self.transcript_highlights,
|
||||
'relevance': self.relevance,
|
||||
'why_relevant': self.why_relevant,
|
||||
'subs': self.subs.to_dict(),
|
||||
@@ -655,6 +657,7 @@ class Report:
|
||||
date_confidence=y.get('date_confidence', 'high'),
|
||||
engagement=eng,
|
||||
transcript_snippet=y.get('transcript_snippet', ''),
|
||||
transcript_highlights=y.get('transcript_highlights', []),
|
||||
relevance=y.get('relevance', 0.7),
|
||||
why_relevant=y.get('why_relevant', ''),
|
||||
subs=subs,
|
||||
|
||||
Reference in New Issue
Block a user