fix: expand entity-grounding haystack to transcripts + top comments
PR #285's entity grounding checked only title + snippet. That missed: - YouTube videos where the entity is mentioned in transcript but not in title (false demotion of on-topic content) - Reddit posts where the entity is in top comments but not in title (false demotion of on-topic discussion) And it also wasn't strong enough to reliably demote items like the 2026-04-19 Nate Herk "Managed Agents" video - which had no Hermes anywhere - because the -25 penalty on rerank_score composed to only -15 on final_score via the 0.60 weight, and engagement bonus partially offset that. Two fixes: 1. _candidate_haystack() now joins title + snippet + metadata[transcript_snippet] + metadata[transcript_highlights] + metadata[top_comments][*].excerpt/text + metadata[comment_insights]. Catches entity mentions wherever they actually live. Guarded with isinstance checks so malformed metadata doesn't raise. 2. ENTITY_MISS_FINAL_PENALTY (20.0) applied directly in _final_score when candidate.explanation contains "entity-miss". This lands the full penalty weight on the composite signal that cluster-scoring consumes, instead of being diluted by the rerank_score weight. Combined effect: entity-miss gap grows from ~15 to ~35 points. Tests: 8 new scenarios covering transcript match, transcript highlight match, top-comment match, comment-insight match, empty-text skip, no-primary-entity no-op, and the dual-penalty composition check.
This commit is contained in:
@@ -256,5 +256,133 @@ class EntityGroundingTests(unittest.TestCase):
|
||||
self.assertNotIn("Primary entity grounding", prompt)
|
||||
|
||||
|
||||
class ExpandedHaystackTests(unittest.TestCase):
|
||||
"""Unit 3: Entity-grounding haystack covers transcript snippets,
|
||||
transcript highlights, top comments, and comment insights - not
|
||||
just title + snippet.
|
||||
"""
|
||||
|
||||
def _youtube_candidate(self, title: str, transcript_snippet: str = "",
|
||||
transcript_highlights: list[str] | None = None) -> schema.Candidate:
|
||||
c = schema.Candidate(
|
||||
candidate_id=f"c-{title[:10]}",
|
||||
item_id="i1",
|
||||
source="youtube",
|
||||
title=title,
|
||||
url="https://youtube.com/watch?v=x",
|
||||
snippet="",
|
||||
subquery_labels=["primary"],
|
||||
native_ranks={"primary:youtube": 1},
|
||||
local_relevance=0.8,
|
||||
freshness=80,
|
||||
engagement=50,
|
||||
source_quality=0.7,
|
||||
rrf_score=0.02,
|
||||
)
|
||||
c.metadata = {}
|
||||
if transcript_snippet:
|
||||
c.metadata["transcript_snippet"] = transcript_snippet
|
||||
if transcript_highlights:
|
||||
c.metadata["transcript_highlights"] = transcript_highlights
|
||||
return c
|
||||
|
||||
def test_entity_found_in_transcript_snippet_avoids_demotion(self):
|
||||
# Title + snippet miss the entity, but the transcript contains it.
|
||||
c = self._youtube_candidate(
|
||||
"Weekly roundup",
|
||||
transcript_snippet="In this video I walk through using Hermes Agent in production.",
|
||||
)
|
||||
rerank._apply_fallback_scores([c], primary_entity="Hermes Agent")
|
||||
self.assertEqual("fallback-local-score", c.explanation)
|
||||
|
||||
def test_entity_found_in_transcript_highlights_avoids_demotion(self):
|
||||
c = self._youtube_candidate(
|
||||
"Some review",
|
||||
transcript_highlights=[
|
||||
"Today we're talking about Hermes Agent",
|
||||
"Let's compare it to the alternatives",
|
||||
],
|
||||
)
|
||||
rerank._apply_fallback_scores([c], primary_entity="Hermes Agent")
|
||||
self.assertEqual("fallback-local-score", c.explanation)
|
||||
|
||||
def test_entity_missing_everywhere_still_demoted_for_video(self):
|
||||
# Nate Herk "Managed Agents" case: no Hermes in title, snippet,
|
||||
# or transcript - demotion fires.
|
||||
c = self._youtube_candidate(
|
||||
"I Tested Claude's New Managed Agents",
|
||||
transcript_snippet="Managed agents are Anthropic's new product with ClickUp and cron...",
|
||||
)
|
||||
rerank._apply_fallback_scores([c], primary_entity="Hermes Agent")
|
||||
self.assertIn("entity-miss", c.explanation)
|
||||
|
||||
def test_entity_found_in_reddit_top_comments_avoids_demotion(self):
|
||||
c = schema.Candidate(
|
||||
candidate_id="r1",
|
||||
item_id="i1",
|
||||
source="reddit",
|
||||
title="Best agent framework?",
|
||||
url="https://reddit.com/r/x",
|
||||
snippet="",
|
||||
subquery_labels=["primary"],
|
||||
native_ranks={"primary:reddit": 1},
|
||||
local_relevance=0.8, freshness=80, engagement=50,
|
||||
source_quality=0.7, rrf_score=0.02,
|
||||
)
|
||||
c.metadata = {
|
||||
"top_comments": [
|
||||
{"excerpt": "I've been using Hermes Agent for a month and it's great"},
|
||||
{"text": "another comment"},
|
||||
],
|
||||
}
|
||||
rerank._apply_fallback_scores([c], primary_entity="Hermes Agent")
|
||||
self.assertEqual("fallback-local-score", c.explanation)
|
||||
|
||||
def test_entity_found_in_comment_insights_avoids_demotion(self):
|
||||
c = schema.Candidate(
|
||||
candidate_id="r2", item_id="i1", source="reddit",
|
||||
title="AI tools", url="https://reddit.com/r/x", snippet="",
|
||||
subquery_labels=["primary"],
|
||||
native_ranks={"primary:reddit": 1},
|
||||
local_relevance=0.8, freshness=80, engagement=50,
|
||||
source_quality=0.7, rrf_score=0.02,
|
||||
)
|
||||
c.metadata = {
|
||||
"comment_insights": ["Consensus: Hermes Agent handles long sessions best"],
|
||||
}
|
||||
rerank._apply_fallback_scores([c], primary_entity="Hermes Agent")
|
||||
self.assertEqual("fallback-local-score", c.explanation)
|
||||
|
||||
def test_truly_empty_candidate_still_skipped(self):
|
||||
# Image-only TikTok with no text anywhere - do not penalize.
|
||||
c = self._youtube_candidate("") # empty title
|
||||
rerank._apply_fallback_scores([c], primary_entity="Hermes Agent")
|
||||
self.assertEqual("fallback-local-score", c.explanation)
|
||||
|
||||
def test_final_score_secondary_penalty_applied_on_entity_miss(self):
|
||||
# When fallback flags entity-miss, final_score gets an ADDITIONAL
|
||||
# -20 penalty beyond the rerank_score reduction. Verify by
|
||||
# comparing final_score for a demoted candidate vs an identical
|
||||
# candidate that matched the entity.
|
||||
off_topic = self._youtube_candidate("Managed Agents from Anthropic")
|
||||
on_topic = self._youtube_candidate(
|
||||
"Hermes Agent walkthrough",
|
||||
transcript_snippet="Hermes Agent review",
|
||||
)
|
||||
rerank._apply_fallback_scores([off_topic, on_topic], primary_entity="Hermes Agent")
|
||||
# Gap should be well above the rerank_score-only path's 0.60 * 25 = 15;
|
||||
# with the secondary penalty it's 15 + 20 = 35 points.
|
||||
gap = on_topic.final_score - off_topic.final_score
|
||||
self.assertGreater(gap, 25.0,
|
||||
f"entity-miss demotion gap only {gap:.1f}; secondary penalty may not be firing")
|
||||
|
||||
def test_secondary_penalty_not_applied_when_entity_match(self):
|
||||
on_topic = self._youtube_candidate("Hermes Agent: use cases")
|
||||
rerank._apply_fallback_scores([on_topic], primary_entity="Hermes Agent")
|
||||
# Explanation does NOT contain entity-miss, so secondary penalty
|
||||
# should not fire; final_score reflects only base signal.
|
||||
self.assertNotIn("entity-miss", on_topic.explanation or "")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
unittest.main()
|
||||
|
||||
Reference in New Issue
Block a user