fix: demote reranker candidates that miss the primary entity

The 2026-04-19 Hermes Agent Use Cases run had a Nate Herk YouTube video
titled "I Tested Claude's New Managed Agents" score 51 and rank #2
with zero Hermes content. The reranker had intent-specific scoring hints
but no entity-grounding check, so topic-vicinity matches (one offhand
OpenClaw mention) drifted to the top.

Add _primary_entity(topic) that strips intent-modifier suffixes ("use
cases", "workflows", etc.) so "Hermes Agent use cases" yields
primary_entity="Hermes Agent". Pass the entity through to both the LLM
and fallback scoring paths.

Fallback path: if primary_entity is not found (case-insensitive) in
title + snippet, subtract ENTITY_MISS_PENALTY (25 pts). Skip the
demotion for candidates with no text at all (image-only TikToks etc.)
to avoid false negatives on thin-text sources.

LLM path: add a "Primary entity grounding" hint to _build_prompt when
primary_entity is non-empty. Instructs the LLM to score candidates
without the entity at <=30.

Tests: 24 rerank tests pass, including 8 new entity-grounding tests.
This commit is contained in:
Matt Van Horn
2026-04-19 09:24:43 -07:00
parent 4d9f29d2ed
commit a709d66e2a
2 changed files with 147 additions and 12 deletions
+72 -11
View File
@@ -3,8 +3,34 @@
from __future__ import annotations from __future__ import annotations
import json import json
import re
from . import http, providers, schema from . import http, providers, query, schema
# Penalty applied when a candidate does not mention the primary entity
# from the topic in its title or snippet. Picked empirically: a typical
# score spread in the shortlist is 30-70, so 25 points reliably pushes
# an off-topic candidate below on-topic ones without fully zeroing out
# marginal matches. See 2026-04-19 Hermes Agent Use Cases failure: a
# Nate Herk "Managed Agents" video scored 51 / ranked #2 with zero
# Hermes content.
ENTITY_MISS_PENALTY = 25.0
# Intent modifiers to strip before extracting the primary entity so that,
# for example, "Hermes Agent use cases" yields primary_entity="hermes agent"
# rather than "hermes agent use cases". Kept in sync with
# planner._INTENT_MODIFIER_PATTERNS.
_INTENT_MODIFIER_RE = re.compile(
r"\b("
r"use cases|use case|workflows|workflow|"
r"examples|example|tutorial|tutorials|"
r"review|reviews|comparison|applications|"
r"in practice|production use|production|"
r"how i use"
r")\b",
re.IGNORECASE,
)
INTENT_SCORING_HINTS: dict[str, str] = { INTENT_SCORING_HINTS: dict[str, str] = {
"comparison": ( "comparison": (
@@ -60,20 +86,21 @@ def rerank_candidates(
) -> list[schema.Candidate]: ) -> list[schema.Candidate]:
"""Rerank the fused shortlist, demoting candidates the reranker scored as irrelevant.""" """Rerank the fused shortlist, demoting candidates the reranker scored as irrelevant."""
shortlisted = candidates[:shortlist_size] shortlisted = candidates[:shortlist_size]
primary_entity = _primary_entity(topic)
if provider and model and shortlisted: if provider and model and shortlisted:
try: try:
response = provider.generate_json(model, _build_prompt(topic, plan, shortlisted)) response = provider.generate_json(model, _build_prompt(topic, plan, shortlisted, primary_entity))
_apply_llm_scores(shortlisted, response) _apply_llm_scores(shortlisted, response)
except (ValueError, KeyError, json.JSONDecodeError, OSError, http.HTTPError) as exc: except (ValueError, KeyError, json.JSONDecodeError, OSError, http.HTTPError) as exc:
import sys import sys
print(f"[Rerank] LLM reranking failed, using local fallback: {type(exc).__name__}: {exc}", file=sys.stderr) print(f"[Rerank] LLM reranking failed, using local fallback: {type(exc).__name__}: {exc}", file=sys.stderr)
_apply_fallback_scores(shortlisted) _apply_fallback_scores(shortlisted, primary_entity=primary_entity)
else: else:
_apply_fallback_scores(shortlisted) _apply_fallback_scores(shortlisted, primary_entity=primary_entity)
if len(candidates) > shortlist_size: if len(candidates) > shortlist_size:
tail = candidates[shortlist_size:] tail = candidates[shortlist_size:]
_apply_fallback_scores(tail) _apply_fallback_scores(tail, primary_entity=primary_entity)
return sorted( return sorted(
candidates, candidates,
@@ -103,7 +130,7 @@ def _fenced_untrusted_content(candidate_block: str) -> str:
) )
def _build_prompt(topic: str, plan: schema.QueryPlan, candidates: list[schema.Candidate]) -> str: def _build_prompt(topic: str, plan: schema.QueryPlan, candidates: list[schema.Candidate], primary_entity: str = "") -> str:
ranking_queries = "\n".join( ranking_queries = "\n".join(
f"- {subquery.label}: {subquery.ranking_query}" f"- {subquery.label}: {subquery.ranking_query}"
for subquery in plan.subqueries for subquery in plan.subqueries
@@ -121,6 +148,16 @@ def _build_prompt(topic: str, plan: schema.QueryPlan, candidates: list[schema.Ca
) )
for candidate in candidates for candidate in candidates
) )
grounding_hint = ""
if primary_entity:
grounding_hint = (
f"\nPrimary entity grounding: the user's primary entity is \"{primary_entity}\". "
"A candidate that does NOT mention this entity (or a clear synonym/abbreviation) "
"in its title or snippet should score no higher than 30, regardless of other "
"signals. Do not let a candidate match the topic vicinity without matching the "
"entity itself. 2026-04-19 Hermes Agent Use Cases failure: a Nate Herk video "
"about Claude's Managed Agents scored 51 with zero Hermes content.\n"
)
return f""" return f"""
Judge search-result relevance for a last-30-days research pipeline. Judge search-result relevance for a last-30-days research pipeline.
@@ -145,7 +182,7 @@ Scoring guidance:
- 70 to 89: clearly relevant and useful - 70 to 89: clearly relevant and useful
- 40 to 69: somewhat relevant but weaker - 40 to 69: somewhat relevant but weaker
- 0 to 39: weak, redundant, or off-target - 0 to 39: weak, redundant, or off-target
{_intent_hint_block(plan)} {grounding_hint}{_intent_hint_block(plan)}
{_fenced_untrusted_content(candidate_block)} {_fenced_untrusted_content(candidate_block)}
""".strip() """.strip()
@@ -169,21 +206,45 @@ def _apply_llm_scores(candidates: list[schema.Candidate], payload: dict) -> None
candidate.final_score = _final_score(candidate) candidate.final_score = _final_score(candidate)
def _apply_fallback_scores(candidates: list[schema.Candidate]) -> None: def _apply_fallback_scores(candidates: list[schema.Candidate], *, primary_entity: str = "") -> None:
for candidate in candidates: for candidate in candidates:
rerank_score, reason = _fallback_tuple(candidate) rerank_score, reason = _fallback_tuple(candidate, primary_entity=primary_entity)
candidate.rerank_score = rerank_score candidate.rerank_score = rerank_score
candidate.explanation = reason candidate.explanation = reason
candidate.final_score = _final_score(candidate) candidate.final_score = _final_score(candidate)
def _fallback_tuple(candidate: schema.Candidate) -> tuple[float, str]: def _fallback_tuple(candidate: schema.Candidate, *, primary_entity: str = "") -> tuple[float, str]:
score = ( score = (
(candidate.local_relevance * 100.0 * 0.7) (candidate.local_relevance * 100.0 * 0.7)
+ (candidate.freshness * 0.2) + (candidate.freshness * 0.2)
+ (candidate.source_quality * 100.0 * 0.1) + (candidate.source_quality * 100.0 * 0.1)
) )
return max(0.0, min(100.0, score)), "fallback-local-score" reason = "fallback-local-score"
# Entity-grounding demotion: if the primary entity (topic minus intent
# modifier) is not present in the candidate's title or snippet, subtract
# ENTITY_MISS_PENALTY. Skip for candidates with no text at all (e.g.,
# image-only TikToks) to avoid penalizing thin-text sources unfairly.
if primary_entity and (candidate.title or candidate.snippet):
haystack = f"{candidate.title} {candidate.snippet}".lower()
if primary_entity.lower() not in haystack:
score -= ENTITY_MISS_PENALTY
reason = "fallback-local-score (entity-miss demotion)"
return max(0.0, min(100.0, score)), reason
def _primary_entity(topic: str) -> str:
"""Extract the primary entity from the topic for grounding checks.
Strips intent-modifier suffixes (see planner._INTENT_MODIFIER_PATTERNS),
trims trailing punctuation, collapses whitespace. Returns the empty
string for topics that are all intent modifier with no entity, so
callers can skip the grounding check.
"""
stripped = _INTENT_MODIFIER_RE.sub(" ", topic)
# Also collapse multiple spaces and strip punctuation.
stripped = re.sub(r"\s+", " ", stripped).strip(" \t\r\n?.,:;!")
return stripped
def _final_score(candidate: schema.Candidate) -> float: def _final_score(candidate: schema.Candidate) -> float:
+75 -1
View File
@@ -178,9 +178,83 @@ class RerankV3Tests(unittest.TestCase):
self.assertEqual("gemini-3.1-flash-lite-preview", provider.model) self.assertEqual("gemini-3.1-flash-lite-preview", provider.model)
self.assertEqual(95.0, first.rerank_score) self.assertEqual(95.0, first.rerank_score)
self.assertEqual("high fit", first.explanation) self.assertEqual("high fit", first.explanation)
self.assertEqual("fallback-local-score", second.explanation) # Tail is scored via the fallback (may or may not carry the entity-miss
# suffix depending on topic-title overlap; assert the base tag is present).
self.assertIn("fallback-local-score", second.explanation or "")
self.assertEqual(first.candidate_id, ranked[0].candidate_id) self.assertEqual(first.candidate_id, ranked[0].candidate_id)
class EntityGroundingTests(unittest.TestCase):
"""Unit 4: Reranker entity-grounding demotion. 2026-04-19 Hermes Agent
Use Cases failure: an off-topic video about Claude Managed Agents
scored 51 and ranked #2 with zero Hermes content.
"""
def _candidate(self, title: str, snippet: str = "") -> schema.Candidate:
return schema.Candidate(
candidate_id=f"c-{title[:10]}",
item_id="i1",
source="youtube",
title=title,
url="https://example.com",
snippet=snippet,
subquery_labels=["primary"],
native_ranks={"primary:youtube": 1},
local_relevance=0.8,
freshness=80,
engagement=50,
source_quality=0.7,
rrf_score=0.02,
)
def test_primary_entity_strips_intent_modifier(self):
self.assertEqual("Hermes Agent", rerank._primary_entity("Hermes Agent use cases"))
self.assertEqual("Hermes Agent Actual", rerank._primary_entity("Hermes Agent Actual Use Cases"))
self.assertEqual("Claude Code", rerank._primary_entity("Claude Code workflows"))
self.assertEqual("DSPy", rerank._primary_entity("DSPy tutorial"))
def test_primary_entity_leaves_bare_entity_unchanged(self):
self.assertEqual("Kanye West", rerank._primary_entity("Kanye West"))
self.assertEqual("Nous Research", rerank._primary_entity("Nous Research"))
def test_fallback_demotes_candidate_without_primary_entity(self):
on_topic = self._candidate("Hermes Agent: Self-Improving AI", "Nous Research Hermes walkthrough")
off_topic = self._candidate("I Tested Claude's Managed Agents", "What you need to know about Anthropic's new managed agents")
rerank._apply_fallback_scores([on_topic, off_topic], primary_entity="Hermes Agent")
self.assertGreater(on_topic.final_score, off_topic.final_score)
self.assertIn("entity-miss", off_topic.explanation or "")
self.assertEqual(on_topic.explanation, "fallback-local-score")
def test_fallback_match_is_case_insensitive(self):
on_topic = self._candidate("HERMES agent rocks", "some text")
rerank._apply_fallback_scores([on_topic], primary_entity="Hermes Agent")
self.assertEqual("fallback-local-score", on_topic.explanation)
def test_fallback_skips_demotion_for_empty_text_candidates(self):
empty = self._candidate("", "")
rerank._apply_fallback_scores([empty], primary_entity="Hermes Agent")
self.assertEqual("fallback-local-score", empty.explanation)
def test_fallback_skips_demotion_when_no_primary_entity(self):
off = self._candidate("Completely unrelated", "snippet")
rerank._apply_fallback_scores([off], primary_entity="")
self.assertEqual("fallback-local-score", off.explanation)
def test_llm_prompt_includes_primary_entity_grounding_hint(self):
candidate = self._candidate("Something", "snippet text")
plan = make_plan()
prompt = rerank._build_prompt(
"Hermes Agent use cases", plan, [candidate], primary_entity="Hermes Agent"
)
self.assertIn("Primary entity grounding", prompt)
self.assertIn("Hermes Agent", prompt)
def test_llm_prompt_omits_grounding_hint_when_no_primary_entity(self):
candidate = self._candidate("Something", "snippet text")
plan = make_plan()
prompt = rerank._build_prompt("", plan, [candidate], primary_entity="")
self.assertNotIn("Primary entity grounding", prompt)
if __name__ == "__main__": if __name__ == "__main__":
unittest.main() unittest.main()