From 1bdc14878ccf9e10d33ab8d109b6b1ca9c935e23 Mon Sep 17 00:00:00 2001 From: Matt Van Horn Date: Sat, 6 Jun 2026 09:44:07 -0700 Subject: [PATCH] fix(reddit): relevance-aware comment-enrichment slot selection in keyless path (#484) * fix(reddit): relevance-aware comment-enrichment slot selection in keyless path * docs(changelog): record relevance-aware enrichment fix under Unreleased --------- Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com> --- CHANGELOG.md | 4 + .../last30days/scripts/lib/reddit_keyless.py | 50 +++++++++- tests/test_reddit_keyless.py | 98 +++++++++++++++++++ 3 files changed, 148 insertions(+), 4 deletions(-) diff --git a/CHANGELOG.md b/CHANGELOG.md index e16cf2f..5b89211 100644 --- a/CHANGELOG.md +++ b/CHANGELOG.md @@ -7,6 +7,10 @@ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0 ## [Unreleased] +### Fixed + +- Keyless Reddit comment enrichment now spends its limited slots on entity-matching posts first (mirroring rerank's entity-miss demotion signal) instead of raw upvote order, so off-topic high-upvote threads from broad subreddits no longer consume the comment budget only to be demoted afterward ([#484](https://github.com/mvanhorn/last30days-skill/pull/484)) + ## [3.3.1] - 2026-05-30 ### Fixed diff --git a/skills/last30days/scripts/lib/reddit_keyless.py b/skills/last30days/scripts/lib/reddit_keyless.py index b1ced97..9879c10 100644 --- a/skills/last30days/scripts/lib/reddit_keyless.py +++ b/skills/last30days/scripts/lib/reddit_keyless.py @@ -163,6 +163,45 @@ def _enrich(posts: List[Dict[str, Any]], depth: str) -> List[Dict[str, Any]]: return enriched + rest +def _slot_priority(topic: str, posts: List[Dict[str, Any]]) -> List[Dict[str, Any]]: + """Order posts for enrichment slots: entity-matching posts first. + + Comment slots (ENRICH_LIMITS) are scarce; spending them on high-upvote + posts that rerank later demotes as entity misses starves the on-topic + posts the user actually sees (2026-06-06 "OpenClaw vs Hermes" run: + 2,000+ upvote Gemma/GPU threads took every slot, then were demoted to + zero). Mirror rerank's demotion signal — the topic's stripped primary + entity contained in the post text — so slots go to posts likely to + survive final ranking. Falls back to token-overlap relevance when the + topic yields no usable primary entity. Within each tier the incoming + (score-first) order is preserved. Never raises; on any failure the + incoming order is returned unchanged. + """ + try: + from . import relevance, rerank + + def _post_text(post: Dict[str, Any]) -> str: + return f"{post.get('title') or ''} {post.get('selftext') or ''}" + + entity = rerank._primary_entity(topic).lower() + if entity: + def _matches(post: Dict[str, Any]) -> bool: + return entity in _post_text(post).lower() + else: + prepared = relevance.PreparedQuery(topic) + + def _matches(post: Dict[str, Any]) -> bool: + return relevance.token_overlap_relevance(prepared, _post_text(post)) > 0.24 + + matches: List[Dict[str, Any]] = [] + misses: List[Dict[str, Any]] = [] + for post in posts: + (matches if _matches(post) else misses).append(post) + return matches + misses + except Exception: + return posts + + def search_and_enrich( topic: str, from_date: str, @@ -194,9 +233,9 @@ def search_and_enrich( if p.get("date") is None or (from_date <= p["date"] <= to_date) ] - # Rank before enrichment by real upvote score (from listing cards / backfill), - # then query relevance, then recency. Posts without a recovered score sort by - # the latter two — same behavior as before scores were available. + # Rank by real upvote score (from listing cards / backfill), then query + # relevance, then recency. Posts without a recovered score sort by the + # latter two — same behavior as before scores were available. posts.sort( key=lambda p: ( p.get("engagement", {}).get("score", 0) or 0, @@ -206,7 +245,10 @@ def search_and_enrich( reverse=True, ) - posts = _enrich(posts, depth) + # Enrichment slot selection is relevance-aware: entity-matching posts + # claim the scarce comment slots first (score order preserved within + # each tier). The score-first sort above still governs within-tier order. + posts = _enrich(_slot_priority(topic, posts), depth) for i, post in enumerate(posts): post["id"] = f"R{i + 1}" diff --git a/tests/test_reddit_keyless.py b/tests/test_reddit_keyless.py index 26de05c..0a36155 100644 --- a/tests/test_reddit_keyless.py +++ b/tests/test_reddit_keyless.py @@ -166,3 +166,101 @@ class TestSearchAndEnrich: reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31", depth="quick") # quick depth enriches only top 3 posts assert fc.call_count == reddit_keyless.ENRICH_LIMITS["quick"] + + +class TestSlotPriority: + """Enrichment slot selection prefers entity-matching posts (R1-R3).""" + + @staticmethod + def _titled(i, title, score=0, selftext=""): + p = _post(i) + p["title"] = title + p["selftext"] = selftext + p["score"] = score + p["engagement"]["score"] = score + return p + + def test_on_topic_low_score_beats_off_topic_high_score(self): + # 3 off-topic monsters + 2 on-topic small threads; quick depth = 3 slots. + posts = [ + self._titled(1, "Stop asking what model to run", score=2662), + self._titled(2, "RTX 4090 PSA", score=2068), + self._titled(3, "Gemma 4 release", score=997), + self._titled(4, "My OpenClaw self-migrated", score=73), + self._titled(5, "Using openclaw with Claude API key is so expensive", score=47), + ] + enriched_urls = [] + + def _capture(url): + enriched_urls.append(url) + return {"top_comments": [], "comment_insights": [], "num_comments": None} + + with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ + mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", + side_effect=_capture): + reddit_keyless.search_and_enrich( + "openclaw", "2026-05-01", "2026-05-31", depth="quick") + assert posts[3]["url"] in enriched_urls + assert posts[4]["url"] in enriched_urls + assert len(enriched_urls) == reddit_keyless.ENRICH_LIMITS["quick"] + + def test_multiword_topic_uses_substring_not_token_overlap(self): + # "claude tips" clears token overlap for "Claude Code" but rerank + # demotes it; the partition must mirror rerank's substring test. + token_only = self._titled(1, "claude tips", score=500) + full_entity = self._titled(2, "Claude Code best setup", score=5) + out = reddit_keyless._slot_priority("Claude Code", [token_only, full_entity]) + assert out[0] is full_entity + assert out[1] is token_only + + def test_intent_modifier_stripped_from_topic(self): + on_topic = self._titled(1, "Hermes Agent v0.13 is great", score=1) + off_topic = self._titled(2, "Hermes Birkin unboxing", score=900) + out = reddit_keyless._slot_priority("Hermes Agent review", [off_topic, on_topic]) + assert out[0] is on_topic + + def test_all_miss_keeps_score_order_and_full_slots(self): + posts = [self._titled(i, f"Gemma thread {i}", score=1000 - i) for i in range(5)] + out = reddit_keyless._slot_priority("openclaw", posts) + assert out == posts # order unchanged + with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ + mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", + return_value={"top_comments": [], "comment_insights": [], + "num_comments": None}) as fc: + reddit_keyless.search_and_enrich( + "openclaw", "2026-05-01", "2026-05-31", depth="quick") + assert fc.call_count == reddit_keyless.ENRICH_LIMITS["quick"] + + def test_same_tier_order_preserved(self): + posts = [self._titled(i, f"openclaw thread {i}", score=100 - i) for i in range(4)] + out = reddit_keyless._slot_priority("openclaw", posts) + assert out == posts + + def test_empty_entity_falls_back_to_token_overlap(self): + # Pure intent-modifier topic yields no primary entity; fallback path + # must not raise and must keep every post. + posts = [self._titled(1, "Post one"), self._titled(2, "review of things")] + out = reddit_keyless._slot_priority("review", posts) + assert len(out) == 2 + assert {p["url"] for p in out} == {p["url"] for p in posts} + + def test_selftext_match_lands_in_match_tier(self): + body_match = self._titled(1, "Need help with my setup", score=2, + selftext="my openclaw agent keeps asking for ssh keys") + off_topic = self._titled(2, "Gemma 4 with QAT", score=700) + out = reddit_keyless._slot_priority("openclaw", [off_topic, body_match]) + assert out[0] is body_match + + def test_none_score_posts_do_not_break_partition(self): + p1 = self._titled(1, "openclaw tips") + p1["engagement"]["score"] = None + p2 = self._titled(2, "Gemma news") + p2["engagement"]["score"] = None + out = reddit_keyless._slot_priority("openclaw", [p2, p1]) + assert out[0] is p1 + + def test_partition_never_raises(self): + posts = [self._titled(1, "openclaw tips", score=1)] + with mock.patch("lib.rerank._primary_entity", side_effect=Exception("boom")): + out = reddit_keyless._slot_priority("openclaw", posts) + assert out == posts