feat: release v3.0.6 - promote plans 003-009 from private beta to public (#277)
Consolidates seven beta-validated plans into the public release. Validated on nine+ topics across GENERAL, COMPARISON, RECOMMENDATIONS, and demographic-shopping classes before ship. Plans bundled in this release: - 003 Engine-emitted Pre-Research Status warning + Polymarket summarization + VOICE CONTRACT LAW 1-5 + Step 0.55 MANDATORY - 004 WebSearch deferred-tool loading (ToolSearch STEP 0) + LAW 5 universal + top-of-file imperative - 005 Supplement floor (2-3 minimum) separate from Step 0.55 pre-research - 006 Step 2.5 MANDATORY raw-file append with canonical format example + count-equality self-check - 007 Restored April 9 canonical comparison template with Quick Verdict, per-entity Strengths/Weaknesses, 9-axis Head-to-Head, Bottom Line, emerging stack + LAW 2/4 COMPARISON exceptions - 008 Person-topic GitHub handle resolution MANDATORY + LAW 1 reinforcement at Step 2 tail and Step 2.5 entry + RECOMMENDATIONS signal-weighted ranking rewrite + Polymarket post-merge topic filter (engine change, filter_items_against_topic helper + vs/versus in _NOISE_WORDS) - 009 Unified pre-flight CHECKLIST + VOICE CONTRACT formatting-authority preface + Step 0.45 Query Quality Pre-Flight (4 keyword-trap classes) + post-synthesis Sources-block self-check Beta validation topics (2026-04-18): Kanye West, Matt Van Horn, CLI vs MCP, OpenClaw vs Paperclip vs Hermes, Paperclip vs Hermes vs Open Claw, Garry Tan, Israel vs Lebanon, Best programming language for AI agents, Peter Steinberger post plan 009, Birthday gift for 42 year old man (Class 1 pre-flight fired correctly), Vincent Koc (passed). No breaking changes. No new CLI flags. No new public API. Plugin name (last30days) and marketplace name (last30days-skill) unchanged. Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com> Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
This commit is contained in:
@@ -117,6 +117,9 @@ _NOISE_WORDS = frozenset({
|
||||
"software", "plugin", "skill", "agent", "bot", "search", "research",
|
||||
# Generic prediction market terms
|
||||
"market", "odds", "prediction", "forecast", "chance", "probability",
|
||||
# Comparison-query conjunctions — should not count as informative filter tokens
|
||||
# when the topic is "X vs Y vs Z"
|
||||
"vs", "versus",
|
||||
})
|
||||
|
||||
|
||||
@@ -165,6 +168,70 @@ def _passes_topic_filter(topic: str, event_title: str) -> bool:
|
||||
return match_count >= min_matches
|
||||
|
||||
|
||||
def _passes_any_informative_word(topic: str, event_title: str) -> bool:
|
||||
"""Looser variant of _passes_topic_filter that keeps an item if ANY
|
||||
informative word from the topic appears in the title.
|
||||
|
||||
Designed for post-merge validation of comparison topics (e.g., "OpenClaw vs
|
||||
Hermes vs Paperclip"), where a market mentioning just one of the entities
|
||||
is still on-topic. The stricter _passes_topic_filter (min_matches=2 for
|
||||
3+ informative words) is correct for single-entity topics like "Mill.com
|
||||
food recycler" but drops legitimate single-entity comparison results.
|
||||
"""
|
||||
core = _extract_core_subject(topic).lower()
|
||||
core_words = [w for w in re.sub(r"[^\w\s]", " ", core).split() if len(w) > 1]
|
||||
if not core_words:
|
||||
return True
|
||||
informative = [w for w in core_words if w not in _NOISE_WORDS]
|
||||
if not informative:
|
||||
return True
|
||||
|
||||
title_lower = " ".join(re.sub(r"[^\w\s]", " ", event_title.lower()).split())
|
||||
title_words = set(title_lower.split())
|
||||
|
||||
for word in informative:
|
||||
if word in title_words:
|
||||
return True
|
||||
if len(word) >= 4 and word in title_lower:
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def filter_items_against_topic(topic: str, items: List[Any]) -> List[Any]:
|
||||
"""Drop items whose title shares no informative word with the original topic.
|
||||
|
||||
Called post-merge from pipeline.py so per-entity subquery results for
|
||||
comparison topics get re-validated against the ORIGINAL full topic before
|
||||
landing in the footer. Prevents noise like WTI crude oil or Elon tweet
|
||||
markets from surviving a loose "Hermes" single-entity subquery match.
|
||||
|
||||
Uses the looser _passes_any_informative_word rule (ANY entity name match
|
||||
is sufficient) so a market mentioning just one of several compared entities
|
||||
still counts as on-topic.
|
||||
|
||||
Accepts a list of either raw dicts (with 'title') or SourceItem-like objects
|
||||
(with .title attribute). Returns the filtered list in the same order.
|
||||
"""
|
||||
if not topic:
|
||||
return items
|
||||
|
||||
filtered = []
|
||||
for item in items:
|
||||
title = getattr(item, "title", None)
|
||||
if title is None and isinstance(item, dict):
|
||||
title = item.get("title", "")
|
||||
title = title or ""
|
||||
|
||||
if _passes_any_informative_word(topic, title):
|
||||
filtered.append(item)
|
||||
|
||||
dropped = len(items) - len(filtered)
|
||||
if dropped:
|
||||
_log(f"Post-merge topic filter dropped {dropped} Polymarket items against full topic '{topic}'")
|
||||
|
||||
return filtered
|
||||
|
||||
|
||||
def _extract_domain_queries(topic: str, events: List[Dict]) -> List[str]:
|
||||
"""Extract domain-indicator search terms from first-pass event tags.
|
||||
|
||||
|
||||
Reference in New Issue
Block a user