949bcf8942
* feat: vs mode runs N full passes; --competitors wraps vs with auto-discovery
Unifies vs-mode and --competitors onto one fanout architecture. A topic
containing "vs" / "versus" now runs N full pipeline.run() calls in parallel
(reverting the one-pass latency optimization that removed per-entity
depth); --competitors becomes a SKILL.md-level shortcut where the hosting
reasoning model (Claude Code, Codex, Hermes, Gemini) discovers N peers via
its own WebSearch, runs Step 0.55 per entity, and invokes the engine with
a vs-topic + --competitors-plan JSON.
Changed:
- vs-mode: N full passes in parallel via fanout (was 1 merged pass).
- --competitors: SKILL.md shortcut for vs-mode-with-discovery. Engine flag
kept for headless/cron use. LAW 7-style stderr reframed to lead with the
hosting-model path (use WebSearch + --competitors-plan) instead of
BRAVE_API_KEY. Footer BRAVE/SERPER nudge suppressed when --plan or
--competitors-plan present (hosting model already has WebSearch).
Added:
- --competitors-plan JSON flag: per-entity {x_handle, x_related, subreddits,
github_user, github_repos, context}. Accepts inline JSON or file path.
subrun_kwargs_for helper is the single source of truth for per-entity
kwargs — no closure-default fallthrough from main scope.
- Per-entity save files: each entity's sub-run produces its own
{slug}-raw.md with a single-row Resolved Entities block.
- --polymarket-keywords filter for ambiguous single-token topics.
Fixed:
- test_competitor_subrun_isolation regression suite locks in 3.0.12's
no-leak invariant (main flags do not inherit into peer sub-runs).
- Updates test_regression.py for the new comparison-mode payload shape.
Bumps plugin.json to 3.0.13. 1,219 tests passing.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
* fix: comparison title attribution — (Last 30 Days) → (/Last30Days)
User feedback on 3.0.13 dogfood runs (Kanye vs Drake, Mercer Island,
Figma): the comparison-mode synthesis title should attribute to the
slash command rather than restate the date range.
Three SKILL.md occurrences updated. Pure documentation change. Bumps to
3.0.14.
Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
---------
Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>
Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
108 lines
4.3 KiB
Python
108 lines
4.3 KiB
Python
import json
|
|
import subprocess
|
|
import sys
|
|
import unittest
|
|
from pathlib import Path
|
|
|
|
|
|
REPO_ROOT = Path(__file__).resolve().parents[1]
|
|
|
|
|
|
def run_mock_json(topic: str) -> dict:
|
|
result = subprocess.run(
|
|
[sys.executable, "scripts/last30days.py", topic, "--mock", "--emit=json"],
|
|
cwd=REPO_ROOT,
|
|
capture_output=True,
|
|
text=True,
|
|
check=False,
|
|
)
|
|
if result.returncode != 0:
|
|
raise AssertionError(f"mock CLI failed for {topic!r}: {result.stderr}")
|
|
return json.loads(result.stdout)
|
|
|
|
|
|
class RegressionTests(unittest.TestCase):
|
|
def assert_common_shape(self, payload: dict) -> None:
|
|
self.assertIn("topic", payload)
|
|
self.assertIn("query_plan", payload)
|
|
self.assertIn("ranked_candidates", payload)
|
|
self.assertIn("clusters", payload)
|
|
self.assertIn("items_by_source", payload)
|
|
|
|
def assert_comparison_shape(self, payload: dict) -> None:
|
|
"""Post-3.0.13: vs-topics produce N full passes, merged output has
|
|
comparison=True + entities list + per-entity report wrapper."""
|
|
self.assertTrue(payload.get("comparison"))
|
|
self.assertIn("entities", payload)
|
|
self.assertIn("reports", payload)
|
|
self.assertEqual(len(payload["entities"]), len(payload["reports"]))
|
|
# Each report entry wraps a single-topic report
|
|
for entry in payload["reports"]:
|
|
self.assertIn("entity", entry)
|
|
self.assertIn("report", entry)
|
|
# Inner report still has the single-topic shape
|
|
inner = entry["report"]
|
|
self.assertIn("topic", inner)
|
|
self.assertIn("query_plan", inner)
|
|
self.assertIn("clusters", inner)
|
|
|
|
def test_openclaw_three_way_comparison_preserves_entities(self):
|
|
payload = run_mock_json("openclaw vs. nanoclaw vs. ironclaw")
|
|
self.assert_comparison_shape(payload)
|
|
entities = [e.lower() for e in payload["entities"]]
|
|
self.assertIn("openclaw", entities)
|
|
self.assertIn("nanoclaw", entities)
|
|
self.assertIn("ironclaw", entities)
|
|
# No cross-entity keyword pollution in any per-entity report's plan
|
|
for entry in payload["reports"]:
|
|
plan = entry["report"]["query_plan"]
|
|
joined = "\n".join(
|
|
sq["search_query"] for sq in plan["subqueries"]
|
|
).lower()
|
|
self.assertNotIn("corsair", joined)
|
|
self.assertNotIn("mouse", joined)
|
|
|
|
def test_how_to_keeps_web_video_and_discussion_sources(self):
|
|
payload = run_mock_json("how to deploy on Fly.io")
|
|
self.assert_common_shape(payload)
|
|
plan = payload["query_plan"]
|
|
self.assertEqual("how_to", plan["intent"])
|
|
sources = set(plan["subqueries"][0]["sources"])
|
|
self.assertIn("youtube", sources)
|
|
self.assertIn("reddit", sources)
|
|
self.assertGreaterEqual(len(sources), 2)
|
|
|
|
def test_breaking_news_query_keeps_expected_shape(self):
|
|
payload = run_mock_json("latest news about React 20")
|
|
self.assert_common_shape(payload)
|
|
plan = payload["query_plan"]
|
|
self.assertEqual("breaking_news", plan["intent"])
|
|
joined_queries = "\n".join(subquery["search_query"] for subquery in plan["subqueries"]).lower()
|
|
self.assertIn("react 20", joined_queries)
|
|
self.assertGreaterEqual(len(plan["subqueries"][0]["sources"]), 2)
|
|
|
|
def test_two_way_comparison_preserves_exact_strings(self):
|
|
payload = run_mock_json("DeepSeek R1 vs GPT-5")
|
|
self.assert_comparison_shape(payload)
|
|
entities_lower = [e.lower() for e in payload["entities"]]
|
|
self.assertIn("deepseek r1", entities_lower)
|
|
self.assertIn("gpt-5", entities_lower)
|
|
# Each per-entity pass has its own entity in its plan
|
|
topics_by_entity = {
|
|
entry["entity"].lower(): entry["report"]["topic"].lower()
|
|
for entry in payload["reports"]
|
|
}
|
|
self.assertEqual(topics_by_entity["deepseek r1"], "deepseek r1")
|
|
self.assertEqual(topics_by_entity["gpt-5"], "gpt-5")
|
|
# No cross-entity pollution
|
|
for entry in payload["reports"]:
|
|
plan = entry["report"]["query_plan"]
|
|
joined = "\n".join(
|
|
sq["search_query"] for sq in plan["subqueries"]
|
|
).lower()
|
|
self.assertNotIn("corsair", joined)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
unittest.main()
|