Files
last30days-skill/tests/test_regression.py
Matt Van Horn 949bcf8942 feat: vs mode N full passes + --competitors auto-discovery + (/Last30Days) title (#312)
* feat: vs mode runs N full passes; --competitors wraps vs with auto-discovery

Unifies vs-mode and --competitors onto one fanout architecture. A topic
containing "vs" / "versus" now runs N full pipeline.run() calls in parallel
(reverting the one-pass latency optimization that removed per-entity
depth); --competitors becomes a SKILL.md-level shortcut where the hosting
reasoning model (Claude Code, Codex, Hermes, Gemini) discovers N peers via
its own WebSearch, runs Step 0.55 per entity, and invokes the engine with
a vs-topic + --competitors-plan JSON.

Changed:
- vs-mode: N full passes in parallel via fanout (was 1 merged pass).
- --competitors: SKILL.md shortcut for vs-mode-with-discovery. Engine flag
  kept for headless/cron use. LAW 7-style stderr reframed to lead with the
  hosting-model path (use WebSearch + --competitors-plan) instead of
  BRAVE_API_KEY. Footer BRAVE/SERPER nudge suppressed when --plan or
  --competitors-plan present (hosting model already has WebSearch).

Added:
- --competitors-plan JSON flag: per-entity {x_handle, x_related, subreddits,
  github_user, github_repos, context}. Accepts inline JSON or file path.
  subrun_kwargs_for helper is the single source of truth for per-entity
  kwargs — no closure-default fallthrough from main scope.
- Per-entity save files: each entity's sub-run produces its own
  {slug}-raw.md with a single-row Resolved Entities block.
- --polymarket-keywords filter for ambiguous single-token topics.

Fixed:
- test_competitor_subrun_isolation regression suite locks in 3.0.12's
  no-leak invariant (main flags do not inherit into peer sub-runs).
- Updates test_regression.py for the new comparison-mode payload shape.

Bumps plugin.json to 3.0.13. 1,219 tests passing.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>

* fix: comparison title attribution — (Last 30 Days) → (/Last30Days)

User feedback on 3.0.13 dogfood runs (Kanye vs Drake, Mercer Island,
Figma): the comparison-mode synthesis title should attribute to the
slash command rather than restate the date range.

Three SKILL.md occurrences updated. Pure documentation change. Bumps to
3.0.14.

Co-Authored-By: Claude Opus 4.7 (1M context) <noreply@anthropic.com>

---------

Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com>
Co-authored-by: Claude Opus 4.7 (1M context) <noreply@anthropic.com>
2026-04-22 21:31:00 -07:00

108 lines
4.3 KiB
Python

import json
import subprocess
import sys
import unittest
from pathlib import Path
REPO_ROOT = Path(__file__).resolve().parents[1]
def run_mock_json(topic: str) -> dict:
result = subprocess.run(
[sys.executable, "scripts/last30days.py", topic, "--mock", "--emit=json"],
cwd=REPO_ROOT,
capture_output=True,
text=True,
check=False,
)
if result.returncode != 0:
raise AssertionError(f"mock CLI failed for {topic!r}: {result.stderr}")
return json.loads(result.stdout)
class RegressionTests(unittest.TestCase):
def assert_common_shape(self, payload: dict) -> None:
self.assertIn("topic", payload)
self.assertIn("query_plan", payload)
self.assertIn("ranked_candidates", payload)
self.assertIn("clusters", payload)
self.assertIn("items_by_source", payload)
def assert_comparison_shape(self, payload: dict) -> None:
"""Post-3.0.13: vs-topics produce N full passes, merged output has
comparison=True + entities list + per-entity report wrapper."""
self.assertTrue(payload.get("comparison"))
self.assertIn("entities", payload)
self.assertIn("reports", payload)
self.assertEqual(len(payload["entities"]), len(payload["reports"]))
# Each report entry wraps a single-topic report
for entry in payload["reports"]:
self.assertIn("entity", entry)
self.assertIn("report", entry)
# Inner report still has the single-topic shape
inner = entry["report"]
self.assertIn("topic", inner)
self.assertIn("query_plan", inner)
self.assertIn("clusters", inner)
def test_openclaw_three_way_comparison_preserves_entities(self):
payload = run_mock_json("openclaw vs. nanoclaw vs. ironclaw")
self.assert_comparison_shape(payload)
entities = [e.lower() for e in payload["entities"]]
self.assertIn("openclaw", entities)
self.assertIn("nanoclaw", entities)
self.assertIn("ironclaw", entities)
# No cross-entity keyword pollution in any per-entity report's plan
for entry in payload["reports"]:
plan = entry["report"]["query_plan"]
joined = "\n".join(
sq["search_query"] for sq in plan["subqueries"]
).lower()
self.assertNotIn("corsair", joined)
self.assertNotIn("mouse", joined)
def test_how_to_keeps_web_video_and_discussion_sources(self):
payload = run_mock_json("how to deploy on Fly.io")
self.assert_common_shape(payload)
plan = payload["query_plan"]
self.assertEqual("how_to", plan["intent"])
sources = set(plan["subqueries"][0]["sources"])
self.assertIn("youtube", sources)
self.assertIn("reddit", sources)
self.assertGreaterEqual(len(sources), 2)
def test_breaking_news_query_keeps_expected_shape(self):
payload = run_mock_json("latest news about React 20")
self.assert_common_shape(payload)
plan = payload["query_plan"]
self.assertEqual("breaking_news", plan["intent"])
joined_queries = "\n".join(subquery["search_query"] for subquery in plan["subqueries"]).lower()
self.assertIn("react 20", joined_queries)
self.assertGreaterEqual(len(plan["subqueries"][0]["sources"]), 2)
def test_two_way_comparison_preserves_exact_strings(self):
payload = run_mock_json("DeepSeek R1 vs GPT-5")
self.assert_comparison_shape(payload)
entities_lower = [e.lower() for e in payload["entities"]]
self.assertIn("deepseek r1", entities_lower)
self.assertIn("gpt-5", entities_lower)
# Each per-entity pass has its own entity in its plan
topics_by_entity = {
entry["entity"].lower(): entry["report"]["topic"].lower()
for entry in payload["reports"]
}
self.assertEqual(topics_by_entity["deepseek r1"], "deepseek r1")
self.assertEqual(topics_by_entity["gpt-5"], "gpt-5")
# No cross-entity pollution
for entry in payload["reports"]:
plan = entry["report"]["query_plan"]
joined = "\n".join(
sq["search_query"] for sq in plan["subqueries"]
).lower()
self.assertNotIn("corsair", joined)
if __name__ == "__main__":
unittest.main()