feat: replace OpenAI Reddit search with ScrapeCreators API
- New scripts/lib/reddit.py: multi-query expansion, global search, subreddit discovery, targeted subreddit search, comment enrichment - 68 results in 17s vs ~15 results in 60-90s (OpenAI) - Cost: ~$0.02/search vs $0.03-0.10 (15-50x cheaper) - Real engagement data (score, comments, dates) from API - No more 429 rate limits on comment enrichment - Falls back to OpenAI if SCRAPECREATORS_API_KEY missing - Registered as last30daysbeta for parallel local testing Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
This commit is contained in:
@@ -1,10 +1,10 @@
|
|||||||
---
|
---
|
||||||
name: last30days
|
name: last30daysbeta
|
||||||
version: "2.8"
|
version: "2.9-beta"
|
||||||
description: "Research a topic from the last 30 days. Also triggered by 'last30'. Sources: Reddit, X, YouTube, TikTok, Instagram, Hacker News, Polymarket, web. Become an expert and write copy-paste-ready prompts."
|
description: "BETA: Research a topic from the last 30 days with ScrapeCreators Reddit. Sources: Reddit, X, YouTube, TikTok, Instagram, Hacker News, Polymarket, web. Become an expert and write copy-paste-ready prompts."
|
||||||
argument-hint: 'last30 AI video tools, last30 best project management tools'
|
argument-hint: 'last30daysbeta AI video tools, last30daysbeta best project management tools'
|
||||||
allowed-tools: Bash, Read, Write, AskUserQuestion, WebSearch
|
allowed-tools: Bash, Read, Write, AskUserQuestion, WebSearch
|
||||||
homepage: https://github.com/mvanhorn/last30days-skill
|
homepage: https://github.com/mvanhorn/last30days-skill-private
|
||||||
user-invocable: true
|
user-invocable: true
|
||||||
metadata:
|
metadata:
|
||||||
clawdbot:
|
clawdbot:
|
||||||
|
|||||||
+55
-13
@@ -140,6 +140,7 @@ from lib import (
|
|||||||
models,
|
models,
|
||||||
normalize,
|
normalize,
|
||||||
openai_reddit,
|
openai_reddit,
|
||||||
|
reddit,
|
||||||
reddit_enrich,
|
reddit_enrich,
|
||||||
render,
|
render,
|
||||||
schema,
|
schema,
|
||||||
@@ -171,19 +172,51 @@ def _search_reddit(
|
|||||||
depth: str,
|
depth: str,
|
||||||
mock: bool,
|
mock: bool,
|
||||||
) -> tuple:
|
) -> tuple:
|
||||||
"""Search Reddit via OpenAI (runs in thread).
|
"""Search Reddit (runs in thread).
|
||||||
|
|
||||||
|
Uses ScrapeCreators when SCRAPECREATORS_API_KEY is available (preferred).
|
||||||
|
Falls back to OpenAI Responses API otherwise.
|
||||||
|
|
||||||
Returns:
|
Returns:
|
||||||
Tuple of (reddit_items, raw_openai, error)
|
Tuple of (reddit_items, raw_response, error, used_scrapecreators)
|
||||||
"""
|
"""
|
||||||
raw_openai = None
|
raw_response = None
|
||||||
reddit_error = None
|
reddit_error = None
|
||||||
|
used_scrapecreators = False
|
||||||
|
|
||||||
|
sc_token = config.get("SCRAPECREATORS_API_KEY")
|
||||||
|
|
||||||
if mock:
|
if mock:
|
||||||
raw_openai = load_fixture("openai_sample.json")
|
raw_response = load_fixture("openai_sample.json")
|
||||||
else:
|
elif sc_token:
|
||||||
|
# === ScrapeCreators path (preferred) ===
|
||||||
|
used_scrapecreators = True
|
||||||
try:
|
try:
|
||||||
raw_openai = openai_reddit.search_reddit(
|
sys.stderr.write("[Reddit] Using ScrapeCreators API\n")
|
||||||
|
sys.stderr.flush()
|
||||||
|
result = reddit.search_and_enrich(
|
||||||
|
topic, from_date, to_date,
|
||||||
|
depth=depth, token=sc_token,
|
||||||
|
)
|
||||||
|
reddit_items = result.get("items", [])
|
||||||
|
if result.get("error"):
|
||||||
|
reddit_error = result["error"]
|
||||||
|
return reddit_items, result, reddit_error, used_scrapecreators
|
||||||
|
except Exception as e:
|
||||||
|
reddit_error = f"ScrapeCreators: {type(e).__name__}: {e}"
|
||||||
|
sys.stderr.write(f"[Reddit] ScrapeCreators failed: {e}\n")
|
||||||
|
sys.stderr.flush()
|
||||||
|
# Fall through to OpenAI if we have that key
|
||||||
|
if not config.get("OPENAI_API_KEY"):
|
||||||
|
return [], {"error": str(e)}, reddit_error, used_scrapecreators
|
||||||
|
used_scrapecreators = False
|
||||||
|
sys.stderr.write("[Reddit] Falling back to OpenAI\n")
|
||||||
|
sys.stderr.flush()
|
||||||
|
|
||||||
|
# === OpenAI path (fallback) ===
|
||||||
|
if not mock:
|
||||||
|
try:
|
||||||
|
raw_response = openai_reddit.search_reddit(
|
||||||
config["OPENAI_API_KEY"],
|
config["OPENAI_API_KEY"],
|
||||||
selected_models["openai"],
|
selected_models["openai"],
|
||||||
topic,
|
topic,
|
||||||
@@ -194,14 +227,14 @@ def _search_reddit(
|
|||||||
account_id=config.get("OPENAI_CHATGPT_ACCOUNT_ID"),
|
account_id=config.get("OPENAI_CHATGPT_ACCOUNT_ID"),
|
||||||
)
|
)
|
||||||
except http.HTTPError as e:
|
except http.HTTPError as e:
|
||||||
raw_openai = {"error": str(e)}
|
raw_response = {"error": str(e)}
|
||||||
reddit_error = f"API error: {e}"
|
reddit_error = f"API error: {e}"
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
raw_openai = {"error": str(e)}
|
raw_response = {"error": str(e)}
|
||||||
reddit_error = f"{type(e).__name__}: {e}"
|
reddit_error = f"{type(e).__name__}: {e}"
|
||||||
|
|
||||||
# Parse response
|
# Parse response
|
||||||
reddit_items = openai_reddit.parse_reddit_response(raw_openai or {})
|
reddit_items = openai_reddit.parse_reddit_response(raw_response or {})
|
||||||
|
|
||||||
# Quick retry with simpler query if few results
|
# Quick retry with simpler query if few results
|
||||||
if len(reddit_items) < 5 and not mock and not reddit_error:
|
if len(reddit_items) < 5 and not mock and not reddit_error:
|
||||||
@@ -218,7 +251,6 @@ def _search_reddit(
|
|||||||
account_id=config.get("OPENAI_CHATGPT_ACCOUNT_ID"),
|
account_id=config.get("OPENAI_CHATGPT_ACCOUNT_ID"),
|
||||||
)
|
)
|
||||||
retry_items = openai_reddit.parse_reddit_response(retry_raw)
|
retry_items = openai_reddit.parse_reddit_response(retry_raw)
|
||||||
# Add items not already found (by URL)
|
|
||||||
existing_urls = {item.get("url") for item in reddit_items}
|
existing_urls = {item.get("url") for item in reddit_items}
|
||||||
for item in retry_items:
|
for item in retry_items:
|
||||||
if item.get("url") not in existing_urls:
|
if item.get("url") not in existing_urls:
|
||||||
@@ -245,7 +277,7 @@ def _search_reddit(
|
|||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
return reddit_items, raw_openai, reddit_error
|
return reddit_items, raw_response, reddit_error, used_scrapecreators
|
||||||
|
|
||||||
|
|
||||||
def _search_x(
|
def _search_x(
|
||||||
@@ -887,10 +919,11 @@ def run_research(
|
|||||||
)
|
)
|
||||||
|
|
||||||
# Collect results (with timeouts to prevent indefinite blocking)
|
# Collect results (with timeouts to prevent indefinite blocking)
|
||||||
|
reddit_used_sc = False # Track if ScrapeCreators was used for Reddit
|
||||||
if reddit_future:
|
if reddit_future:
|
||||||
reddit_timeout = timeouts.get("reddit_future", future_timeout)
|
reddit_timeout = timeouts.get("reddit_future", future_timeout)
|
||||||
try:
|
try:
|
||||||
reddit_items, raw_openai, reddit_error = reddit_future.result(timeout=reddit_timeout)
|
reddit_items, raw_openai, reddit_error, reddit_used_sc = reddit_future.result(timeout=reddit_timeout)
|
||||||
if reddit_error and progress:
|
if reddit_error and progress:
|
||||||
progress.show_error(f"Reddit error: {reddit_error}")
|
progress.show_error(f"Reddit error: {reddit_error}")
|
||||||
except TimeoutError:
|
except TimeoutError:
|
||||||
@@ -1022,11 +1055,19 @@ def run_research(
|
|||||||
sys.stderr.flush()
|
sys.stderr.flush()
|
||||||
|
|
||||||
# Enrich Reddit items with real data (parallel, capped)
|
# Enrich Reddit items with real data (parallel, capped)
|
||||||
|
# Skip enrichment if ScrapeCreators already provided comments + engagement
|
||||||
enrich_max = timeouts["enrich_max_items"]
|
enrich_max = timeouts["enrich_max_items"]
|
||||||
enrich_total_timeout = timeouts["enrich_total"]
|
enrich_total_timeout = timeouts["enrich_total"]
|
||||||
items_to_enrich = reddit_items[:enrich_max]
|
items_to_enrich = reddit_items[:enrich_max]
|
||||||
rate_limited = False # Set True if Reddit returns 429 during enrichment
|
rate_limited = False # Set True if Reddit returns 429 during enrichment
|
||||||
|
|
||||||
|
if reddit_used_sc and items_to_enrich:
|
||||||
|
# ScrapeCreators already enriched items with comments — just copy to raw list
|
||||||
|
sys.stderr.write(f"[Reddit] Skipping old enrichment — ScrapeCreators already provided comments\n")
|
||||||
|
sys.stderr.flush()
|
||||||
|
raw_reddit_enriched = list(reddit_items[:enrich_max])
|
||||||
|
items_to_enrich = [] # Skip the enrichment block below
|
||||||
|
|
||||||
if items_to_enrich:
|
if items_to_enrich:
|
||||||
if progress:
|
if progress:
|
||||||
progress.start_reddit_enrich(1, len(items_to_enrich))
|
progress.start_reddit_enrich(1, len(items_to_enrich))
|
||||||
@@ -1101,11 +1142,12 @@ def run_research(
|
|||||||
|
|
||||||
# Phase 2: Supplemental search based on entities from Phase 1
|
# Phase 2: Supplemental search based on entities from Phase 1
|
||||||
# Skip on --quick (speed matters), mock mode, or if Reddit is rate-limiting
|
# Skip on --quick (speed matters), mock mode, or if Reddit is rate-limiting
|
||||||
|
# Also skip Reddit supplemental when ScrapeCreators was used (subreddit drilling already done)
|
||||||
if depth != "quick" and not mock and (reddit_items or x_items):
|
if depth != "quick" and not mock and (reddit_items or x_items):
|
||||||
sup_reddit, sup_x = _run_supplemental(
|
sup_reddit, sup_x = _run_supplemental(
|
||||||
topic, reddit_items, x_items,
|
topic, reddit_items, x_items,
|
||||||
from_date, to_date, depth, x_source, progress,
|
from_date, to_date, depth, x_source, progress,
|
||||||
skip_reddit=rate_limited,
|
skip_reddit=(rate_limited or reddit_used_sc),
|
||||||
resolved_handle=resolved_handle,
|
resolved_handle=resolved_handle,
|
||||||
)
|
)
|
||||||
if sup_reddit:
|
if sup_reddit:
|
||||||
|
|||||||
+33
-9
@@ -220,18 +220,42 @@ def config_exists() -> bool:
|
|||||||
return CONFIG_FILE.exists()
|
return CONFIG_FILE.exists()
|
||||||
|
|
||||||
|
|
||||||
|
def is_reddit_available(config: Dict[str, Any]) -> bool:
|
||||||
|
"""Check if Reddit search is available.
|
||||||
|
|
||||||
|
Reddit can use either ScrapeCreators (preferred) or OpenAI.
|
||||||
|
"""
|
||||||
|
has_sc = bool(config.get('SCRAPECREATORS_API_KEY'))
|
||||||
|
has_openai = bool(config.get('OPENAI_API_KEY')) and config.get('OPENAI_AUTH_STATUS') == AUTH_STATUS_OK
|
||||||
|
return has_sc or has_openai
|
||||||
|
|
||||||
|
|
||||||
|
def get_reddit_source(config: Dict[str, Any]) -> Optional[str]:
|
||||||
|
"""Determine which Reddit backend to use.
|
||||||
|
|
||||||
|
Priority: ScrapeCreators (cheaper, faster) > OpenAI (legacy)
|
||||||
|
|
||||||
|
Returns: 'scrapecreators', 'openai', or None
|
||||||
|
"""
|
||||||
|
if config.get('SCRAPECREATORS_API_KEY'):
|
||||||
|
return 'scrapecreators'
|
||||||
|
if config.get('OPENAI_API_KEY') and config.get('OPENAI_AUTH_STATUS') == AUTH_STATUS_OK:
|
||||||
|
return 'openai'
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
def get_available_sources(config: Dict[str, Any]) -> str:
|
def get_available_sources(config: Dict[str, Any]) -> str:
|
||||||
"""Determine which sources are available based on API keys.
|
"""Determine which sources are available based on API keys.
|
||||||
|
|
||||||
Returns: 'all', 'both', 'reddit', 'reddit-web', 'x', 'x-web', 'web', or 'none'
|
Returns: 'all', 'both', 'reddit', 'reddit-web', 'x', 'x-web', 'web', or 'none'
|
||||||
"""
|
"""
|
||||||
has_openai = bool(config.get('OPENAI_API_KEY')) and config.get('OPENAI_AUTH_STATUS') == AUTH_STATUS_OK
|
has_reddit = is_reddit_available(config)
|
||||||
has_xai = bool(config.get('XAI_API_KEY'))
|
has_xai = bool(config.get('XAI_API_KEY'))
|
||||||
has_web = has_web_search_keys(config)
|
has_web = has_web_search_keys(config)
|
||||||
|
|
||||||
if has_openai and has_xai:
|
if has_reddit and has_xai:
|
||||||
return 'all' if has_web else 'both'
|
return 'all' if has_web else 'both'
|
||||||
elif has_openai:
|
elif has_reddit:
|
||||||
return 'reddit-web' if has_web else 'reddit'
|
return 'reddit-web' if has_web else 'reddit'
|
||||||
elif has_xai:
|
elif has_xai:
|
||||||
return 'x-web' if has_web else 'x'
|
return 'x-web' if has_web else 'x'
|
||||||
@@ -263,11 +287,11 @@ def get_web_search_source(config: Dict[str, Any]) -> Optional[str]:
|
|||||||
|
|
||||||
|
|
||||||
def get_missing_keys(config: Dict[str, Any]) -> str:
|
def get_missing_keys(config: Dict[str, Any]) -> str:
|
||||||
"""Determine which sources are missing (accounting for Bird).
|
"""Determine which sources are missing (accounting for Bird and ScrapeCreators).
|
||||||
|
|
||||||
Returns: 'all', 'both', 'reddit', 'x', 'web', or 'none'
|
Returns: 'all', 'both', 'reddit', 'x', 'web', or 'none'
|
||||||
"""
|
"""
|
||||||
has_openai = bool(config.get('OPENAI_API_KEY')) and config.get('OPENAI_AUTH_STATUS') == AUTH_STATUS_OK
|
has_reddit = is_reddit_available(config)
|
||||||
has_xai = bool(config.get('XAI_API_KEY'))
|
has_xai = bool(config.get('XAI_API_KEY'))
|
||||||
has_web = has_web_search_keys(config)
|
has_web = has_web_search_keys(config)
|
||||||
|
|
||||||
@@ -277,14 +301,14 @@ def get_missing_keys(config: Dict[str, Any]) -> str:
|
|||||||
|
|
||||||
has_x = has_xai or has_bird
|
has_x = has_xai or has_bird
|
||||||
|
|
||||||
if has_openai and has_x and has_web:
|
if has_reddit and has_x and has_web:
|
||||||
return 'none'
|
return 'none'
|
||||||
elif has_openai and has_x:
|
elif has_reddit and has_x:
|
||||||
return 'web' # Missing web search keys
|
return 'web' # Missing web search keys
|
||||||
elif has_openai:
|
elif has_reddit:
|
||||||
return 'x' # Missing X source (and possibly web)
|
return 'x' # Missing X source (and possibly web)
|
||||||
elif has_x:
|
elif has_x:
|
||||||
return 'reddit' # Missing OpenAI key (and possibly web)
|
return 'reddit' # Missing Reddit source (and possibly web)
|
||||||
else:
|
else:
|
||||||
return 'all' # Missing everything
|
return 'all' # Missing everything
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,562 @@
|
|||||||
|
"""Reddit search via ScrapeCreators API for /last30days.
|
||||||
|
|
||||||
|
Uses ScrapeCreators REST API to search Reddit globally, discover relevant
|
||||||
|
subreddits, run targeted subreddit searches, and fetch comment trees.
|
||||||
|
|
||||||
|
Replaces openai_reddit.py as the primary Reddit search backend.
|
||||||
|
Falls back to openai_reddit.py if SCRAPECREATORS_API_KEY is missing but
|
||||||
|
OPENAI_API_KEY is present.
|
||||||
|
|
||||||
|
Requires SCRAPECREATORS_API_KEY in config (same key as TikTok + Instagram).
|
||||||
|
API docs: https://scrapecreators.com/docs
|
||||||
|
"""
|
||||||
|
|
||||||
|
import re
|
||||||
|
import sys
|
||||||
|
from collections import Counter
|
||||||
|
from datetime import datetime, timezone
|
||||||
|
from typing import Any, Dict, List, Optional, Set
|
||||||
|
|
||||||
|
try:
|
||||||
|
import requests as _requests
|
||||||
|
except ImportError:
|
||||||
|
_requests = None
|
||||||
|
|
||||||
|
from . import http
|
||||||
|
|
||||||
|
SCRAPECREATORS_BASE = "https://api.scrapecreators.com/v1/reddit"
|
||||||
|
|
||||||
|
# Depth configurations: how many API calls per phase
|
||||||
|
DEPTH_CONFIG = {
|
||||||
|
"quick": {
|
||||||
|
"global_searches": 1,
|
||||||
|
"subreddit_searches": 2,
|
||||||
|
"comment_enrichments": 3,
|
||||||
|
"timeframe": "week",
|
||||||
|
},
|
||||||
|
"default": {
|
||||||
|
"global_searches": 2,
|
||||||
|
"subreddit_searches": 3,
|
||||||
|
"comment_enrichments": 5,
|
||||||
|
"timeframe": "month",
|
||||||
|
},
|
||||||
|
"deep": {
|
||||||
|
"global_searches": 3,
|
||||||
|
"subreddit_searches": 5,
|
||||||
|
"comment_enrichments": 8,
|
||||||
|
"timeframe": "month",
|
||||||
|
},
|
||||||
|
}
|
||||||
|
|
||||||
|
# Stopwords for query extraction
|
||||||
|
NOISE_WORDS = frozenset({
|
||||||
|
'best', 'top', 'good', 'great', 'awesome', 'killer',
|
||||||
|
'latest', 'new', 'news', 'update', 'updates',
|
||||||
|
'trending', 'hottest', 'popular',
|
||||||
|
'practices', 'features', 'tips',
|
||||||
|
'recommendations', 'advice',
|
||||||
|
'prompt', 'prompts', 'prompting',
|
||||||
|
'methods', 'strategies', 'approaches',
|
||||||
|
'how', 'to', 'the', 'a', 'an', 'for', 'with',
|
||||||
|
'of', 'in', 'on', 'is', 'are', 'what', 'which',
|
||||||
|
'guide', 'tutorial', 'using',
|
||||||
|
})
|
||||||
|
|
||||||
|
|
||||||
|
def _log(msg: str):
|
||||||
|
"""Log to stderr."""
|
||||||
|
sys.stderr.write(f"[Reddit/SC] {msg}\n")
|
||||||
|
sys.stderr.flush()
|
||||||
|
|
||||||
|
|
||||||
|
def _sc_headers(token: str) -> Dict[str, str]:
|
||||||
|
"""Build ScrapeCreators request headers."""
|
||||||
|
return {
|
||||||
|
"x-api-key": token,
|
||||||
|
"Content-Type": "application/json",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _extract_core_subject(topic: str) -> str:
|
||||||
|
"""Extract core subject from verbose query.
|
||||||
|
|
||||||
|
Strips meta/research words to keep only the core product/concept name.
|
||||||
|
"""
|
||||||
|
text = topic.lower().strip()
|
||||||
|
|
||||||
|
# Strip multi-word prefixes
|
||||||
|
prefixes = [
|
||||||
|
'what are the best', 'what is the best', 'what are the latest',
|
||||||
|
'what are people saying about', 'what do people think about',
|
||||||
|
'how do i use', 'how to use', 'how to',
|
||||||
|
'what are', 'what is', 'tips for', 'best practices for',
|
||||||
|
]
|
||||||
|
for p in prefixes:
|
||||||
|
if text.startswith(p + ' '):
|
||||||
|
text = text[len(p):].strip()
|
||||||
|
|
||||||
|
words = text.split()
|
||||||
|
filtered = [w for w in words if w not in NOISE_WORDS]
|
||||||
|
|
||||||
|
result = ' '.join(filtered) if filtered else text
|
||||||
|
return result.rstrip('?!.')
|
||||||
|
|
||||||
|
|
||||||
|
def expand_reddit_queries(topic: str, depth: str) -> List[str]:
|
||||||
|
"""Generate multiple Reddit search queries from a topic.
|
||||||
|
|
||||||
|
Uses local logic (no LLM call needed):
|
||||||
|
1. Extract core subject (strip noise words)
|
||||||
|
2. Include original topic if different from core
|
||||||
|
3. For default/deep: add casual/review variant
|
||||||
|
4. For deep: add problem/issues variant
|
||||||
|
|
||||||
|
Returns 1-4 query strings depending on depth.
|
||||||
|
"""
|
||||||
|
core = _extract_core_subject(topic)
|
||||||
|
queries = [core]
|
||||||
|
|
||||||
|
# Broader variant: include more context from original topic
|
||||||
|
original_clean = topic.strip().rstrip('?!.')
|
||||||
|
if core.lower() != original_clean.lower() and len(original_clean.split()) <= 8:
|
||||||
|
queries.append(original_clean)
|
||||||
|
|
||||||
|
if depth in ("default", "deep"):
|
||||||
|
queries.append(f"{core} worth it OR thoughts OR review")
|
||||||
|
|
||||||
|
if depth == "deep":
|
||||||
|
queries.append(f"{core} issues OR problems OR bug OR broken")
|
||||||
|
|
||||||
|
return queries
|
||||||
|
|
||||||
|
|
||||||
|
def discover_subreddits(results: List[Dict[str, Any]], max_subs: int = 5) -> List[str]:
|
||||||
|
"""Extract top subreddits from global search results by frequency.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
results: List of post dicts from global search
|
||||||
|
max_subs: Maximum subreddits to return
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Top subreddit names sorted by post count
|
||||||
|
"""
|
||||||
|
counts = Counter()
|
||||||
|
for post in results:
|
||||||
|
sub = post.get("subreddit", "")
|
||||||
|
if sub:
|
||||||
|
counts[sub] += 1
|
||||||
|
|
||||||
|
return [sub for sub, _ in counts.most_common(max_subs)]
|
||||||
|
|
||||||
|
|
||||||
|
def _parse_date(created_utc) -> Optional[str]:
|
||||||
|
"""Convert Unix timestamp to YYYY-MM-DD."""
|
||||||
|
if not created_utc:
|
||||||
|
return None
|
||||||
|
try:
|
||||||
|
dt = datetime.fromtimestamp(float(created_utc), tz=timezone.utc)
|
||||||
|
return dt.strftime("%Y-%m-%d")
|
||||||
|
except (ValueError, TypeError, OSError):
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
def _normalize_post(post: Dict[str, Any], idx: int, source_label: str = "global") -> Dict[str, Any]:
|
||||||
|
"""Normalize a ScrapeCreators Reddit post to our internal format."""
|
||||||
|
permalink = post.get("permalink", "")
|
||||||
|
url = f"https://www.reddit.com{permalink}" if permalink else post.get("url", "")
|
||||||
|
|
||||||
|
# Ensure URL looks like a Reddit thread
|
||||||
|
if url and "reddit.com" not in url:
|
||||||
|
url = ""
|
||||||
|
|
||||||
|
return {
|
||||||
|
"id": f"R{idx}",
|
||||||
|
"reddit_id": post.get("id", ""),
|
||||||
|
"title": str(post.get("title", "")).strip(),
|
||||||
|
"url": url,
|
||||||
|
"subreddit": str(post.get("subreddit", "")).strip(),
|
||||||
|
"date": _parse_date(post.get("created_utc")),
|
||||||
|
"engagement": {
|
||||||
|
"score": post.get("ups") or post.get("score", 0),
|
||||||
|
"num_comments": post.get("num_comments", 0),
|
||||||
|
"upvote_ratio": post.get("upvote_ratio"),
|
||||||
|
},
|
||||||
|
"relevance": 0.7,
|
||||||
|
"why_relevant": f"Reddit {source_label} search",
|
||||||
|
"selftext": str(post.get("selftext", ""))[:500],
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
def _global_search(
|
||||||
|
query: str,
|
||||||
|
token: str,
|
||||||
|
sort: str = "relevance",
|
||||||
|
timeframe: str = "month",
|
||||||
|
) -> List[Dict[str, Any]]:
|
||||||
|
"""Search across all of Reddit via ScrapeCreators global search.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
query: Search query
|
||||||
|
token: ScrapeCreators API key
|
||||||
|
sort: Sort order (relevance, hot, top, new)
|
||||||
|
timeframe: Time filter (hour, day, week, month, year, all)
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of post dicts
|
||||||
|
"""
|
||||||
|
if not _requests:
|
||||||
|
_log("requests library not installed, falling back to urllib")
|
||||||
|
# Use stdlib http module as fallback
|
||||||
|
try:
|
||||||
|
from urllib.parse import urlencode
|
||||||
|
params = urlencode({"query": query, "sort": sort, "timeframe": timeframe})
|
||||||
|
url = f"{SCRAPECREATORS_BASE}/search?{params}"
|
||||||
|
headers = _sc_headers(token)
|
||||||
|
headers["User-Agent"] = http.USER_AGENT
|
||||||
|
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||||
|
return data.get("posts", data.get("data", []))
|
||||||
|
except Exception as e:
|
||||||
|
_log(f"Global search error (urllib): {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
try:
|
||||||
|
resp = _requests.get(
|
||||||
|
f"{SCRAPECREATORS_BASE}/search",
|
||||||
|
params={"query": query, "sort": sort, "timeframe": timeframe},
|
||||||
|
headers=_sc_headers(token),
|
||||||
|
timeout=30,
|
||||||
|
)
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
return data.get("posts", data.get("data", []))
|
||||||
|
except Exception as e:
|
||||||
|
_log(f"Global search error: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def _subreddit_search(
|
||||||
|
subreddit: str,
|
||||||
|
query: str,
|
||||||
|
token: str,
|
||||||
|
sort: str = "relevance",
|
||||||
|
timeframe: str = "month",
|
||||||
|
) -> List[Dict[str, Any]]:
|
||||||
|
"""Search within a specific subreddit via ScrapeCreators.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
subreddit: Subreddit name (without r/)
|
||||||
|
query: Search query
|
||||||
|
token: ScrapeCreators API key
|
||||||
|
sort: Sort order
|
||||||
|
timeframe: Time filter
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of post dicts
|
||||||
|
"""
|
||||||
|
if not _requests:
|
||||||
|
try:
|
||||||
|
from urllib.parse import urlencode
|
||||||
|
params = urlencode({
|
||||||
|
"subreddit": subreddit, "query": query,
|
||||||
|
"sort": sort, "timeframe": timeframe,
|
||||||
|
})
|
||||||
|
url = f"{SCRAPECREATORS_BASE}/subreddit/search?{params}"
|
||||||
|
headers = _sc_headers(token)
|
||||||
|
headers["User-Agent"] = http.USER_AGENT
|
||||||
|
data = http.get(url, headers=headers, timeout=30, retries=2)
|
||||||
|
return data.get("posts", data.get("data", []))
|
||||||
|
except Exception as e:
|
||||||
|
_log(f"Subreddit search error (urllib) for r/{subreddit}: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
try:
|
||||||
|
resp = _requests.get(
|
||||||
|
f"{SCRAPECREATORS_BASE}/subreddit/search",
|
||||||
|
params={
|
||||||
|
"subreddit": subreddit,
|
||||||
|
"query": query,
|
||||||
|
"sort": sort,
|
||||||
|
"timeframe": timeframe,
|
||||||
|
},
|
||||||
|
headers=_sc_headers(token),
|
||||||
|
timeout=30,
|
||||||
|
)
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
return data.get("posts", data.get("data", []))
|
||||||
|
except Exception as e:
|
||||||
|
_log(f"Subreddit search error for r/{subreddit}: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def fetch_post_comments(
|
||||||
|
url: str,
|
||||||
|
token: str,
|
||||||
|
) -> List[Dict[str, Any]]:
|
||||||
|
"""Fetch comments for a Reddit post via ScrapeCreators.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
url: Reddit post URL or permalink
|
||||||
|
token: ScrapeCreators API key
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
List of comment dicts with score, author, body, etc.
|
||||||
|
"""
|
||||||
|
if not _requests:
|
||||||
|
try:
|
||||||
|
from urllib.parse import urlencode
|
||||||
|
params = urlencode({"url": url})
|
||||||
|
api_url = f"{SCRAPECREATORS_BASE}/post/comments?{params}"
|
||||||
|
headers = _sc_headers(token)
|
||||||
|
headers["User-Agent"] = http.USER_AGENT
|
||||||
|
data = http.get(api_url, headers=headers, timeout=30, retries=2)
|
||||||
|
return data.get("comments", data.get("data", []))
|
||||||
|
except Exception as e:
|
||||||
|
_log(f"Comment fetch error (urllib): {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
try:
|
||||||
|
resp = _requests.get(
|
||||||
|
f"{SCRAPECREATORS_BASE}/post/comments",
|
||||||
|
params={"url": url},
|
||||||
|
headers=_sc_headers(token),
|
||||||
|
timeout=30,
|
||||||
|
)
|
||||||
|
resp.raise_for_status()
|
||||||
|
data = resp.json()
|
||||||
|
return data.get("comments", data.get("data", []))
|
||||||
|
except Exception as e:
|
||||||
|
_log(f"Comment fetch error: {e}")
|
||||||
|
return []
|
||||||
|
|
||||||
|
|
||||||
|
def _dedupe_posts(posts: List[Dict[str, Any]]) -> List[Dict[str, Any]]:
|
||||||
|
"""Deduplicate posts by reddit_id, keeping first occurrence."""
|
||||||
|
seen_ids = set()
|
||||||
|
seen_urls = set()
|
||||||
|
unique = []
|
||||||
|
for post in posts:
|
||||||
|
rid = post.get("reddit_id", "")
|
||||||
|
url = post.get("url", "")
|
||||||
|
if rid and rid in seen_ids:
|
||||||
|
continue
|
||||||
|
if url and url in seen_urls:
|
||||||
|
continue
|
||||||
|
if rid:
|
||||||
|
seen_ids.add(rid)
|
||||||
|
if url:
|
||||||
|
seen_urls.add(url)
|
||||||
|
unique.append(post)
|
||||||
|
return unique
|
||||||
|
|
||||||
|
|
||||||
|
def search_reddit(
|
||||||
|
topic: str,
|
||||||
|
from_date: str,
|
||||||
|
to_date: str,
|
||||||
|
depth: str = "default",
|
||||||
|
token: str = None,
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
"""Full Reddit search: multi-query global discovery + subreddit drill-down.
|
||||||
|
|
||||||
|
This is the main entry point. Replaces openai_reddit.search_reddit().
|
||||||
|
|
||||||
|
Args:
|
||||||
|
topic: Search topic
|
||||||
|
from_date: Start date (YYYY-MM-DD)
|
||||||
|
to_date: End date (YYYY-MM-DD)
|
||||||
|
depth: 'quick', 'default', or 'deep'
|
||||||
|
token: ScrapeCreators API key
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict with 'items' list and optional 'error'.
|
||||||
|
"""
|
||||||
|
if not token:
|
||||||
|
return {"items": [], "error": "No SCRAPECREATORS_API_KEY configured"}
|
||||||
|
|
||||||
|
config = DEPTH_CONFIG.get(depth, DEPTH_CONFIG["default"])
|
||||||
|
timeframe = config["timeframe"]
|
||||||
|
|
||||||
|
# === Phase 1: Query Expansion ===
|
||||||
|
queries = expand_reddit_queries(topic, depth)
|
||||||
|
_log(f"Expanded '{topic}' into {len(queries)} queries: {queries}")
|
||||||
|
|
||||||
|
# === Phase 2: Global Discovery ===
|
||||||
|
all_raw_posts = []
|
||||||
|
max_global = config["global_searches"]
|
||||||
|
|
||||||
|
for i, query in enumerate(queries[:max_global]):
|
||||||
|
sort = "relevance" if i == 0 else "top"
|
||||||
|
_log(f"Global search {i+1}/{max_global}: '{query}' (sort={sort})")
|
||||||
|
posts = _global_search(query, token, sort=sort, timeframe=timeframe)
|
||||||
|
_log(f" -> {len(posts)} results")
|
||||||
|
all_raw_posts.extend(posts)
|
||||||
|
|
||||||
|
# Normalize all posts
|
||||||
|
all_items = []
|
||||||
|
for i, post in enumerate(all_raw_posts):
|
||||||
|
item = _normalize_post(post, i + 1, "global")
|
||||||
|
all_items.append(item)
|
||||||
|
|
||||||
|
# === Phase 3: Subreddit Discovery + Targeted Search ===
|
||||||
|
discovered_subs = discover_subreddits(all_raw_posts, max_subs=config["subreddit_searches"])
|
||||||
|
_log(f"Discovered subreddits: {discovered_subs}")
|
||||||
|
|
||||||
|
core = _extract_core_subject(topic)
|
||||||
|
for sub in discovered_subs[:config["subreddit_searches"]]:
|
||||||
|
_log(f"Subreddit search: r/{sub} for '{core}'")
|
||||||
|
sub_posts = _subreddit_search(sub, core, token, sort="relevance", timeframe=timeframe)
|
||||||
|
_log(f" -> {len(sub_posts)} results from r/{sub}")
|
||||||
|
for j, post in enumerate(sub_posts):
|
||||||
|
item = _normalize_post(post, len(all_items) + j + 1, f"r/{sub}")
|
||||||
|
all_items.append(item)
|
||||||
|
|
||||||
|
# === Phase 4: Deduplicate ===
|
||||||
|
all_items = _dedupe_posts(all_items)
|
||||||
|
_log(f"After dedup: {len(all_items)} unique posts")
|
||||||
|
|
||||||
|
# === Phase 5: Date filter ===
|
||||||
|
in_range = []
|
||||||
|
out_of_range = 0
|
||||||
|
for item in all_items:
|
||||||
|
if item["date"] and from_date <= item["date"] <= to_date:
|
||||||
|
in_range.append(item)
|
||||||
|
elif item["date"] is None:
|
||||||
|
in_range.append(item) # Keep unknown dates
|
||||||
|
else:
|
||||||
|
out_of_range += 1
|
||||||
|
|
||||||
|
if in_range:
|
||||||
|
all_items = in_range
|
||||||
|
if out_of_range:
|
||||||
|
_log(f"Filtered {out_of_range} posts outside date range")
|
||||||
|
else:
|
||||||
|
_log(f"No posts within date range, keeping all {len(all_items)}")
|
||||||
|
|
||||||
|
# === Phase 6: Sort by engagement ===
|
||||||
|
all_items.sort(
|
||||||
|
key=lambda x: (x.get("engagement", {}).get("score", 0) or 0),
|
||||||
|
reverse=True,
|
||||||
|
)
|
||||||
|
|
||||||
|
# Re-index IDs
|
||||||
|
for i, item in enumerate(all_items):
|
||||||
|
item["id"] = f"R{i+1}"
|
||||||
|
|
||||||
|
_log(f"Final: {len(all_items)} Reddit posts")
|
||||||
|
return {"items": all_items}
|
||||||
|
|
||||||
|
|
||||||
|
def enrich_with_comments(
|
||||||
|
items: List[Dict[str, Any]],
|
||||||
|
token: str,
|
||||||
|
depth: str = "default",
|
||||||
|
) -> List[Dict[str, Any]]:
|
||||||
|
"""Enrich top items with comment data from ScrapeCreators.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
items: Reddit items from search_reddit()
|
||||||
|
token: ScrapeCreators API key
|
||||||
|
depth: Depth for comment limit
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Items with top_comments and comment_insights added.
|
||||||
|
"""
|
||||||
|
config = DEPTH_CONFIG.get(depth, DEPTH_CONFIG["default"])
|
||||||
|
max_comments = config["comment_enrichments"]
|
||||||
|
|
||||||
|
if not items or not token:
|
||||||
|
return items
|
||||||
|
|
||||||
|
top_items = items[:max_comments]
|
||||||
|
_log(f"Enriching comments for {len(top_items)} posts")
|
||||||
|
|
||||||
|
for item in top_items:
|
||||||
|
url = item.get("url", "")
|
||||||
|
if not url:
|
||||||
|
continue
|
||||||
|
|
||||||
|
raw_comments = fetch_post_comments(url, token)
|
||||||
|
if not raw_comments:
|
||||||
|
continue
|
||||||
|
|
||||||
|
# Parse comments into our format
|
||||||
|
top_comments = []
|
||||||
|
insights = []
|
||||||
|
|
||||||
|
for c in raw_comments[:10]: # Take top 10 comments
|
||||||
|
body = c.get("body", "")
|
||||||
|
if not body or body in ("[deleted]", "[removed]"):
|
||||||
|
continue
|
||||||
|
|
||||||
|
score = c.get("ups") or c.get("score", 0)
|
||||||
|
author = c.get("author", "[deleted]")
|
||||||
|
permalink = c.get("permalink", "")
|
||||||
|
comment_url = f"https://reddit.com{permalink}" if permalink else ""
|
||||||
|
|
||||||
|
top_comments.append({
|
||||||
|
"score": score,
|
||||||
|
"date": _parse_date(c.get("created_utc")),
|
||||||
|
"author": author,
|
||||||
|
"excerpt": body[:300],
|
||||||
|
"url": comment_url,
|
||||||
|
})
|
||||||
|
|
||||||
|
# Extract insights from substantive comments
|
||||||
|
if len(body) >= 30 and author not in ("[deleted]", "[removed]", "AutoModerator"):
|
||||||
|
insight = body[:150]
|
||||||
|
if len(body) > 150:
|
||||||
|
for i, char in enumerate(insight):
|
||||||
|
if char in '.!?' and i > 50:
|
||||||
|
insight = insight[:i+1]
|
||||||
|
break
|
||||||
|
else:
|
||||||
|
insight = insight.rstrip() + "..."
|
||||||
|
insights.append(insight)
|
||||||
|
|
||||||
|
# Sort comments by score
|
||||||
|
top_comments.sort(key=lambda c: c.get("score", 0), reverse=True)
|
||||||
|
|
||||||
|
item["top_comments"] = top_comments[:10]
|
||||||
|
item["comment_insights"] = insights[:7]
|
||||||
|
|
||||||
|
return items
|
||||||
|
|
||||||
|
|
||||||
|
def search_and_enrich(
|
||||||
|
topic: str,
|
||||||
|
from_date: str,
|
||||||
|
to_date: str,
|
||||||
|
depth: str = "default",
|
||||||
|
token: str = None,
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
"""Full Reddit pipeline: search + comment enrichment.
|
||||||
|
|
||||||
|
This is the convenience function that does everything.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
topic: Search topic
|
||||||
|
from_date: Start date (YYYY-MM-DD)
|
||||||
|
to_date: End date (YYYY-MM-DD)
|
||||||
|
depth: 'quick', 'default', or 'deep'
|
||||||
|
token: ScrapeCreators API key
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict with 'items' list. Items include top_comments and comment_insights.
|
||||||
|
"""
|
||||||
|
result = search_reddit(topic, from_date, to_date, depth, token)
|
||||||
|
items = result.get("items", [])
|
||||||
|
|
||||||
|
if items and token:
|
||||||
|
items = enrich_with_comments(items, token, depth)
|
||||||
|
result["items"] = items
|
||||||
|
|
||||||
|
return result
|
||||||
|
|
||||||
|
|
||||||
|
def parse_reddit_response(response: Dict[str, Any]) -> List[Dict[str, Any]]:
|
||||||
|
"""Parse ScrapeCreators response to item list.
|
||||||
|
|
||||||
|
Compatibility shim matching openai_reddit.parse_reddit_response() signature.
|
||||||
|
"""
|
||||||
|
return response.get("items", [])
|
||||||
@@ -1,4 +1,9 @@
|
|||||||
"""Reddit thread enrichment with real engagement metrics."""
|
"""Reddit thread enrichment with real engagement metrics.
|
||||||
|
|
||||||
|
Supports two backends:
|
||||||
|
1. ScrapeCreators API (preferred) - no rate limits, 1 credit/call
|
||||||
|
2. reddit.com/.json (fallback) - free but 429-prone
|
||||||
|
"""
|
||||||
|
|
||||||
import re
|
import re
|
||||||
from typing import Any, Dict, List, Optional
|
from typing import Any, Dict, List, Optional
|
||||||
@@ -254,3 +259,67 @@ def enrich_reddit_item(
|
|||||||
item["comment_insights"] = extract_comment_insights(top_comments)
|
item["comment_insights"] = extract_comment_insights(top_comments)
|
||||||
|
|
||||||
return item
|
return item
|
||||||
|
|
||||||
|
|
||||||
|
def enrich_reddit_item_sc(
|
||||||
|
item: Dict[str, Any],
|
||||||
|
token: str,
|
||||||
|
timeout: int = 30,
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
"""Enrich a Reddit item using ScrapeCreators comment API.
|
||||||
|
|
||||||
|
No rate limit risk. Uses 1 credit per call.
|
||||||
|
|
||||||
|
Args:
|
||||||
|
item: Reddit item dict (already has engagement from search)
|
||||||
|
token: ScrapeCreators API key
|
||||||
|
timeout: HTTP timeout
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Enriched item with top_comments and comment_insights
|
||||||
|
"""
|
||||||
|
from . import reddit as reddit_mod
|
||||||
|
|
||||||
|
url = item.get("url", "")
|
||||||
|
if not url:
|
||||||
|
return item
|
||||||
|
|
||||||
|
raw_comments = reddit_mod.fetch_post_comments(url, token)
|
||||||
|
if not raw_comments:
|
||||||
|
return item
|
||||||
|
|
||||||
|
top_comments = []
|
||||||
|
for c in raw_comments[:10]:
|
||||||
|
body = c.get("body", "")
|
||||||
|
if not body or body in ("[deleted]", "[removed]"):
|
||||||
|
continue
|
||||||
|
|
||||||
|
score = c.get("ups") or c.get("score", 0)
|
||||||
|
author = c.get("author", "[deleted]")
|
||||||
|
permalink = c.get("permalink", "")
|
||||||
|
comment_url = f"https://reddit.com{permalink}" if permalink else ""
|
||||||
|
|
||||||
|
top_comments.append({
|
||||||
|
"score": score,
|
||||||
|
"date": dates.timestamp_to_date(c.get("created_utc")) if c.get("created_utc") else None,
|
||||||
|
"author": author,
|
||||||
|
"body": body[:300],
|
||||||
|
"excerpt": body[:200],
|
||||||
|
"url": comment_url,
|
||||||
|
})
|
||||||
|
|
||||||
|
top_comments.sort(key=lambda c: c.get("score", 0), reverse=True)
|
||||||
|
|
||||||
|
item["top_comments"] = []
|
||||||
|
for c in top_comments:
|
||||||
|
item["top_comments"].append({
|
||||||
|
"score": c.get("score", 0),
|
||||||
|
"date": c.get("date"),
|
||||||
|
"author": c.get("author", ""),
|
||||||
|
"excerpt": c.get("excerpt", ""),
|
||||||
|
"url": c.get("url", ""),
|
||||||
|
})
|
||||||
|
|
||||||
|
item["comment_insights"] = extract_comment_insights(top_comments)
|
||||||
|
|
||||||
|
return item
|
||||||
|
|||||||
Reference in New Issue
Block a user