feat: v3.0.0 - intelligent search, GitHub person/project mode, ELI5, 13+ sources

v3 rewrites the search engine from the ground up:

- Intelligent pre-research: resolves X handles, GitHub repos, subreddits,
  TikTok hashtags, and YouTube channels before searching
- GitHub person-mode: PR velocity, top repos by stars, release notes
- GitHub project-mode: live star counts, README, releases, top issues
- ELI5 mode: plain language synthesis, no jargon
- 13+ sources: Reddit, X, YouTube, TikTok, Instagram, HN, Polymarket,
  GitHub, Threads, Pinterest, Perplexity, Bluesky, Web
- Free Reddit comments via public JSON (no API key needed)
- Fun judge v2: humor scoring baked into narrative
- Cookie consent before browser scanning
- 10,000 free ScrapeCreators calls
- 1,012 tests

Thank you to the community contributors whose issues and PRs shaped v3:
@uppinote20 (#143), @zerone0x (#134, #136), @thinkun (#116),
@thomasmktong (#124), @fanispoulinakisai-boop (#100), @pejmanjohn (#78),
@zl190 (#115), @hnshah (#84, #85, #86)

Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>
This commit is contained in:
Matt Van Horn
2026-04-08 10:52:23 -07:00
parent 61904b31e3
commit 0a9ff16dfc
397 changed files with 21427 additions and 53106 deletions
+221 -413
View File
@@ -1,39 +1,16 @@
"""Environment and API key management for last30days skill."""
from __future__ import annotations
import base64
import binascii
import json
import logging
import os
import sys
import time
from dataclasses import dataclass
from pathlib import Path
from typing import Optional, Dict, Any, List, Literal
logger = logging.getLogger(__name__)
# ---------------------------------------------------------------------------
# Cookie domain registry: maps source names to browser cookie extraction params.
# Each entry: (domain, cookie_names, config_key_mapping)
# config_key_mapping: {cookie_name: config_key} so we know which config key
# each extracted cookie should populate.
# ---------------------------------------------------------------------------
COOKIE_DOMAINS: Dict[str, Dict[str, Any]] = {
"x": {
"domain": ".x.com",
"cookies": ["auth_token", "ct0"],
"mapping": {
"auth_token": "AUTH_TOKEN",
"ct0": "CT0",
},
},
"truthsocial": {
"domain": ".truthsocial.com",
"cookies": ["_session_id"],
"mapping": {
"_session_id": "TRUTHSOCIAL_TOKEN",
},
},
}
from typing import Any, Literal
# Allow override via environment variable for testing
# Set LAST30DAYS_CONFIG_DIR="" for clean/no-config mode
@@ -67,10 +44,10 @@ AUTH_STATUS_MISSING_ACCOUNT_ID: AuthStatus = "missing_account_id"
@dataclass(frozen=True)
class OpenAIAuth:
token: Optional[str]
token: str | None
source: AuthSource
status: AuthStatus
account_id: Optional[str]
account_id: str | None
codex_auth_file: str
@@ -80,17 +57,17 @@ def _check_file_permissions(path: Path) -> None:
mode = path.stat().st_mode
# Check if group or other can read (bits 0o044)
if mode & 0o044:
import sys
sys.stderr.write(
f"[last30days] WARNING: {path} is readable by other users. "
f"Run: chmod 600 {path}\n"
)
sys.stderr.flush()
except OSError:
pass
except OSError as exc:
sys.stderr.write(f"[last30days] WARNING: could not stat {path}: {exc}\n")
sys.stderr.flush()
def load_env_file(path: Path) -> Dict[str, str]:
def load_env_file(path: Path) -> dict[str, str]:
"""Load environment variables from a file."""
env = {}
if not path or not path.exists():
@@ -114,7 +91,7 @@ def load_env_file(path: Path) -> Dict[str, str]:
return env
def _decode_jwt_payload(token: str) -> Optional[Dict[str, Any]]:
def _decode_jwt_payload(token: str) -> dict[str, Any] | None:
"""Decode JWT payload without verification."""
try:
parts = token.split(".")
@@ -124,7 +101,9 @@ def _decode_jwt_payload(token: str) -> Optional[Dict[str, Any]]:
pad = "=" * (-len(payload_b64) % 4)
decoded = base64.urlsafe_b64decode(payload_b64 + pad)
return json.loads(decoded.decode("utf-8"))
except Exception:
except (json.JSONDecodeError, UnicodeDecodeError, binascii.Error, IndexError) as exc:
sys.stderr.write(f"[last30days] WARNING: malformed JWT token: {exc}\n")
sys.stderr.flush()
return None
@@ -139,7 +118,7 @@ def _token_expired(token: str, leeway_seconds: int = 60) -> bool:
return exp <= (time.time() + leeway_seconds)
def extract_chatgpt_account_id(access_token: str) -> Optional[str]:
def extract_chatgpt_account_id(access_token: str) -> str | None:
"""Extract chatgpt_account_id from JWT token."""
payload = _decode_jwt_payload(access_token)
if not payload:
@@ -150,18 +129,22 @@ def extract_chatgpt_account_id(access_token: str) -> Optional[str]:
return None
def load_codex_auth(path: Path = CODEX_AUTH_FILE) -> Dict[str, Any]:
def load_codex_auth(path: Path = CODEX_AUTH_FILE) -> dict[str, Any]:
"""Load Codex auth JSON."""
if not path.exists():
return {}
try:
with open(path, "r") as f:
return json.load(f)
except Exception:
except json.JSONDecodeError:
sys.stderr.write(
f"[last30days] WARNING: {path} exists but contains invalid JSON -- ignoring\n"
)
sys.stderr.flush()
return {}
def get_codex_access_token() -> tuple[Optional[str], str]:
def get_codex_access_token() -> tuple[str | None, str]:
"""Get Codex access token from auth.json.
Returns:
@@ -182,7 +165,7 @@ def get_codex_access_token() -> tuple[Optional[str], str]:
return token, AUTH_STATUS_OK
def get_openai_auth(file_env: Dict[str, str]) -> OpenAIAuth:
def get_openai_auth(file_env: dict[str, str]) -> OpenAIAuth:
"""Resolve OpenAI auth from API key or Codex login."""
api_key = os.environ.get('OPENAI_API_KEY') or file_env.get('OPENAI_API_KEY')
if api_key:
@@ -194,35 +177,20 @@ def get_openai_auth(file_env: Dict[str, str]) -> OpenAIAuth:
codex_auth_file=str(CODEX_AUTH_FILE),
)
codex_token, codex_status = get_codex_access_token()
if codex_token:
account_id = extract_chatgpt_account_id(codex_token)
if account_id:
return OpenAIAuth(
token=codex_token,
source=AUTH_SOURCE_CODEX,
status=AUTH_STATUS_OK,
account_id=account_id,
codex_auth_file=str(CODEX_AUTH_FILE),
)
return OpenAIAuth(
token=None,
source=AUTH_SOURCE_CODEX,
status=AUTH_STATUS_MISSING_ACCOUNT_ID,
account_id=None,
codex_auth_file=str(CODEX_AUTH_FILE),
)
# Codex auth (chatgpt.com backend) intentionally skipped.
# The endpoint is unstable and causes crashes when the token expires.
# Users who want OpenAI should set OPENAI_API_KEY explicitly.
return OpenAIAuth(
token=None,
source=AUTH_SOURCE_NONE,
status=codex_status,
status=AUTH_STATUS_MISSING,
account_id=None,
codex_auth_file=str(CODEX_AUTH_FILE),
)
def _find_project_env() -> Optional[Path]:
def _find_project_env() -> Path | None:
"""Find per-project .env by walking up from cwd.
Searches for .claude/last30days.env in each parent directory,
@@ -239,95 +207,13 @@ def _find_project_env() -> Optional[Path]:
return None
def extract_browser_credentials(config: Dict[str, Any]) -> Dict[str, str]:
"""Extract credentials from browser cookies for sources that need them.
Checks the FROM_BROWSER config key to decide whether/how to extract:
- 'auto': try browsers in platform order
- 'firefox', 'chrome', 'safari': try only that browser
- 'off': skip extraction entirely
If SETUP_COMPLETE is not set AND FROM_BROWSER is not explicitly set,
defaults to 'off' (wizard hasn't run yet — no extraction without consent).
If SETUP_COMPLETE is set and FROM_BROWSER is not set, defaults to 'auto'.
Explicit env var/config values always take priority over extracted cookies.
Returns:
Dict of {config_key: value} for credentials discovered from cookies.
"""
setup_complete = config.get("SETUP_COMPLETE")
from_browser = config.get("FROM_BROWSER")
# Determine effective browser setting
if from_browser is None:
if setup_complete:
from_browser = "auto"
else:
from_browser = "off"
from_browser = from_browser.lower().strip() if isinstance(from_browser, str) else "off"
if from_browser == "off":
return {}
# Lazy import to avoid loading cookie_extract at module level
try:
from . import cookie_extract
except Exception:
logger.debug("cookie_extract module not available")
return {}
credentials: Dict[str, str] = {}
for source_name, spec in COOKIE_DOMAINS.items():
domain = spec["domain"]
cookie_names: List[str] = spec["cookies"]
mapping: Dict[str, str] = spec["mapping"]
# Skip if ALL mapped config keys already have values
all_present = all(config.get(config_key) for config_key in mapping.values())
if all_present:
logger.debug(
"Skipping cookie extraction for %s: credentials already set",
source_name,
)
continue
try:
result = cookie_extract.extract_cookies_with_source(from_browser, domain, cookie_names)
except Exception as exc:
logger.debug(
"Cookie extraction failed for %s: %s", source_name, exc
)
continue
if result is None:
continue
cookies, browser_name = result
filled_any = False
for cookie_name, config_key in mapping.items():
# Only fill in keys not already present
if not config.get(config_key) and cookie_name in cookies:
credentials[config_key] = cookies[cookie_name]
filled_any = True
# Track which browser provided the credentials for this source
if filled_any:
credentials[f"__{source_name.upper()}_BROWSER"] = browser_name
return credentials
def get_config() -> Dict[str, Any]:
def get_config() -> dict[str, Any]:
"""Load configuration from multiple sources.
Priority (highest wins):
1. Environment variables (os.environ)
2. .claude/last30days.env (per-project config)
3. ~/.config/last30days/.env (global config)
4. Browser cookies (only fills in missing keys)
"""
# Load from global config file
file_env = load_env_file(CONFIG_FILE) if CONFIG_FILE else {}
@@ -355,15 +241,13 @@ def get_config() -> Dict[str, Any]:
('GOOGLE_API_KEY', None),
('GEMINI_API_KEY', None),
('GOOGLE_GENAI_API_KEY', None),
('OPENROUTER_API_KEY', None),
('PARALLEL_API_KEY', None),
('BRAVE_API_KEY', None),
('EXA_API_KEY', None),
('XIAOHONGSHU_API_BASE', None),
('GEMINI_MODEL', None),
('OPENAI_MODEL_POLICY', 'auto'),
('LAST30DAYS_REASONING_PROVIDER', 'auto'),
('LAST30DAYS_PLANNER_MODEL', None),
('LAST30DAYS_RERANK_MODEL', None),
('LAST30DAYS_X_MODEL', None),
('LAST30DAYS_X_BACKEND', None),
('OPENAI_MODEL_PIN', None),
('XAI_MODEL_POLICY', 'latest'),
('XAI_MODEL_PIN', None),
('SCRAPECREATORS_API_KEY', None),
('APIFY_API_TOKEN', None),
@@ -372,6 +256,11 @@ def get_config() -> Dict[str, Any]:
('BSKY_HANDLE', None),
('BSKY_APP_PASSWORD', None),
('TRUTHSOCIAL_TOKEN', None),
('BRAVE_API_KEY', None),
('EXA_API_KEY', None),
('SERPER_API_KEY', None),
('OPENROUTER_API_KEY', None),
('PARALLEL_API_KEY', None),
('FROM_BROWSER', None),
('SETUP_COMPLETE', None),
('INCLUDE_SOURCES', None),
@@ -380,24 +269,6 @@ def get_config() -> Dict[str, Any]:
for key, default in keys:
config[key] = os.environ.get(key) or merged_env.get(key, default)
# Inject browser cookies for any credentials not already set
browser_creds = extract_browser_credentials(config)
for key, value in browser_creds.items():
if not config.get(key):
config[key] = value
# Track AUTH_TOKEN source for status reporting
if config.get('AUTH_TOKEN'):
if os.environ.get('AUTH_TOKEN') or merged_env.get('AUTH_TOKEN'):
config['_AUTH_TOKEN_SOURCE'] = 'env'
elif browser_creds.get('AUTH_TOKEN'):
browser_name = browser_creds.get('__X_BROWSER', 'unknown')
config['_AUTH_TOKEN_SOURCE'] = f'browser-{browser_name}'
else:
config['_AUTH_TOKEN_SOURCE'] = 'env' # fallback
else:
config['_AUTH_TOKEN_SOURCE'] = None
# Track which config source was used
if project_env_path:
config['_CONFIG_SOURCE'] = f'project:{project_env_path}'
@@ -406,9 +277,87 @@ def get_config() -> Dict[str, Any]:
else:
config['_CONFIG_SOURCE'] = 'env_only'
# Extract browser credentials if configured
browser_creds = extract_browser_credentials(config)
for key, value in browser_creds.items():
if not config.get(key):
config[key] = value
config[f"_{key}_SOURCE"] = "browser"
return config
# ---------------------------------------------------------------------------
# Browser cookie extraction
# ---------------------------------------------------------------------------
COOKIE_DOMAINS: dict[str, dict[str, Any]] = {
"x": {
"domain": ".x.com",
"cookies": ["auth_token", "ct0"],
"mapping": {"auth_token": "AUTH_TOKEN", "ct0": "CT0"},
},
"truthsocial": {
"domain": ".truthsocial.com",
"cookies": ["_session_id"],
"mapping": {"_session_id": "TRUTHSOCIAL_TOKEN"},
},
}
def extract_browser_credentials(config: dict[str, Any]) -> dict[str, str]:
"""Extract auth cookies from local browsers.
Default behavior (FROM_BROWSER unset): tries Firefox and Safari only.
These read local files silently with no system dialogs. Chrome is
skipped because ``security find-generic-password`` triggers a macOS
Keychain prompt that cannot be reliably suppressed.
Set ``FROM_BROWSER=auto`` to also try Chrome (accepts the dialog),
or ``FROM_BROWSER=off`` to disable extraction entirely.
"""
from_browser = (config.get("FROM_BROWSER") or "").strip().lower()
if from_browser == "off":
return {}
try:
from . import cookie_extract
except ImportError:
return {}
# Determine which browsers to try
if from_browser in ("firefox", "chrome", "safari"):
browsers = [from_browser]
elif from_browser == "auto":
browsers = ["firefox", "safari", "chrome"]
else:
# Default: silent browsers only (no Keychain dialog)
browsers = ["firefox", "safari"]
extracted: dict[str, str] = {}
for _service, spec in COOKIE_DOMAINS.items():
if all(config.get(env_key) for env_key in spec["mapping"].values()):
continue
for browser in browsers:
try:
cookies = cookie_extract.extract_cookies(browser, spec["domain"], spec["cookies"])
except Exception:
continue
if cookies:
for cookie_name, env_key in spec["mapping"].items():
if cookie_name in cookies and not config.get(env_key):
extracted[env_key] = cookies[cookie_name]
break # Found cookies for this service, stop trying browsers
return extracted
def get_x_source_with_method(config: dict[str, Any]) -> tuple[str | None, str]:
"""Return (source, method) for X search, where method describes the auth origin."""
if config.get("XAI_API_KEY"):
return "xai", "xai"
if config.get("AUTH_TOKEN") and config.get("CT0"):
method = config.get("_AUTH_TOKEN_SOURCE", "env")
return "bird", method
return None, "none"
def config_exists() -> bool:
"""Check if any configuration source exists."""
if _find_project_env():
@@ -418,235 +367,60 @@ def config_exists() -> bool:
return False
def is_reddit_available(config: Dict[str, Any]) -> bool:
def is_reddit_available(config: dict[str, Any]) -> bool:
"""Check if Reddit search is available.
Reddit can use either ScrapeCreators (preferred) or OpenAI.
v3 uses ScrapeCreators only.
"""
has_sc = bool(config.get('SCRAPECREATORS_API_KEY'))
has_openai = bool(config.get('OPENAI_API_KEY')) and config.get('OPENAI_AUTH_STATUS') == AUTH_STATUS_OK
return has_sc or has_openai
return bool(config.get('SCRAPECREATORS_API_KEY'))
def get_reddit_source(config: Dict[str, Any]) -> Optional[str]:
def get_reddit_source(config: dict[str, Any]) -> str | None:
"""Determine which Reddit backend to use.
Priority: ScrapeCreators (cheaper, faster) > OpenAI (legacy)
Returns: 'scrapecreators', 'openai', or None
Returns: 'scrapecreators' or None
"""
if config.get('SCRAPECREATORS_API_KEY'):
return 'scrapecreators'
if config.get('OPENAI_API_KEY') and config.get('OPENAI_AUTH_STATUS') == AUTH_STATUS_OK:
return 'openai'
return None
def get_available_sources(config: Dict[str, Any]) -> str:
"""Determine which sources are available.
def get_x_source(config: dict[str, Any]) -> str | None:
"""Determine the best available explicit X/Twitter source.
X is available if ANY auth method works: AUTH_TOKEN/CT0 (env or cookies),
XAI_API_KEY, or Bird installed+authenticated.
Reddit is always available (public JSON fallback).
HN and Polymarket are always available.
YouTube available if yt-dlp installed.
Priority: explicit backend pin, then xAI, then Bird with explicit cookies.
Returns: 'all', 'both', 'reddit', 'reddit-web', 'x', 'x-web', 'web', or 'none'
"""
has_reddit = True
has_x = get_x_source(config) is not None
has_web = has_web_search_keys(config)
if has_reddit and has_x:
return 'all' if has_web else 'both'
elif has_reddit:
return 'reddit-web' if has_web else 'reddit'
return 'web' if has_web else 'none'
def has_web_search_keys(config: Dict[str, Any]) -> bool:
"""Check if any web search API keys are configured."""
return bool(config.get('EXA_API_KEY') or config.get('OPENROUTER_API_KEY') or config.get('PARALLEL_API_KEY') or config.get('BRAVE_API_KEY'))
def get_web_search_source(config: Dict[str, Any]) -> Optional[str]:
"""Determine the best available web search backend.
Priority: Exa (free) > Parallel AI > Brave > OpenRouter/Sonar Pro
Returns: 'exa', 'parallel', 'brave', 'openrouter', or None
"""
if config.get('EXA_API_KEY'):
return 'exa'
if config.get('PARALLEL_API_KEY'):
return 'parallel'
if config.get('BRAVE_API_KEY'):
return 'brave'
if config.get('OPENROUTER_API_KEY'):
return 'openrouter'
return None
def get_missing_keys(config: Dict[str, Any]) -> str:
"""Determine which sources are missing (accounting for Bird).
Returns: 'all', 'both', 'reddit', 'x', 'web', or 'none'
"""
has_reddit = True
has_xai = bool(config.get('XAI_API_KEY'))
has_web = has_web_search_keys(config)
# Check if Bird provides X access (import here to avoid circular dependency)
from . import bird_x
has_bird = bird_x.is_bird_installed() and bird_x.is_bird_authenticated()
has_x = has_xai or has_bird
if has_reddit and has_x and has_web:
return 'none'
elif has_reddit and has_x:
return 'web' # Missing web search keys
elif has_reddit and has_web:
return 'x' # Missing X source
elif has_reddit:
return 'x' # Missing X source (and possibly web)
return 'all'
def validate_sources(requested: str, available: str, include_web: bool = False) -> tuple[str, Optional[str]]:
"""Validate requested sources against available keys.
Args:
requested: 'auto', 'reddit', 'x', 'both', or 'web'
available: Result from get_available_sources()
include_web: If True, add WebSearch to available sources
Returns:
Tuple of (effective_sources, error_message)
"""
has_reddit = available in ('reddit', 'both', 'reddit-web', 'all')
has_x = available in ('x', 'both', 'x-web', 'all')
has_web = available in ('web', 'reddit-web', 'x-web', 'all')
if requested == 'auto':
if has_reddit and has_x:
base = 'both'
elif has_reddit:
base = 'reddit'
elif has_x:
base = 'x'
elif has_web:
base = 'web'
else:
return 'none', "No sources are available."
if include_web:
if base == 'both':
return 'all', None
if base == 'reddit':
return 'reddit-web', None
if base == 'x':
return 'x-web', None
return base, None
if requested == 'web':
return 'web', None
if requested == 'both':
if not (has_reddit and has_x):
return 'none', "Requested both sources but X source is missing."
if include_web:
return 'all', None
return 'both', None
if requested == 'reddit':
if not has_reddit:
return 'none', "Requested Reddit but only xAI key is available."
if include_web:
return 'reddit-web', None
return 'reddit', None
if requested == 'x':
if not has_x:
return 'none', "Requested X but no X source is available (need Bird auth or XAI_API_KEY)."
if include_web:
return 'x-web', None
return 'x', None
return requested, None
def get_x_source(config: Dict[str, Any]) -> Optional[str]:
"""Determine the best available X/Twitter source.
Priority chain:
1. AUTH_TOKEN/CT0 from env var or .env file → Bird with method "env"
2. AUTH_TOKEN/CT0 from browser cookie extraction → Bird with method "browser-{browser}"
3. XAI_API_KEY → xAI with method "api"
4. None
Use get_x_source_with_method() to also get the method string.
Browser-cookie probing is intentionally not used here. Automatic Keychain
access causes popups during normal pipeline runs. Bird is only considered
available when AUTH_TOKEN and CT0 are present explicitly.
Args:
config: Configuration dict from get_config()
Returns:
'bird' if Bird is installed and authenticated,
'bird' if Bird is installed and explicit cookies are configured,
'xai' if XAI_API_KEY is configured,
None if no X source available.
"""
source, _method = get_x_source_with_method(config)
return source
def get_x_source_with_method(config: Dict[str, Any]) -> tuple[Optional[str], Optional[str]]:
"""Determine the best available X/Twitter source and auth method.
Priority chain:
1. AUTH_TOKEN/CT0 (env var or .env) → Bird with method "env"
2. AUTH_TOKEN/CT0 (browser cookies) → Bird with method "browser-{browser}"
3. XAI_API_KEY → xAI with method "api"
4. None
Args:
config: Configuration dict from get_config()
Returns:
Tuple of (source, method) where source is 'bird', 'xai', or None
and method is 'env', 'browser-chrome', 'browser-firefox', 'browser-safari', 'api', or None.
"""
# Import here to avoid circular dependency
from . import bird_x
setup_complete = config.get('SETUP_COMPLETE')
preferred = (config.get('LAST30DAYS_X_BACKEND') or '').lower()
has_bird_creds = bool(config.get('AUTH_TOKEN') and config.get('CT0'))
if has_bird_creds:
bird_x.set_credentials(config.get('AUTH_TOKEN'), config.get('CT0'))
# Check Bird first (free option — uses AUTH_TOKEN/CT0 from any source)
if bird_x.is_bird_installed():
auth_source = config.get('_AUTH_TOKEN_SOURCE')
if preferred == 'xai':
return 'xai' if config.get('XAI_API_KEY') else None
if preferred == 'bird':
return 'bird' if has_bird_creds and bird_x.is_bird_installed() else None
# If SETUP_COMPLETE is not set, only allow explicit env var credentials.
# Do NOT call is_bird_authenticated() for browser-cookie probing —
# that requires user consent via the setup wizard.
if not setup_complete:
# Explicit AUTH_TOKEN from env var / .env file is always allowed
if auth_source == 'env' and config.get('AUTH_TOKEN'):
username = bird_x.is_bird_authenticated()
if username:
return 'bird', 'env'
else:
# SETUP_COMPLETE is set — normal flow, probe cookies if needed
username = bird_x.is_bird_authenticated()
if username:
if auth_source and auth_source.startswith('browser-'):
method = auth_source # e.g. "browser-firefox"
else:
method = 'env'
return 'bird', method
# Fall back to xAI if key exists
if config.get('XAI_API_KEY'):
return 'xai', 'api'
return 'xai'
if has_bird_creds and bird_x.is_bird_installed():
return 'bird'
return None, None
return None
def is_ytdlp_available() -> bool:
@@ -655,6 +429,25 @@ def is_ytdlp_available() -> bool:
return youtube_yt.is_ytdlp_installed()
def is_youtube_comments_available(config: dict[str, Any]) -> bool:
"""Check if YouTube comment enrichment is available.
Requires SCRAPECREATORS_API_KEY AND youtube_comments in INCLUDE_SOURCES.
"""
if not config.get('SCRAPECREATORS_API_KEY'):
return False
include = _parse_include_sources(config)
return 'youtube_comments' in include
def is_youtube_sc_available(config: dict[str, Any]) -> bool:
"""Check if ScrapeCreators YouTube search fallback is available.
Used when yt-dlp is not installed or fails.
"""
return bool(config.get('SCRAPECREATORS_API_KEY'))
def is_hackernews_available() -> bool:
"""Check if Hacker News source is available.
@@ -663,7 +456,7 @@ def is_hackernews_available() -> bool:
return True
def is_bluesky_available(config: Dict[str, Any]) -> bool:
def is_bluesky_available(config: dict[str, Any]) -> bool:
"""Check if Bluesky source is available.
Requires BSKY_HANDLE and BSKY_APP_PASSWORD (app password from bsky.app/settings).
@@ -671,7 +464,7 @@ def is_bluesky_available(config: Dict[str, Any]) -> bool:
return bool(config.get('BSKY_HANDLE') and config.get('BSKY_APP_PASSWORD'))
def is_truthsocial_available(config: Dict[str, Any]) -> bool:
def is_truthsocial_available(config: dict[str, Any]) -> bool:
"""Check if Truth Social source is available.
Requires TRUTHSOCIAL_TOKEN (bearer token from browser dev tools).
@@ -687,7 +480,7 @@ def is_polymarket_available() -> bool:
return True
def is_tiktok_available(config: Dict[str, Any]) -> bool:
def is_tiktok_available(config: dict[str, Any]) -> bool:
"""Check if TikTok source is available (ScrapeCreators or legacy Apify).
Returns True if SCRAPECREATORS_API_KEY or APIFY_API_TOKEN is set.
@@ -695,12 +488,29 @@ def is_tiktok_available(config: Dict[str, Any]) -> bool:
return bool(config.get('SCRAPECREATORS_API_KEY') or config.get('APIFY_API_TOKEN'))
def get_tiktok_token(config: Dict[str, Any]) -> str:
def get_tiktok_token(config: dict[str, Any]) -> str:
"""Get TikTok API token, preferring ScrapeCreators over legacy Apify."""
return config.get('SCRAPECREATORS_API_KEY') or config.get('APIFY_API_TOKEN') or ''
def is_instagram_available(config: Dict[str, Any]) -> bool:
def _parse_include_sources(config: dict[str, Any]) -> set[str]:
"""Parse INCLUDE_SOURCES config value into a set of lowercase source names."""
raw = config.get('INCLUDE_SOURCES') or ''
return {s.strip().lower() for s in raw.split(',') if s.strip()}
def is_threads_available(config: dict[str, Any]) -> bool:
"""Check if Threads source is available.
Requires SCRAPECREATORS_API_KEY AND 'threads' in INCLUDE_SOURCES.
Threads is an opt-in source - it is not activated by default.
"""
if not config.get('SCRAPECREATORS_API_KEY'):
return False
return 'threads' in _parse_include_sources(config)
def is_instagram_available(config: dict[str, Any]) -> bool:
"""Check if Instagram source is available (ScrapeCreators).
Returns True if SCRAPECREATORS_API_KEY is set.
@@ -709,12 +519,12 @@ def is_instagram_available(config: Dict[str, Any]) -> bool:
return bool(config.get('SCRAPECREATORS_API_KEY'))
def get_instagram_token(config: Dict[str, Any]) -> str:
def get_instagram_token(config: dict[str, Any]) -> str:
"""Get Instagram API token (same ScrapeCreators key as TikTok)."""
return config.get('SCRAPECREATORS_API_KEY') or ''
def get_xiaohongshu_api_base(config: Dict[str, Any]) -> str:
def get_xiaohongshu_api_base(config: dict[str, Any]) -> str:
"""Get Xiaohongshu HTTP API base URL.
Defaults to host.docker.internal so OpenClaw Docker can reach host service.
@@ -722,7 +532,7 @@ def get_xiaohongshu_api_base(config: Dict[str, Any]) -> str:
return (config.get('XIAOHONGSHU_API_BASE') or "http://host.docker.internal:18060").rstrip("/")
def is_xiaohongshu_available(config: Dict[str, Any]) -> bool:
def is_xiaohongshu_available(config: dict[str, Any]) -> bool:
"""Check whether Xiaohongshu HTTP API is reachable and logged in."""
# Import here to avoid heavy imports at module load.
from . import http
@@ -744,7 +554,14 @@ def is_xiaohongshu_available(config: Dict[str, Any]) -> bool:
if isinstance(login, dict) else False
)
return bool(is_logged_in)
except Exception:
except (OSError, http.HTTPError):
return False
except Exception as exc:
sys.stderr.write(
f"[last30days] WARNING: unexpected error checking Xiaohongshu: "
f"{type(exc).__name__}: {exc}\n"
)
sys.stderr.flush()
return False
@@ -752,56 +569,47 @@ def is_xiaohongshu_available(config: Dict[str, Any]) -> bool:
is_apify_available = is_tiktok_available
def get_x_source_status(config: Dict[str, Any]) -> Dict[str, Any]:
def get_x_source_status(config: dict[str, Any]) -> dict[str, Any]:
"""Get detailed X source status for UI decisions.
Returns:
Dict with keys: source, method, bird_installed, bird_authenticated,
Dict with keys: source, bird_installed, bird_authenticated,
bird_username, xai_available, can_install_bird
The ``method`` field indicates HOW the active source is authenticated:
- "env" — AUTH_TOKEN came from an env var or .env file
- "browser-chrome", "browser-firefox", "browser-safari" — from cookie extraction
- "api" — using xAI API key
- None — no X source available
"""
from . import bird_x
setup_complete = config.get('SETUP_COMPLETE')
bird_status = bird_x.get_bird_status()
xai_available = bool(config.get('XAI_API_KEY'))
if not setup_complete:
# Before consent: do NOT call get_bird_status() which probes cookies.
# Only check if Bird is installed (no cookie probing) and use the
# gated get_x_source_with_method() which blocks cookie detection.
bird_installed = bird_x.is_bird_installed()
source, method = get_x_source_with_method(config)
# Bird "authenticated" only if get_x_source_with_method found explicit creds
bird_authenticated = (source == 'bird')
return {
"source": source,
"method": method,
"bird_installed": bird_installed,
"bird_authenticated": bird_authenticated,
"bird_username": None if not bird_authenticated else "env AUTH_TOKEN",
"xai_available": xai_available,
"can_install_bird": True,
}
# SETUP_COMPLETE is set — normal flow
bird_status = bird_x.get_bird_status()
# Use the unified resolution function for source + method
source, method = get_x_source_with_method(config)
# Determine active source
if bird_status["authenticated"]:
source = 'bird'
elif xai_available:
source = 'xai'
else:
source = None
return {
"source": source,
"method": method,
"bird_installed": bird_status["installed"],
"bird_authenticated": bird_status["authenticated"],
"bird_username": bird_status["username"],
"xai_available": xai_available,
"can_install_bird": bird_status["can_install"],
}
# Pinterest
def is_pinterest_available(config: dict[str, Any]) -> bool:
"""Check if Pinterest source is available.
Returns True when SCRAPECREATORS_API_KEY is set AND 'pinterest' is in
INCLUDE_SOURCES (or requested_sources at the pipeline level). Pinterest
is opt-in because not every topic benefits from visual pin results.
"""
return bool(config.get('SCRAPECREATORS_API_KEY'))
def get_pinterest_token(config: dict[str, Any]) -> str:
"""Get Pinterest API token (same ScrapeCreators key as TikTok/Instagram)."""
return config.get('SCRAPECREATORS_API_KEY') or ''