diff --git a/fixtures/reddit_listing_cards_sample.html b/fixtures/reddit_listing_cards_sample.html
new file mode 100644
index 0000000..0f176b3
--- /dev/null
+++ b/fixtures/reddit_listing_cards_sample.html
@@ -0,0 +1,8 @@
+
+
+
+
+
+
+
+
diff --git a/fixtures/reddit_search_rss_sample.xml b/fixtures/reddit_search_rss_sample.xml
new file mode 100644
index 0000000..982c9cd
--- /dev/null
+++ b/fixtures/reddit_search_rss_sample.xml
@@ -0,0 +1,7 @@
+
+2026-05-29T14:14:32+00:00https://www.redditstatic.com/icon.png//r/Rakuten/top.rss?t=monthThis is an unofficial subreddit for Rakuten Rewards, the cash back website. We are not affiliated with, endorsed by, or sponsored by Rakuten or any of its subsidiaries.top scoring links : Rakuten/u/InternetUser52https://www.reddit.com/user/InternetUser52<!-- SC_OFF --><div class="md"><p>I'm rich!!</p> </div><!-- SC_ON -->   submitted by   <a href="https://www.reddit.com/user/InternetUser52"> /u/InternetUser52 </a> <br/> <span><a href="https://i.redd.it/q8fgmxs29c2h1.jpeg">[link]</a></span>   <span><a href="https://www.reddit.com/r/Rakuten/comments/1tiv013/lets_goo_002/">[comments]</a></span>t3_1tiv0132026-05-20T18:48:31+00:002026-05-20T18:48:31+00:00LETS GOO! $0.02!!!
+/u/Immediate-Duck-6351https://www.reddit.com/user/Immediate-Duck-6351<!-- SC_OFF --><div class="md"><p>I don’t travel and I’m buying a house in a few weeks so cash back is amazing 🙌 hoping to keep the pace in the next quarter so I can buy new kitchen appliances lol. </p> </div><!-- SC_ON -->   submitted by   <a href="https://www.reddit.com/user/Immediate-Duck-6351"> /u/Immediate-Duck-6351 </a> <br/> <span><a href="https://i.redd.it/d2a4s0ipvb1h1.jpeg">[link]</a></span>   <span><a href="https://www.reddit.com/r/Rakuten/comments/1te1fp8/so_excited/">[comments]</a></span>t3_1te1fp82026-05-15T16:29:28+00:002026-05-15T16:29:28+00:00So excited 🥳
+/u/gnibgnibhttps://www.reddit.com/user/gnibgnib<!-- SC_OFF --><div class="md"><p>128k for the May transfer</p> <p>41k pending for August </p> <p>Got another 9k at Asics not showing but overall pretty happy with Rakuten</p> <p>P2 was able to secure 85k for May transfer</p> </div><!-- SC_ON -->   submitted by   <a href="https://www.reddit.com/user/gnibgnib"> /u/gnibgnib </a> <br/> <span><a href="https://www.reddit.com/gallery/1tb8674">[link]</a></span>   <span><a href="https://www.reddit.com/r/Rakuten/comments/1tb8674/had_a_great_run_so_far_this_year_thanks_to_this/">[comments]</a></span>t3_1tb86742026-05-12T17:17:19+00:002026-05-12T17:17:19+00:00Had a great run so far this year thanks to this sub!
+/u/TravelVet93https://www.reddit.com/user/TravelVet93  submitted by   <a href="https://www.reddit.com/user/TravelVet93"> /u/TravelVet93 </a> <br/> <span><a href="https://i.redd.it/x6b9whvupb1h1.jpeg">[link]</a></span>   <span><a href="https://www.reddit.com/r/Rakuten/comments/1te0hom/my_best_payout_so_far/">[comments]</a></span>t3_1te0hom2026-05-15T15:56:40+00:002026-05-15T15:56:40+00:00My best payout so far
+/u/Beautiful-Piece-4252https://www.reddit.com/user/Beautiful-Piece-4252<!-- SC_OFF --><div class="md"><p>The amount of $$ available in sign up bonuses is amazing. It's kind of a part time job ensuring Rakuten captures everything, but my August and November payout should be sizeable. I'm new to this and it always seemed like a lot of work for little reward. I know it's not sustainable, but wow!</p> </div><!-- SC_ON -->   submitted by   <a href="https://www.reddit.com/user/Beautiful-Piece-4252"> /u/Beautiful-Piece-4252 </a> <br/> <span><a href="https://i.redd.it/1vqvajsci42h1.jpeg">[link]</a></span>   <span><a href="https://www.reddit.com/r/Rakuten/comments/1thsnm1/how_can_this_be_real/">[comments]</a></span>t3_1thsnm12026-05-19T16:46:17+00:002026-05-19T16:46:17+00:00How can this be real?
+
diff --git a/fixtures/reddit_shreddit_comments_sample.html b/fixtures/reddit_shreddit_comments_sample.html
new file mode 100644
index 0000000..97308e6
--- /dev/null
+++ b/fixtures/reddit_shreddit_comments_sample.html
@@ -0,0 +1,29 @@
+
+
+
+
+
Where do you find $750? The highest available package for Total was $284.99 when I did the lifelock promotion. I did get the full 284.99 from Rakuten.
+
+
+
It went to pending ($712.49)
+
+
+
Hey I PM’d. can I get the screenshot ?
+
+
+
Family plan
+
+
+
Price seems to change every time I go to the page but I see only 249.99-369.99 for Total/Advanced. No where near your $750. Just saying the Total plan for 299.99 worked for me and I got 284.99 which is 95%.
+
+
+
I did that one. Let’s pray
+
+
+
[removed]
+
+
+
A downvoted but real reply with negative score for edge-case coverage.
+
+
diff --git a/skills/last30days/scripts/lib/http.py b/skills/last30days/scripts/lib/http.py
index 8afd464..a8d694f 100644
--- a/skills/last30days/scripts/lib/http.py
+++ b/skills/last30days/scripts/lib/http.py
@@ -223,6 +223,53 @@ def post_raw(url: str, json_data: Dict[str, Any], headers: Optional[Dict[str, st
return request("POST", url, headers=headers, json_data=json_data, raw=True, **kwargs)
+BROWSER_USER_AGENT = (
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
+ "AppleWebKit/537.36 (KHTML, like Gecko) "
+ "Chrome/124.0.0.0 Safari/537.36"
+)
+
+
+def get_text(
+ url: str,
+ timeout: int = DEFAULT_TIMEOUT,
+ retries: int = 2,
+ accept: str = "*/*",
+ headers: Optional[Dict[str, str]] = None,
+) -> Optional[str]:
+ """Fetch a URL and return decoded text, or None on any failure.
+
+ Keyless helper for Reddit RSS and shreddit HTML endpoints — the free path
+ that replaced the now-403 ``.json`` endpoints. Sends a browser User-Agent
+ and never raises: returns None on HTTP error, network failure, or timeout
+ so tiered callers can fall through to the next source.
+
+ Args:
+ url: Request URL
+ timeout: HTTP timeout per attempt in seconds
+ retries: Number of retries on failure (kept low — these tiers fail fast)
+ accept: Accept header value (e.g. "application/atom+xml", "text/html")
+ headers: Optional extra headers merged over the defaults
+
+ Returns:
+ Decoded response body as text, or None on failure.
+ """
+ merged = {
+ "User-Agent": BROWSER_USER_AGENT,
+ "Accept": accept,
+ "Accept-Language": "en-US,en;q=0.9",
+ }
+ if headers:
+ merged.update(headers)
+ try:
+ return request(
+ "GET", url, headers=merged, timeout=timeout, retries=retries, raw=True
+ )
+ except HTTPError as e:
+ log(f"get_text failed ({e}): {url}")
+ return None
+
+
def scrapecreators_headers(token: str) -> Dict[str, str]:
"""Build ScrapeCreators request headers (x-api-key + JSON content type)."""
return {
diff --git a/skills/last30days/scripts/lib/reddit_keyless.py b/skills/last30days/scripts/lib/reddit_keyless.py
new file mode 100644
index 0000000..b1ced97
--- /dev/null
+++ b/skills/last30days/scripts/lib/reddit_keyless.py
@@ -0,0 +1,214 @@
+"""Keyless Reddit pipeline: tiered free search + comment enrichment.
+
+Replaces the dead ``.json`` free path. Discovery tiers, cheapest/most-likely
+first; enrichment then runs on whatever was discovered:
+
+ Tier 0 one-shot legacy ``.json`` search — demoted. Datacenter IPs get 403,
+ but a residential machine (where the skill usually runs) may still
+ get 200, so it is worth one cheap try. Honors the "brute-force .json"
+ intent without depending on it.
+ Tier 1 RSS discovery (reddit_rss) — keyless, robust, the load-bearing path.
+ Tier 2 shreddit comment + count enrichment (reddit_shreddit) for top posts.
+
+Returns ``[]`` (never raises) so ``pipeline.py`` can fall through to the
+ScrapeCreators backup when every keyless tier comes up empty.
+"""
+
+import concurrent.futures
+import sys
+from concurrent.futures import ThreadPoolExecutor
+from typing import Any, Dict, List, Optional
+
+from collections import Counter
+
+from . import reddit_rss, reddit_shreddit, reddit_listing
+
+ENRICH_LIMITS = reddit_shreddit.ENRICH_LIMITS
+ENRICH_BUDGET = 45 # seconds total across all enrichment threads
+MAX_ENRICH_WORKERS = 4
+MAX_DERIVED_SUBS = 5 # subreddits derived from RSS results for score backfill
+
+
+def _log(msg: str) -> None:
+ sys.stderr.write(f"[RedditKeyless] {msg}\n")
+ sys.stderr.flush()
+
+
+def _tier0_json(topic: str, depth: str) -> List[Dict[str, Any]]:
+ """One cheap global ``.json`` discovery attempt. Returns [] on the 403 wall."""
+ try:
+ from . import reddit_public
+ return reddit_public.search(topic, depth=depth) or []
+ except Exception as e: # never let the demoted tier sink the run
+ _log(f"Tier 0 (.json) unavailable: {e}")
+ return []
+
+
+def _top_subreddits(posts: List[Dict[str, Any]], limit: int = MAX_DERIVED_SUBS) -> List[str]:
+ """Most frequent subreddits across discovered posts (for score backfill)."""
+ counts = Counter(p.get("subreddit", "") for p in posts if p.get("subreddit"))
+ return [sub for sub, _ in counts.most_common(limit)]
+
+
+def _apply_scores(post: Dict[str, Any], scored: Dict[str, int]) -> None:
+ post["score"] = scored["score"]
+ post["num_comments"] = scored["num_comments"]
+ post.setdefault("engagement", {})["score"] = scored["score"]
+ post["engagement"]["num_comments"] = scored["num_comments"]
+
+
+def _discover(topic: str, depth: str, subreddits: Optional[List[str]]) -> List[Dict[str, Any]]:
+ # Tier 0: demoted one-shot .json (dead for normal users too, but free to try).
+ posts = _tier0_json(topic, depth)
+ if posts:
+ _log(f"Tier 0 (.json) returned {len(posts)} posts")
+ return posts
+
+ # Tier 1: keyless discovery. RSS gives breadth (incl. global keyword search);
+ # the listing partials give real upvote scores.
+ rss_posts = reddit_rss.search_rss(topic, depth=depth, subreddits=subreddits)
+
+ if subreddits:
+ # Targeted run: the caller chose these subreddits, so their listing cards
+ # are on-topic — include them as scored discovery AND as a score source.
+ listing_posts = reddit_listing.fetch_listings(subreddits, depth=depth, query=topic)
+ score_source = listing_posts
+ else:
+ # Bare global run: subreddits derived from noisy RSS results are NOT
+ # reliably on-topic, so their listings are used ONLY to backfill scores
+ # onto the keyword-matched RSS posts — never merged as discovery, which
+ # would flood results with high-upvote but irrelevant posts.
+ listing_posts = []
+ derived = _top_subreddits(rss_posts)
+ score_source = reddit_listing.fetch_listings(derived, depth=depth, query=topic)
+ _log(
+ f"Tier 1 (RSS) {len(rss_posts)} posts; "
+ f"{'listing discovery ' + str(len(listing_posts)) if subreddits else 'score-only'}; "
+ f"{len(score_source)} scored cards"
+ )
+
+ # Score lookup by post id, from the scored listing cards.
+ score_map: Dict[str, Dict[str, int]] = {}
+ for p in score_source:
+ pid = p.get("metadata", {}).get("post_id", "")
+ if pid:
+ score_map[pid] = {"score": p["score"], "num_comments": p["num_comments"]}
+
+ # Merge: scored listing posts first (targeted only), then RSS breadth,
+ # backfilled with real scores where the post appears in a listing.
+ merged: List[Dict[str, Any]] = []
+ seen: set = set()
+ for p in listing_posts:
+ if p["url"] not in seen:
+ seen.add(p["url"])
+ merged.append(p)
+ for p in rss_posts:
+ if p["url"] in seen:
+ continue
+ pid = reddit_listing._post_id(p["url"])
+ if pid in score_map:
+ _apply_scores(p, score_map[pid])
+ seen.add(p["url"])
+ merged.append(p)
+ return merged
+
+
+def _enrich_one(post: Dict[str, Any]) -> Dict[str, Any]:
+ """Attach shreddit comments + real comment count. Never raises."""
+ try:
+ data = reddit_shreddit.fetch_comments(post.get("url", ""))
+ if data.get("top_comments"):
+ post["top_comments"] = data["top_comments"]
+ if data.get("comment_insights"):
+ post["comment_insights"] = data["comment_insights"]
+ num = data.get("num_comments")
+ if num is not None:
+ post["num_comments"] = num
+ post.setdefault("engagement", {})["num_comments"] = num
+ except Exception:
+ pass # keep the post with whatever discovery gave us
+ return post
+
+
+def _enrich(posts: List[Dict[str, Any]], depth: str) -> List[Dict[str, Any]]:
+ """Enrich the top N posts with comments under a total time budget."""
+ limit = ENRICH_LIMITS.get(depth, ENRICH_LIMITS["default"])
+ to_enrich = posts[:limit]
+ rest = posts[limit:]
+ if not to_enrich:
+ return posts
+
+ result_map: Dict[int, Dict[str, Any]] = {}
+ try:
+ with ThreadPoolExecutor(max_workers=min(limit, MAX_ENRICH_WORKERS)) as executor:
+ futures = {
+ executor.submit(_enrich_one, post): i
+ for i, post in enumerate(to_enrich)
+ }
+ done, not_done = concurrent.futures.wait(futures, timeout=ENRICH_BUDGET)
+ for future in done:
+ idx = futures[future]
+ try:
+ result_map[idx] = future.result(timeout=0)
+ except Exception:
+ result_map[idx] = to_enrich[idx]
+ for future in not_done:
+ idx = futures[future]
+ result_map[idx] = to_enrich[idx]
+ future.cancel()
+ enriched = [result_map[i] for i in range(len(to_enrich))]
+ except Exception:
+ enriched = to_enrich
+
+ return enriched + rest
+
+
+def search_and_enrich(
+ topic: str,
+ from_date: str,
+ to_date: str,
+ depth: str = "default",
+ subreddits: Optional[List[str]] = None,
+) -> List[Dict[str, Any]]:
+ """Full keyless Reddit pipeline: discover (Tier 0/1) then enrich (Tier 2).
+
+ Args:
+ topic: Search topic
+ from_date: Start date (YYYY-MM-DD)
+ to_date: End date (YYYY-MM-DD)
+ depth: 'quick', 'default', or 'deep'
+ subreddits: Optional pre-resolved subreddit names (without r/)
+
+ Returns:
+ List of normalized item dicts matching the reddit_public output shape,
+ with top_comments/comment_insights attached on enriched posts.
+ Empty list when all keyless tiers fail (so SC backup can engage).
+ """
+ posts = _discover(topic, depth, subreddits)
+ if not posts:
+ return []
+
+ # Date filter: keep posts in range or with unknown dates (mirrors reddit_public).
+ posts = [
+ p for p in posts
+ if p.get("date") is None or (from_date <= p["date"] <= to_date)
+ ]
+
+ # Rank before enrichment by real upvote score (from listing cards / backfill),
+ # then query relevance, then recency. Posts without a recovered score sort by
+ # the latter two — same behavior as before scores were available.
+ posts.sort(
+ key=lambda p: (
+ p.get("engagement", {}).get("score", 0) or 0,
+ p.get("relevance", 0) or 0,
+ p.get("date") or "",
+ ),
+ reverse=True,
+ )
+
+ posts = _enrich(posts, depth)
+
+ for i, post in enumerate(posts):
+ post["id"] = f"R{i + 1}"
+
+ return posts
diff --git a/skills/last30days/scripts/lib/reddit_listing.py b/skills/last30days/scripts/lib/reddit_listing.py
new file mode 100644
index 0000000..7f2fccc
--- /dev/null
+++ b/skills/last30days/scripts/lib/reddit_listing.py
@@ -0,0 +1,183 @@
+"""Keyless Reddit listing scrape via shreddit /svc partials — with real scores.
+
+The subreddit listing partial
+``/svc/shreddit/community-more-posts/{sort}/?name={sub}[&t={range}]`` serves
+HTTP 200 with no API key and **server-renders each post's upvote score**, which
+neither RSS nor the comments endpoint provides. Each post is a
+```` element whose start-tag attributes carry ``score``,
+``comment-count``, ``post-title``, ``permalink``, ``author``, ``subreddit-name``
+and ``created-timestamp``.
+
+This is the keyless source of post-level upvotes. It works for normal users on
+ordinary connections (verified), so reddit_keyless uses it both as a scored
+discovery source and to backfill scores onto RSS-discovered posts.
+"""
+
+import html as _html
+import re
+import sys
+from datetime import datetime, timezone
+from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeoutError
+from typing import Any, Dict, List, Optional
+
+from . import http
+from .relevance import token_overlap_relevance
+
+# Listing sorts pulled per subreddit, by depth.
+LISTING_SORTS = {
+ "quick": ["top"],
+ "default": ["top", "hot"],
+ "deep": ["top", "hot", "new"],
+}
+DEPTH_LIMITS = {"quick": 10, "default": 25, "deep": 50}
+TIMEFRAME = "month"
+MAX_WORKERS = 4
+LISTING_TIMEOUT = 15
+
+_POST_CARD = re.compile(r"])[^>]*>")
+
+
+def _log(msg: str) -> None:
+ sys.stderr.write(f"[RedditListing] {msg}\n")
+ sys.stderr.flush()
+
+
+def _attr(tag: str, name: str) -> Optional[str]:
+ m = re.search(rf'\b{name}="([^"]*)"', tag)
+ return _html.unescape(m.group(1)) if m else None
+
+
+def _to_date(value: Optional[str]) -> Optional[str]:
+ if not value:
+ return None
+ try:
+ return datetime.fromisoformat(value.strip()).date().isoformat()
+ except (ValueError, TypeError):
+ return None
+
+
+def _to_epoch(value: Optional[str]) -> Optional[float]:
+ if not value:
+ return None
+ try:
+ dt = datetime.fromisoformat(value.strip())
+ if dt.tzinfo is None:
+ dt = dt.replace(tzinfo=timezone.utc)
+ return dt.timestamp()
+ except (ValueError, TypeError):
+ return None
+
+
+def _post_id(permalink: str) -> str:
+ m = re.search(r"/comments/([A-Za-z0-9]+)", permalink or "")
+ return m.group(1) if m else ""
+
+
+def parse_cards(html_text: str, query: str = "") -> List[Dict[str, Any]]:
+ """Parse cards into normalized post dicts with real scores."""
+ posts: List[Dict[str, Any]] = []
+ for m in _POST_CARD.finditer(html_text or ""):
+ tag = m.group(0)
+ permalink = _attr(tag, "permalink") or ""
+ if "/comments/" not in permalink:
+ continue
+ try:
+ score = int(_attr(tag, "score") or 0)
+ except ValueError:
+ score = 0
+ try:
+ num_comments = int(_attr(tag, "comment-count") or 0)
+ except ValueError:
+ num_comments = 0
+ title = _attr(tag, "post-title") or ""
+ author = _attr(tag, "author") or "[deleted]"
+ subreddit = _attr(tag, "subreddit-name") or ""
+ created = _attr(tag, "created-timestamp")
+ url = f"https://www.reddit.com{permalink}"
+
+ posts.append({
+ "id": "",
+ "title": title,
+ "url": url,
+ "score": score,
+ "num_comments": num_comments,
+ "subreddit": subreddit,
+ "created_utc": _to_epoch(created),
+ "author": author if author not in ("[deleted]", "[removed]") else "[deleted]",
+ "selftext": "",
+ "date": _to_date(created),
+ "engagement": {
+ "score": score,
+ "num_comments": num_comments,
+ "upvote_ratio": None,
+ },
+ "relevance": round(token_overlap_relevance(query, title), 3) if query else 0.0,
+ "why_relevant": "Reddit listing",
+ "metadata": {"post_id": _post_id(permalink)},
+ })
+ return posts
+
+
+def _listing_url(subreddit: str, sort: str) -> str:
+ sub = subreddit.removeprefix("r/").strip()
+ url = f"https://www.reddit.com/svc/shreddit/community-more-posts/{sort}/?name={sub}"
+ if sort == "top":
+ url += f"&t={TIMEFRAME}"
+ return url
+
+
+def _fetch_one(subreddit: str, sort: str, query: str) -> List[Dict[str, Any]]:
+ try:
+ text = http.get_text(_listing_url(subreddit, sort), timeout=LISTING_TIMEOUT,
+ accept="text/html")
+ return parse_cards(text, query) if text else []
+ except Exception as e:
+ _log(f"listing fetch failed r/{subreddit} {sort}: {e}")
+ return []
+
+
+def fetch_listings(
+ subreddits: List[str],
+ depth: str = "default",
+ query: str = "",
+) -> List[Dict[str, Any]]:
+ """Fetch scored post cards across subreddits × depth-appropriate sorts.
+
+ Returns deduped normalized posts (with real scores), unranked/unsliced —
+ the caller merges these with other sources, ranks, and slices.
+ """
+ if not subreddits:
+ return []
+ sorts = LISTING_SORTS.get(depth, LISTING_SORTS["default"])
+ jobs = [(sub, sort) for sub in subreddits for sort in sorts]
+ all_posts: List[Dict[str, Any]] = []
+ with ThreadPoolExecutor(max_workers=min(MAX_WORKERS, len(jobs)) or 1) as executor:
+ futures = {executor.submit(_fetch_one, sub, sort, query): (sub, sort)
+ for sub, sort in jobs}
+ for future in futures:
+ try:
+ all_posts.extend(future.result(timeout=LISTING_TIMEOUT + 5))
+ except (Exception, FuturesTimeoutError) as e:
+ _log(f"listing future failed: {e}")
+
+ seen: set = set()
+ unique: List[Dict[str, Any]] = []
+ for p in all_posts:
+ if p["url"] not in seen:
+ seen.add(p["url"])
+ unique.append(p)
+ return unique
+
+
+def score_index(subreddits: List[str], depth: str = "default") -> Dict[str, Dict[str, int]]:
+ """Build a {post_id: {score, num_comments}} map from subreddit listings.
+
+ Used to backfill real scores onto posts discovered via RSS, which carries
+ no engagement numbers.
+ """
+ index: Dict[str, Dict[str, int]] = {}
+ for p in fetch_listings(subreddits, depth=depth):
+ pid = p.get("metadata", {}).get("post_id") or _post_id(p["url"])
+ if pid:
+ index[pid] = {"score": p["score"], "num_comments": p["num_comments"]}
+ return index
diff --git a/skills/last30days/scripts/lib/reddit_public.py b/skills/last30days/scripts/lib/reddit_public.py
index e8f3dc8..b6789b0 100644
--- a/skills/last30days/scripts/lib/reddit_public.py
+++ b/skills/last30days/scripts/lib/reddit_public.py
@@ -1,9 +1,16 @@
-"""Standalone Reddit public JSON search module.
+"""Reddit public ``.json`` search module (demoted to keyless Tier 0).
-Searches Reddit using the free public JSON endpoints (no API key required).
-Promoted from last-resort fallback to robust primary free path.
+Reddit's public ``.json`` endpoints now return HTTP 403 from most contexts
+(shreddit anti-bot), so this is no longer the primary free path. The keyless
+pipeline (see reddit_keyless.py) still calls ``search`` as a cheap one-shot
+Tier 0 attempt — a residential machine may occasionally get a 200 — before
+falling through to RSS discovery (reddit_rss.py) and shreddit comment
+enrichment (reddit_shreddit.py).
-Endpoints:
+``search_reddit_public`` is retained as a compatibility shim that delegates to
+the keyless pipeline, so existing callers (pipeline.py) need no change.
+
+Endpoints (Tier 0):
- Global: https://www.reddit.com/search.json?q={query}&sort=relevance&t=month&limit={limit}
- Subreddit: https://www.reddit.com/r/{sub}/search.json?q={query}&restrict_sr=on&sort=relevance&t=month
@@ -18,7 +25,6 @@ import time
import urllib.error
import urllib.parse
import urllib.request
-from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeoutError
from typing import Any, Dict, List, Optional
@@ -35,13 +41,6 @@ DEPTH_LIMITS = {
"deep": 50,
}
-# How many top posts to enrich with comments, by depth
-ENRICH_LIMITS = {
- "quick": 3,
- "default": 5,
- "deep": 8,
-}
-
MAX_RETRIES = 3
BASE_BACKOFF = 2.0 # seconds
@@ -237,78 +236,6 @@ def search(
return unique[:limit]
-def _enrich_post(item: Dict[str, Any], timeout: int = 10) -> Dict[str, Any]:
- """Enrich a single post with top comments. Never raises."""
- try:
- from . import reddit_enrich
- thread_data = reddit_enrich.fetch_thread_data(item["url"], timeout=timeout)
- if not thread_data:
- return item
- parsed = reddit_enrich.parse_thread_data(thread_data)
- comments = parsed.get("comments", [])
- top = reddit_enrich.get_top_comments(comments)
- item["top_comments"] = [
- {
- "score": c.get("score", 0),
- "excerpt": (c.get("body") or "")[:200],
- "author": c.get("author", ""),
- }
- for c in top[:10]
- ]
- except Exception:
- # Never discard — keep post with empty metadata
- pass
- return item
-
-
-def _enrich_posts(posts: List[Dict[str, Any]], depth: str = "default") -> List[Dict[str, Any]]:
- """Enrich top N posts with comment data using threads. Total budget 45s."""
- limit = ENRICH_LIMITS.get(depth, ENRICH_LIMITS["default"])
- to_enrich = posts[:limit]
- rest = posts[limit:]
-
- if not to_enrich:
- return posts
-
- enriched = []
- try:
- with ThreadPoolExecutor(max_workers=min(limit, 4)) as executor:
- futures = {
- executor.submit(_enrich_post, post, 10): i
- for i, post in enumerate(to_enrich)
- }
- # Collect results with 45s total budget
- import concurrent.futures
- done, not_done = concurrent.futures.wait(futures, timeout=45)
- # Build result list preserving order
- result_map: Dict[int, Dict[str, Any]] = {}
- for future in done:
- idx = futures[future]
- try:
- result_map[idx] = future.result(timeout=0)
- except Exception:
- result_map[idx] = to_enrich[idx]
- # Any not-done futures: keep original post
- for future in not_done:
- idx = futures[future]
- result_map[idx] = to_enrich[idx]
- future.cancel()
- enriched = [result_map[i] for i in range(len(to_enrich))]
- except Exception:
- enriched = to_enrich
-
- return enriched + rest
-
-
-def _search_subreddit(sub: str, topic: str, depth: str, timeout: int = 15) -> List[Dict[str, Any]]:
- """Search a single subreddit. Never raises."""
- try:
- return search(topic, depth=depth, subreddit=sub, timeout=timeout)
- except Exception as e:
- _log(f"Subreddit search failed for r/{sub}: {e}")
- return []
-
-
def search_reddit_public(
topic: str,
from_date: str,
@@ -316,12 +243,17 @@ def search_reddit_public(
depth: str = "default",
subreddits: Optional[List[str]] = None,
) -> List[Dict[str, Any]]:
- """High-level Reddit public search matching the openai_reddit interface.
+ """High-level free Reddit search + enrichment (keyless).
- When subreddits are provided (from agent planning), searches each targeted
- sub first, then does global search, and deduplicates across both. This
- mirrors the SC search_and_enrich() flow where pre-resolved subreddits get
- priority.
+ Thin compatibility shim over the tiered keyless pipeline: the legacy
+ ``.json`` search/enrichment endpoints now return HTTP 403, so this delegates
+ to ``reddit_keyless.search_and_enrich`` (Tier 0 one-shot ``.json`` →
+ Tier 1 RSS discovery → Tier 2 shreddit comment enrichment). The name and
+ signature are preserved so ``pipeline.py`` and other callers need no change
+ and the ScrapeCreators backup still engages when this returns empty.
+
+ The module-level ``search`` / ``_parse_posts`` helpers remain in use as the
+ keyless pipeline's demoted Tier 0 ``.json`` attempt.
Args:
topic: Search topic
@@ -332,57 +264,9 @@ def search_reddit_public(
Returns:
List of normalized item dicts matching ScrapeCreators output format.
+ Empty list on total failure (so SC backup can engage).
"""
- all_posts: List[Dict[str, Any]] = []
-
- # Phase 1: Search targeted subreddits in parallel (if provided)
- if subreddits:
- _log(f"Searching {len(subreddits)} targeted subreddits: {subreddits}")
- workers = min(4, len(subreddits))
- with ThreadPoolExecutor(max_workers=workers) as executor:
- futures = {
- executor.submit(_search_subreddit, sub, topic, depth): sub
- for sub in subreddits
- }
- for future in futures:
- sub = futures[future]
- try:
- sub_posts = future.result(timeout=30)
- _log(f" -> {len(sub_posts)} results from r/{sub}")
- all_posts.extend(sub_posts)
- except (Exception, FuturesTimeoutError) as e:
- _log(f" -> r/{sub} failed: {e}")
-
- # Phase 2: Global search
- global_posts = search(topic, depth=depth)
- all_posts.extend(global_posts)
-
- # Deduplicate by URL (targeted results keep priority since they come first)
- seen_urls: set = set()
- results: List[Dict[str, Any]] = []
- for post in all_posts:
- if post["url"] not in seen_urls:
- seen_urls.add(post["url"])
- results.append(post)
-
- # Date filter: keep posts in range or with unknown dates
- filtered = []
- for item in results:
- d = item.get("date")
- if d is None or (from_date <= d <= to_date):
- filtered.append(item)
-
- # Sort by engagement (score desc)
- filtered.sort(
- key=lambda x: x.get("engagement", {}).get("score", 0),
- reverse=True,
+ from . import reddit_keyless
+ return reddit_keyless.search_and_enrich(
+ topic, from_date, to_date, depth=depth, subreddits=subreddits
)
-
- # Enrich top posts with comments
- filtered = _enrich_posts(filtered, depth=depth)
-
- # Re-index IDs
- for i, item in enumerate(filtered):
- item["id"] = f"R{i + 1}"
-
- return filtered
diff --git a/skills/last30days/scripts/lib/reddit_rss.py b/skills/last30days/scripts/lib/reddit_rss.py
new file mode 100644
index 0000000..b7c3576
--- /dev/null
+++ b/skills/last30days/scripts/lib/reddit_rss.py
@@ -0,0 +1,224 @@
+"""Keyless Reddit discovery via public RSS/Atom feeds.
+
+Reddit's ``.json`` search endpoints now return HTTP 403 (shreddit anti-bot).
+RSS feeds still serve HTTP 200 with no API key, so this module uses them for
+post discovery, replacing ``reddit_public.search`` as the free search path.
+
+Two feed families are combined and deduped:
+- search: /search.rss?q=... and /r/{sub}/search.rss?q=...&restrict_sr=on
+- listing: /r/{sub}/{top,hot}.rss?t=month
+
+RSS entries carry no engagement score, so ``score``/``num_comments`` start at 0
+and are backfilled during shreddit enrichment (see reddit_shreddit.py). Output
+dicts match the normalized shape emitted by ``reddit_public._parse_posts`` so
+downstream code (pipeline, renderer) is unaffected.
+"""
+
+import sys
+import xml.etree.ElementTree as ET
+from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeoutError
+from datetime import datetime, timezone
+from typing import Any, Dict, List, Optional
+from urllib.parse import quote_plus
+
+from . import http
+from .relevance import token_overlap_relevance
+
+ATOM = "{http://www.w3.org/2005/Atom}"
+
+# Mirror reddit_public depth-aware limits so the two free paths behave alike.
+DEPTH_LIMITS = {
+ "quick": 10,
+ "default": 25,
+ "deep": 50,
+}
+
+# Listing sorts pulled per subreddit (in addition to search), for volume.
+LISTING_SORTS = {
+ "quick": ["top"],
+ "default": ["top", "hot"],
+ "deep": ["top", "hot", "new"],
+}
+
+MAX_WORKERS = 4
+FEED_TIMEOUT = 15
+
+
+def _log(msg: str) -> None:
+ sys.stderr.write(f"[RedditRSS] {msg}\n")
+ sys.stderr.flush()
+
+
+def _iso_to_date(value: Optional[str]) -> Optional[str]:
+ """Parse an ISO-8601 timestamp (e.g. 2026-05-20T18:48:31+00:00) to YYYY-MM-DD."""
+ if not value:
+ return None
+ try:
+ dt = datetime.fromisoformat(value.strip())
+ return dt.date().isoformat()
+ except (ValueError, TypeError):
+ return None
+
+
+def _iso_to_epoch(value: Optional[str]) -> Optional[float]:
+ if not value:
+ return None
+ try:
+ dt = datetime.fromisoformat(value.strip())
+ if dt.tzinfo is None:
+ dt = dt.replace(tzinfo=timezone.utc)
+ return dt.timestamp()
+ except (ValueError, TypeError):
+ return None
+
+
+def _subreddit_from(category: str, url: str) -> str:
+ """Derive subreddit name from the entry category or, failing that, the URL."""
+ if category:
+ return category
+ # URL form: https://www.reddit.com/r/{sub}/comments/{id}/...
+ parts = url.split("/r/", 1)
+ if len(parts) == 2:
+ return parts[1].split("/", 1)[0]
+ return ""
+
+
+def _parse_feed(xml_text: str, query: str = "") -> List[Dict[str, Any]]:
+ """Parse an Atom feed string into normalized post dicts. Never raises."""
+ if not xml_text:
+ return []
+ try:
+ root = ET.fromstring(xml_text)
+ except ET.ParseError as e:
+ _log(f"feed parse error: {e}")
+ return []
+
+ posts: List[Dict[str, Any]] = []
+ for entry in root.iter(f"{ATOM}entry"):
+ link_el = entry.find(f"{ATOM}link")
+ url = link_el.get("href", "").strip() if link_el is not None else ""
+ if not url or "/comments/" not in url:
+ continue
+
+ title_el = entry.find(f"{ATOM}title")
+ title = (title_el.text or "").strip() if title_el is not None else ""
+
+ author = ""
+ author_el = entry.find(f"{ATOM}author/{ATOM}name")
+ if author_el is not None and author_el.text:
+ author = author_el.text.strip().removeprefix("/u/").removeprefix("u/")
+ if author in ("[deleted]", "[removed]", ""):
+ author = "[deleted]"
+
+ cat_el = entry.find(f"{ATOM}category")
+ category = cat_el.get("term", "").strip() if cat_el is not None else ""
+ subreddit = _subreddit_from(category, url)
+
+ updated_el = entry.find(f"{ATOM}updated")
+ updated = (updated_el.text or "").strip() if updated_el is not None else ""
+
+ content_el = entry.find(f"{ATOM}content")
+ selftext = ""
+ if content_el is not None and content_el.text:
+ # Strip the simplest HTML; renderer only needs an excerpt.
+ import re as _re
+ selftext = _re.sub(r"<[^>]+>", " ", content_el.text)
+ selftext = _re.sub(r"\s+", " ", selftext).strip()[:500]
+
+ relevance = round(token_overlap_relevance(query, title), 3) if query else 0.0
+
+ posts.append({
+ "id": "", # assigned after dedup
+ "title": title,
+ "url": url,
+ "score": 0, # backfilled by shreddit enrichment
+ "num_comments": 0, # backfilled by shreddit enrichment
+ "subreddit": subreddit,
+ "created_utc": _iso_to_epoch(updated),
+ "author": author,
+ "selftext": selftext,
+ "date": _iso_to_date(updated),
+ "engagement": {
+ "score": 0,
+ "num_comments": 0,
+ "upvote_ratio": None,
+ },
+ "relevance": relevance,
+ "why_relevant": "Reddit RSS",
+ "metadata": {},
+ })
+
+ return posts
+
+
+def _build_urls(query: str, depth: str, subreddits: Optional[List[str]]) -> List[str]:
+ """Build the keyless RSS feed URLs to fan out across."""
+ q = quote_plus(query)
+ urls: List[str] = [
+ f"https://www.reddit.com/search.rss?q={q}&sort=relevance&t=month"
+ ]
+ for raw_sub in (subreddits or []):
+ sub = raw_sub.removeprefix("r/").strip()
+ if not sub:
+ continue
+ urls.append(
+ f"https://www.reddit.com/r/{sub}/search.rss"
+ f"?q={q}&restrict_sr=on&sort=relevance&t=month"
+ )
+ for sort in LISTING_SORTS.get(depth, LISTING_SORTS["default"]):
+ urls.append(f"https://www.reddit.com/r/{sub}/{sort}.rss?t=month")
+ return urls
+
+
+def _fetch_feed(url: str, query: str) -> List[Dict[str, Any]]:
+ """Fetch and parse one feed. Never raises."""
+ try:
+ text = http.get_text(url, timeout=FEED_TIMEOUT, accept="application/atom+xml")
+ return _parse_feed(text, query) if text else []
+ except Exception as e: # defensive: a single bad feed must not sink the run
+ _log(f"feed fetch failed for {url}: {e}")
+ return []
+
+
+def search_rss(
+ query: str,
+ depth: str = "default",
+ subreddits: Optional[List[str]] = None,
+) -> List[Dict[str, Any]]:
+ """Discover Reddit posts for a query via keyless RSS feeds.
+
+ Args:
+ query: Search query string
+ depth: 'quick', 'default', or 'deep' — controls result limit and feeds
+ subreddits: Optional pre-resolved subreddit names (without r/) to target
+
+ Returns:
+ List of normalized post dicts (deduped by URL, capped by depth),
+ with placeholder scores to be backfilled during enrichment.
+ Empty list on any failure.
+ """
+ limit = DEPTH_LIMITS.get(depth, DEPTH_LIMITS["default"])
+ urls = _build_urls(query, depth, subreddits)
+
+ all_posts: List[Dict[str, Any]] = []
+ workers = min(MAX_WORKERS, len(urls)) or 1
+ with ThreadPoolExecutor(max_workers=workers) as executor:
+ futures = {executor.submit(_fetch_feed, url, query): url for url in urls}
+ for future in futures:
+ try:
+ all_posts.extend(future.result(timeout=FEED_TIMEOUT + 5))
+ except (Exception, FuturesTimeoutError) as e:
+ _log(f"feed future failed: {e}")
+
+ # Dedupe by URL (first occurrence wins).
+ seen: set = set()
+ unique: List[Dict[str, Any]] = []
+ for post in all_posts:
+ if post["url"] not in seen:
+ seen.add(post["url"])
+ unique.append(post)
+
+ for i, post in enumerate(unique):
+ post["id"] = f"R{i + 1}"
+
+ return unique[:limit]
diff --git a/skills/last30days/scripts/lib/reddit_shreddit.py b/skills/last30days/scripts/lib/reddit_shreddit.py
new file mode 100644
index 0000000..d222ac0
--- /dev/null
+++ b/skills/last30days/scripts/lib/reddit_shreddit.py
@@ -0,0 +1,184 @@
+"""Keyless Reddit comment enrichment via shreddit /svc endpoints.
+
+Reddit's ``{thread}.json`` endpoint now returns HTTP 403. The shreddit partial
+endpoint ``/svc/shreddit/comments/r/{sub}/t3_{id}`` still serves HTTP 200 HTML
+with no API key, embedding each comment as a ```` custom
+element whose start-tag attributes carry ``score`` / ``author`` / ``created`` /
+``permalink``, and whose body lives in a ``
``
+block. This module parses that markup into top comments, matching the
+``top_comments`` / ``comment_insights`` shape produced by ``reddit_enrich`` so
+the renderer is unaffected.
+
+Limitation: the comments endpoint carries the real comment count
+(``total-comments``) but not the post's upvote score, so post-level ``score``
+cannot be recovered keylessly here (ScrapeCreators backup still provides it).
+"""
+
+import html as _html
+import re
+import sys
+from datetime import datetime
+from typing import Any, Dict, List, Optional
+
+from . import http
+from . import reddit_enrich
+
+# Up to N posts enriched per run, by depth (mirrors reddit_public.ENRICH_LIMITS).
+ENRICH_LIMITS = {
+ "quick": 3,
+ "default": 5,
+ "deep": 8,
+}
+
+# Max comments returned per post (independent of how many posts get enriched).
+MAX_COMMENTS = 10
+
+SVC_TIMEOUT = 12
+
+# Match the exact element start tag, not
+# or (lookahead requires whitespace or '>').
+_COMMENT_START = re.compile(r"])[^>]*>")
+_TOTAL_COMMENTS = re.compile(r'total-comments="(\d+)"')
+_PARA = re.compile(r"
]*>(.*?)
", re.S)
+_TAG = re.compile(r"<[^>]+>")
+_WS = re.compile(r"\s+")
+_NEXT_RTJSON = re.compile(r'id="t1_[A-Za-z0-9]+-(?:comment|post)-rtjson-content"')
+
+
+def _log(msg: str) -> None:
+ sys.stderr.write(f"[RedditShreddit] {msg}\n")
+ sys.stderr.flush()
+
+
+def extract_post_ref(url: str) -> Optional[tuple]:
+ """Return (subreddit, post_id) from a Reddit thread URL, or None."""
+ m = re.search(r"/r/([^/]+)/comments/([A-Za-z0-9]+)", url or "")
+ if not m:
+ return None
+ return m.group(1), m.group(2)
+
+
+def _svc_url(subreddit: str, post_id: str) -> str:
+ # sort=top guarantees Reddit front-loads the highest-scored comments on the
+ # first page, so the true top comments are captured even on huge threads
+ # (we still re-sort by score locally as a backstop).
+ return (
+ f"https://www.reddit.com/svc/shreddit/comments/r/{subreddit}/t3_{post_id}"
+ f"?sort=top"
+ )
+
+
+def _attr(tag: str, name: str) -> str:
+ m = re.search(rf'\b{name}="([^"]*)"', tag)
+ return _html.unescape(m.group(1)) if m else ""
+
+
+def _iso_to_date(value: str) -> Optional[str]:
+ if not value:
+ return None
+ try:
+ return datetime.fromisoformat(value.strip()).date().isoformat()
+ except (ValueError, TypeError):
+ return None
+
+
+def _body_for(html_text: str, thing_id: str) -> str:
+ """Extract a comment's text body, anchored on its unique thingId.
+
+ The body div id embeds the comment's thingId, so this assigns body→comment
+ correctly even for nested replies. The slice is bounded by the next
+ comment's rtjson anchor to avoid swallowing child-comment text.
+ """
+ if not thing_id:
+ return ""
+ anchor = f'id="{thing_id}-post-rtjson-content"'
+ idx = html_text.find(anchor)
+ if idx == -1:
+ return ""
+ window = html_text[idx + len(anchor): idx + len(anchor) + 8000]
+ nxt = _NEXT_RTJSON.search(window)
+ if nxt:
+ window = window[: nxt.start()]
+ paras = _PARA.findall(window)
+ if not paras:
+ return ""
+ text = " ".join(_TAG.sub("", p) for p in paras)
+ return _WS.sub(" ", _html.unescape(text)).strip()
+
+
+def parse_comments(html_text: str, limit: int = MAX_COMMENTS) -> List[Dict[str, Any]]:
+ """Parse elements into scored comment dicts (sorted desc)."""
+ comments: List[Dict[str, Any]] = []
+ for m in _COMMENT_START.finditer(html_text or ""):
+ tag = m.group(0)
+ author = _attr(tag, "author") or "[deleted]"
+ if author in ("[deleted]", "[removed]"):
+ continue
+ thing_id = _attr(tag, "thingId")
+ body = _body_for(html_text, thing_id)
+ if not body or body in ("[deleted]", "[removed]"):
+ continue
+ try:
+ score = int(_attr(tag, "score") or 0)
+ except ValueError:
+ score = 0
+ permalink = _attr(tag, "permalink")
+ comments.append({
+ "score": score,
+ "author": author,
+ "body": body[:300],
+ "excerpt": body[:200],
+ "permalink": permalink,
+ "date": _iso_to_date(_attr(tag, "created")),
+ "url": f"https://reddit.com{permalink}" if permalink else "",
+ })
+
+ comments.sort(key=lambda c: c.get("score", 0), reverse=True)
+ return comments[:limit]
+
+
+def _total_comments(html_text: str) -> Optional[int]:
+ m = _TOTAL_COMMENTS.search(html_text or "")
+ return int(m.group(1)) if m else None
+
+
+def fetch_comments(
+ post_url: str,
+ timeout: int = SVC_TIMEOUT,
+) -> Dict[str, Any]:
+ """Fetch and parse top comments for a Reddit post via the shreddit endpoint.
+
+ Args:
+ post_url: Reddit thread URL (…/r/{sub}/comments/{id}/…)
+ timeout: HTTP timeout in seconds
+
+ Returns:
+ Dict with 'top_comments' (list, reddit_enrich shape), 'comment_insights'
+ (list[str]), and 'num_comments' (int or None). Empty/None on any
+ failure — never raises, so the caller can fall through to SC backup.
+ """
+ ref = extract_post_ref(post_url)
+ if not ref:
+ return {"top_comments": [], "comment_insights": [], "num_comments": None}
+ sub, post_id = ref
+
+ html_text = http.get_text(_svc_url(sub, post_id), timeout=timeout, accept="text/html")
+ if not html_text:
+ return {"top_comments": [], "comment_insights": [], "num_comments": None}
+
+ comments = parse_comments(html_text, limit=MAX_COMMENTS)
+ insights = reddit_enrich.extract_comment_insights(comments)
+ return {
+ "top_comments": [
+ {
+ "score": c["score"],
+ "date": c["date"],
+ "author": c["author"],
+ "excerpt": c["excerpt"],
+ "url": c["url"],
+ }
+ for c in comments
+ ],
+ "comment_insights": insights,
+ "num_comments": _total_comments(html_text),
+ }
diff --git a/tests/test_reddit_keyless.py b/tests/test_reddit_keyless.py
new file mode 100644
index 0000000..26de05c
--- /dev/null
+++ b/tests/test_reddit_keyless.py
@@ -0,0 +1,168 @@
+"""Tests for scripts/lib/reddit_keyless.py — tiered keyless Reddit pipeline."""
+
+from unittest import mock
+
+from lib import reddit_keyless
+
+
+def _post(i, date="2026-05-20", rel=0.0):
+ url = f"https://www.reddit.com/r/test/comments/{i:06d}/post_{i}/"
+ return {
+ "id": "", "title": f"Post {i}", "url": url, "score": 0, "num_comments": 0,
+ "subreddit": "test", "created_utc": None, "author": "u", "selftext": "",
+ "date": date, "engagement": {"score": 0, "num_comments": 0, "upvote_ratio": None},
+ "relevance": rel, "why_relevant": "Reddit RSS", "metadata": {},
+ }
+
+
+def _scored(i, score, ncmt=0):
+ p = _post(i)
+ p["score"] = score
+ p["num_comments"] = ncmt
+ p["engagement"]["score"] = score
+ p["engagement"]["num_comments"] = ncmt
+ p["why_relevant"] = "Reddit listing"
+ p["metadata"] = {"post_id": f"{i:06d}"}
+ return p
+
+
+class TestDiscoveryTierOrder:
+ """Tier 0 (.json) is tried first; RSS + scored listings are the keyless path."""
+
+ def test_tier0_success_skips_keyless(self):
+ with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[_post(1)]) as t0, \
+ mock.patch.object(reddit_keyless.reddit_rss, "search_rss") as rss, \
+ mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings") as lst:
+ out = reddit_keyless._discover("topic", "default", None)
+ assert len(out) == 1
+ t0.assert_called_once()
+ rss.assert_not_called()
+ lst.assert_not_called()
+
+ def test_tier0_empty_falls_to_keyless(self):
+ with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \
+ mock.patch.object(reddit_keyless.reddit_rss, "search_rss",
+ return_value=[_post(1), _post(2)]) as rss, \
+ mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings",
+ return_value=[]):
+ out = reddit_keyless._discover("topic", "default", ["test"])
+ assert len(out) == 2
+ rss.assert_called_once()
+
+ def test_listing_scores_backfill_rss_posts(self):
+ # RSS finds post 1 (no score); listing card for post 1 carries the score.
+ rss_post = _post(1)
+ listing_post = _scored(1, score=52692, ncmt=1743)
+ with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \
+ mock.patch.object(reddit_keyless.reddit_rss, "search_rss",
+ return_value=[rss_post]), \
+ mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings",
+ return_value=[listing_post]):
+ out = reddit_keyless._discover("topic", "default", ["test"])
+ # listing post (scored) is kept; RSS dup of same url is dropped
+ assert len(out) == 1
+ assert out[0]["engagement"]["score"] == 52692
+ assert out[0]["num_comments"] == 1743
+
+ def test_scores_flow_to_distinct_rss_posts(self):
+ # Distinct RSS post whose id matches a listing card gets backfilled.
+ rss_post = _post(7) # url .../000007/...
+ listing_post = _scored(7, score=999)
+ listing_post["url"] = "https://www.reddit.com/r/test/comments/zzzzzz/other/"
+ with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \
+ mock.patch.object(reddit_keyless.reddit_rss, "search_rss",
+ return_value=[rss_post]), \
+ mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings",
+ return_value=[listing_post]):
+ out = reddit_keyless._discover("topic", "default", ["test"])
+ backfilled = [p for p in out if p["url"] == rss_post["url"]][0]
+ assert backfilled["engagement"]["score"] == 999
+
+ def test_bare_query_does_not_merge_listing_discovery(self):
+ # No subreddits provided: derived-subreddit listings must NOT be added as
+ # results (avoids flooding with off-topic high-upvote posts) — only used
+ # to backfill scores onto the keyword-matched RSS posts.
+ rss_post = _post(1) # on-topic keyword match
+ offtopic_listing = _scored(99, score=88888) # high score, unrelated sub
+ offtopic_listing["url"] = "https://www.reddit.com/r/random/comments/zzz999/x/"
+ with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \
+ mock.patch.object(reddit_keyless.reddit_rss, "search_rss",
+ return_value=[rss_post]), \
+ mock.patch.object(reddit_keyless, "_top_subreddits", return_value=["random"]), \
+ mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings",
+ return_value=[offtopic_listing]):
+ out = reddit_keyless._discover("topic", "default", None)
+ urls = [p["url"] for p in out]
+ assert rss_post["url"] in urls
+ assert offtopic_listing["url"] not in urls # not merged as discovery
+
+ def test_tier0_never_raises(self):
+ with mock.patch("lib.reddit_public.search", side_effect=Exception("boom")), \
+ mock.patch.object(reddit_keyless.reddit_rss, "search_rss", return_value=[]), \
+ mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", return_value=[]):
+ assert reddit_keyless._discover("t", "default", None) == []
+
+
+class TestSearchAndEnrich:
+ """Full pipeline: discover -> date filter -> rank -> enrich -> reindex."""
+
+ def _patch_enrich_passthrough(self):
+ return mock.patch.object(
+ reddit_keyless.reddit_shreddit, "fetch_comments",
+ return_value={"top_comments": [], "comment_insights": [], "num_comments": None},
+ )
+
+ def test_returns_empty_when_no_discovery(self):
+ with mock.patch.object(reddit_keyless, "_discover", return_value=[]):
+ assert reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") == []
+
+ def test_date_filter_keeps_in_range_and_unknown(self):
+ posts = [_post(1, date="2026-05-10"), _post(2, date="2020-01-01"),
+ _post(3, date=None)]
+ with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \
+ self._patch_enrich_passthrough():
+ out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31")
+ titles = {p["title"] for p in out}
+ assert "Post 1" in titles and "Post 3" in titles
+ assert "Post 2" not in titles
+
+ def test_reindexes_ids(self):
+ posts = [_post(1), _post(2), _post(3)]
+ with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \
+ self._patch_enrich_passthrough():
+ out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31")
+ assert [p["id"] for p in out] == ["R1", "R2", "R3"]
+
+ def test_enrichment_attaches_comments(self):
+ posts = [_post(1)]
+ enriched = {
+ "top_comments": [{"score": 9, "date": "2026-05-19", "author": "a",
+ "excerpt": "great", "url": "https://reddit.com/x"}],
+ "comment_insights": ["great point about X"],
+ "num_comments": 14,
+ }
+ with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \
+ mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments",
+ return_value=enriched):
+ out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31")
+ assert out[0]["top_comments"][0]["score"] == 9
+ assert out[0]["num_comments"] == 14
+ assert out[0]["engagement"]["num_comments"] == 14
+
+ def test_enrichment_failure_keeps_posts(self):
+ posts = [_post(i) for i in range(8)]
+ with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \
+ mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments",
+ side_effect=Exception("svc down")):
+ out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31")
+ assert len(out) == 8 # all posts retained despite enrichment failure
+
+ def test_only_top_n_enriched_by_depth(self):
+ posts = [_post(i, rel=1.0 - i / 100) for i in range(10)]
+ with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \
+ mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments",
+ return_value={"top_comments": [], "comment_insights": [],
+ "num_comments": None}) as fc:
+ reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31", depth="quick")
+ # quick depth enriches only top 3 posts
+ assert fc.call_count == reddit_keyless.ENRICH_LIMITS["quick"]
diff --git a/tests/test_reddit_listing.py b/tests/test_reddit_listing.py
new file mode 100644
index 0000000..fdb0e2a
--- /dev/null
+++ b/tests/test_reddit_listing.py
@@ -0,0 +1,85 @@
+"""Tests for scripts/lib/reddit_listing.py — keyless scored listing scrape."""
+
+from pathlib import Path
+from unittest import mock
+
+from lib import reddit_listing as rl
+
+FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "reddit_listing_cards_sample.html"
+
+
+def _html():
+ return FIXTURE.read_text(encoding="utf-8")
+
+
+class TestParseCards:
+ """parse_cards reads cards into scored post dicts."""
+
+ def test_parses_five_cards(self):
+ posts = rl.parse_cards(_html(), query="netherlands")
+ assert len(posts) == 5
+
+ def test_real_score_and_count(self):
+ posts = rl.parse_cards(_html())
+ top = posts[0]
+ assert top["score"] == 52692 # the real upvote count
+ assert top["engagement"]["score"] == 52692
+ assert top["num_comments"] == 1743
+ assert top["engagement"]["num_comments"] == 1743
+
+ def test_normalized_shape(self):
+ post = rl.parse_cards(_html())[0]
+ required = {"id", "title", "url", "score", "num_comments", "subreddit",
+ "created_utc", "author", "selftext", "date",
+ "engagement", "relevance", "why_relevant", "metadata"}
+ assert required.issubset(set(post.keys()))
+ assert post["why_relevant"] == "Reddit listing"
+ assert post["metadata"]["post_id"] # post id captured for backfill
+
+ def test_fields_populated(self):
+ post = rl.parse_cards(_html())[0]
+ assert post["title"]
+ assert post["author"] == "AdSpecialist6598"
+ assert post["subreddit"] == "technology"
+ assert "/comments/" in post["url"]
+ assert post["date"] and len(post["date"]) == 10
+
+ def test_empty_html_returns_empty(self):
+ assert rl.parse_cards("") == []
+ assert rl.parse_cards("
no cards
") == []
+
+
+class TestListingUrl:
+ def test_top_includes_timeframe(self):
+ u = rl._listing_url("technology", "top")
+ assert "community-more-posts/top/" in u and "name=technology" in u and "t=month" in u
+
+ def test_hot_no_timeframe(self):
+ u = rl._listing_url("r/technology", "hot")
+ assert "community-more-posts/hot/" in u and "name=technology" in u and "t=" not in u
+ assert ".json" not in u
+
+
+class TestFetchListings:
+ def test_dedupes_across_sorts(self):
+ with mock.patch.object(rl.http, "get_text", return_value=_html()):
+ posts = rl.fetch_listings(["technology"], depth="default")
+ urls = [p["url"] for p in posts]
+ assert len(urls) == len(set(urls)) # top + hot return same cards -> deduped
+
+ def test_no_subreddits_returns_empty(self):
+ assert rl.fetch_listings([], depth="default") == []
+
+ def test_all_fetches_fail_returns_empty(self):
+ with mock.patch.object(rl.http, "get_text", return_value=None):
+ assert rl.fetch_listings(["technology"]) == []
+
+
+class TestScoreIndex:
+ def test_builds_post_id_to_score_map(self):
+ with mock.patch.object(rl.http, "get_text", return_value=_html()):
+ idx = rl.score_index(["technology"], depth="quick")
+ assert idx # non-empty
+ first = next(iter(idx.values()))
+ assert set(first.keys()) == {"score", "num_comments"}
+ assert any(v["score"] == 52692 for v in idx.values())
diff --git a/tests/test_reddit_public.py b/tests/test_reddit_public.py
index d8574d5..10b9194 100644
--- a/tests/test_reddit_public.py
+++ b/tests/test_reddit_public.py
@@ -326,83 +326,24 @@ class TestMissingSubreddit:
assert results == []
# ---------------------------------------------------------------------------
-# Tests for comment enrichment (Unit 2)
+# search_reddit_public is now a thin shim over the keyless pipeline.
+# Full discovery + enrichment behavior is covered in test_reddit_keyless.py.
# ---------------------------------------------------------------------------
-class TestEnrichmentIntegration:
- """search_reddit_public enriches top posts with comments."""
+class TestSearchRedditPublicDelegatesToKeyless:
+ """search_reddit_public delegates to reddit_keyless.search_and_enrich."""
- @mock.patch("lib.reddit_public._enrich_post")
- @mock.patch("lib.reddit_public.urllib.request.urlopen")
- def test_search_enriches_top_5_by_default(self, mock_urlopen, mock_enrich):
- listing = _make_reddit_listing([
- {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/",
- "score": 100 - i, "created_utc": 1711670400}
- for i in range(10)
- ])
- mock_urlopen.return_value = _mock_urlopen_ok(listing)
- mock_enrich.side_effect = lambda item, timeout=10: item # pass-through
+ def test_delegates_with_all_args(self):
+ with mock.patch("lib.reddit_keyless.search_and_enrich") as mock_keyless:
+ mock_keyless.return_value = [{"id": "R1", "title": "x"}]
+ results = reddit_public.search_reddit_public(
+ "test", "2024-03-01", "2024-03-31",
+ depth="quick", subreddits=["ClaudeAI"],
+ )
- results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31")
-
- assert len(results) == 10
- # Default depth enriches top 5
- assert mock_enrich.call_count == 5
-
- @mock.patch("lib.reddit_public._enrich_post")
- @mock.patch("lib.reddit_public.urllib.request.urlopen")
- def test_enrichment_timeout_keeps_posts(self, mock_urlopen, mock_enrich):
- listing = _make_reddit_listing([
- {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/",
- "score": 100 - i, "created_utc": 1711670400}
- for i in range(10)
- ])
- mock_urlopen.return_value = _mock_urlopen_ok(listing)
-
- # Some enrichments raise, some succeed
- call_count = {"n": 0}
- def _side_effect(item, timeout=10):
- call_count["n"] += 1
- if call_count["n"] % 2 == 0:
- raise TimeoutError("enrichment timed out")
- return item
- mock_enrich.side_effect = _side_effect
-
- results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31")
-
- # All 10 posts should still be returned
- assert len(results) == 10
-
- @mock.patch("lib.reddit_public._enrich_post")
- @mock.patch("lib.reddit_public.urllib.request.urlopen")
- def test_all_enrichment_fails_all_posts_returned(self, mock_urlopen, mock_enrich):
- listing = _make_reddit_listing([
- {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/",
- "score": 100 - i, "created_utc": 1711670400}
- for i in range(10)
- ])
- mock_urlopen.return_value = _mock_urlopen_ok(listing)
- mock_enrich.side_effect = Exception("total failure")
-
- results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31")
-
- # All posts returned despite enrichment failure
- assert len(results) == 10
-
- @mock.patch("lib.reddit_public._enrich_post")
- @mock.patch("lib.reddit_public.urllib.request.urlopen")
- def test_quick_depth_enriches_top_3(self, mock_urlopen, mock_enrich):
- listing = _make_reddit_listing([
- {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/",
- "score": 100 - i, "created_utc": 1711670400}
- for i in range(10)
- ])
- mock_urlopen.return_value = _mock_urlopen_ok(listing)
- mock_enrich.side_effect = lambda item, timeout=10: item
-
- results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31", depth="quick")
-
- assert len(results) == 10
- # Quick depth enriches only top 3
- assert mock_enrich.call_count == 3
+ assert results == [{"id": "R1", "title": "x"}]
+ mock_keyless.assert_called_once_with(
+ "test", "2024-03-01", "2024-03-31",
+ depth="quick", subreddits=["ClaudeAI"],
+ )
diff --git a/tests/test_reddit_rss.py b/tests/test_reddit_rss.py
new file mode 100644
index 0000000..cb8377c
--- /dev/null
+++ b/tests/test_reddit_rss.py
@@ -0,0 +1,96 @@
+"""Tests for scripts/lib/reddit_rss.py — keyless Reddit RSS discovery."""
+
+from pathlib import Path
+from unittest import mock
+
+from lib import reddit_rss
+
+FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "reddit_search_rss_sample.xml"
+
+
+def _feed_text():
+ return FIXTURE.read_text(encoding="utf-8")
+
+
+class TestParseFeed:
+ """_parse_feed turns Atom entries into normalized post dicts."""
+
+ def test_parses_entries(self):
+ posts = reddit_rss._parse_feed(_feed_text(), query="lifelock")
+ assert len(posts) == 5
+ for p in posts:
+ assert p["title"]
+ assert "/comments/" in p["url"]
+ assert p["url"].startswith("https://www.reddit.com/")
+
+ def test_normalized_shape_matches_scrapecreators(self):
+ post = reddit_rss._parse_feed(_feed_text(), query="x")[0]
+ required = {"id", "title", "url", "score", "num_comments", "subreddit",
+ "created_utc", "author", "selftext", "date",
+ "engagement", "relevance", "why_relevant", "metadata"}
+ assert required.issubset(set(post.keys()))
+ assert set(post["engagement"].keys()) == {"score", "num_comments", "upvote_ratio"}
+ assert post["why_relevant"] == "Reddit RSS"
+
+ def test_score_is_placeholder_zero(self):
+ # RSS carries no engagement score; it is backfilled during enrichment.
+ for p in reddit_rss._parse_feed(_feed_text(), query="x"):
+ assert p["score"] == 0
+ assert p["engagement"]["score"] == 0
+
+ def test_subreddit_derivation(self):
+ post = reddit_rss._parse_feed(_feed_text(), query="x")[0]
+ assert post["subreddit"] == "Rakuten"
+
+ def test_date_parsed_to_iso(self):
+ post = reddit_rss._parse_feed(_feed_text(), query="x")[0]
+ assert post["date"] and len(post["date"]) == 10 # YYYY-MM-DD
+ assert isinstance(post["created_utc"], float)
+
+ def test_author_strips_u_prefix(self):
+ authors = [p["author"] for p in reddit_rss._parse_feed(_feed_text(), query="x")]
+ assert all(not a.startswith("/u/") and not a.startswith("u/") for a in authors)
+
+ def test_empty_and_malformed_feed_never_raises(self):
+ assert reddit_rss._parse_feed("", query="x") == []
+ assert reddit_rss._parse_feed("", query="x") == []
+
+ def test_entry_without_comments_link_skipped(self):
+ feed = (
+ ''
+ 'Subreddit itself'
+ ''
+ '2026-05-20T00:00:00+00:00'
+ )
+ assert reddit_rss._parse_feed(feed, query="x") == []
+
+
+class TestSearchRss:
+ """search_rss fans out, dedupes, assigns IDs, and honors depth limits."""
+
+ def test_dedupe_and_ids(self):
+ # Same feed returned for every URL -> deduped to 5 unique posts.
+ with mock.patch.object(reddit_rss.http, "get_text", return_value=_feed_text()):
+ posts = reddit_rss.search_rss("lifelock", depth="default",
+ subreddits=["Rakuten", "ConsumerAdvice"])
+ urls = [p["url"] for p in posts]
+ assert len(urls) == len(set(urls)) # no duplicates
+ assert [p["id"] for p in posts] == [f"R{i+1}" for i in range(len(posts))]
+
+ def test_depth_limit_quick(self):
+ with mock.patch.object(reddit_rss.http, "get_text", return_value=_feed_text()):
+ posts = reddit_rss.search_rss("lifelock", depth="quick")
+ assert len(posts) <= reddit_rss.DEPTH_LIMITS["quick"]
+
+ def test_all_feeds_fail_returns_empty(self):
+ with mock.patch.object(reddit_rss.http, "get_text", return_value=None):
+ posts = reddit_rss.search_rss("lifelock", subreddits=["Rakuten"])
+ assert posts == []
+
+ def test_builds_keyless_rss_urls(self):
+ urls = reddit_rss._build_urls("life lock", "default", ["Rakuten"])
+ assert any("search.rss?q=life+lock" in u and "/r/" not in u.split("?")[0] for u in urls)
+ assert any("/r/Rakuten/search.rss" in u and "restrict_sr=on" in u for u in urls)
+ assert any("/r/Rakuten/top.rss" in u for u in urls)
+ assert all(".json" not in u for u in urls) # never the dead endpoint
diff --git a/tests/test_reddit_shreddit.py b/tests/test_reddit_shreddit.py
new file mode 100644
index 0000000..3bd423e
--- /dev/null
+++ b/tests/test_reddit_shreddit.py
@@ -0,0 +1,103 @@
+"""Tests for scripts/lib/reddit_shreddit.py — keyless shreddit comment scrape."""
+
+from pathlib import Path
+from unittest import mock
+
+from lib import reddit_shreddit as rs
+
+FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "reddit_shreddit_comments_sample.html"
+
+
+def _html():
+ return FIXTURE.read_text(encoding="utf-8")
+
+
+class TestExtractPostRef:
+ def test_extracts_sub_and_id(self):
+ ref = rs.extract_post_ref("https://www.reddit.com/r/Rakuten/comments/1taeiw0/title/")
+ assert ref == ("Rakuten", "1taeiw0")
+
+ def test_non_thread_url_returns_none(self):
+ assert rs.extract_post_ref("https://www.reddit.com/r/Rakuten/") is None
+ assert rs.extract_post_ref("") is None
+
+ def test_svc_url_shape(self):
+ # sort=top guarantees the highest-scored comments land on page 1.
+ assert rs._svc_url("Rakuten", "1taeiw0") == (
+ "https://www.reddit.com/svc/shreddit/comments/r/Rakuten/t3_1taeiw0?sort=top"
+ )
+
+
+class TestParseComments:
+ """parse_comments reads elements into scored dicts."""
+
+ def test_happy_path(self):
+ comments = rs.parse_comments(_html())
+ assert len(comments) >= 1
+ for c in comments:
+ assert isinstance(c["score"], int)
+ assert c["author"] and c["author"] not in ("[deleted]", "[removed]")
+ assert c["body"]
+
+ def test_sorted_by_score_desc(self):
+ scores = [c["score"] for c in rs.parse_comments(_html())]
+ assert scores == sorted(scores, reverse=True)
+
+ def test_deleted_and_removed_filtered(self):
+ authors = [c["author"] for c in rs.parse_comments(_html())]
+ assert "[deleted]" not in authors and "[removed]" not in authors
+
+ def test_negative_score_retained(self):
+ scores = [c["score"] for c in rs.parse_comments(_html())]
+ assert -7 in scores # synthetic downvoted-but-real comment
+
+ def test_limit_honored(self):
+ assert len(rs.parse_comments(_html(), limit=2)) == 2
+
+ def test_body_text_extracted(self):
+ bodies = [c["body"] for c in rs.parse_comments(_html())]
+ assert any("$750" in b or "pending" in b for b in bodies)
+
+ def test_comment_url_built(self):
+ for c in rs.parse_comments(_html()):
+ if c["url"]:
+ assert c["url"].startswith("https://reddit.com/r/")
+
+ def test_empty_html_returns_empty(self):
+ assert rs.parse_comments("") == []
+ assert rs.parse_comments("no comments here") == []
+
+
+class TestTotalComments:
+ def test_reads_total(self):
+ assert rs._total_comments(_html()) == 14
+
+ def test_missing_returns_none(self):
+ assert rs._total_comments("") is None
+
+
+class TestFetchComments:
+ """fetch_comments wires URL -> svc fetch -> parse, never raising."""
+
+ def test_happy_path(self):
+ url = "https://www.reddit.com/r/Rakuten/comments/1taeiw0/title/"
+ with mock.patch.object(rs.http, "get_text", return_value=_html()) as m:
+ out = rs.fetch_comments(url)
+ # svc endpoint, not .json
+ assert "/svc/shreddit/comments/" in m.call_args[0][0]
+ assert ".json" not in m.call_args[0][0]
+ assert out["num_comments"] == 14
+ assert len(out["top_comments"]) >= 1
+ first = out["top_comments"][0]
+ assert {"score", "date", "author", "excerpt", "url"} <= set(first.keys())
+ assert isinstance(out["comment_insights"], list)
+
+ def test_bad_url_returns_empty(self):
+ out = rs.fetch_comments("https://www.reddit.com/r/Rakuten/")
+ assert out["top_comments"] == [] and out["num_comments"] is None
+
+ def test_fetch_failure_returns_empty(self):
+ url = "https://www.reddit.com/r/Rakuten/comments/1taeiw0/title/"
+ with mock.patch.object(rs.http, "get_text", return_value=None):
+ out = rs.fetch_comments(url)
+ assert out["top_comments"] == [] and out["num_comments"] is None