From 8d3a9e4368163b8a2aef17de0dc06c3e944bfe60 Mon Sep 17 00:00:00 2001 From: Matt Van Horn Date: Fri, 29 May 2026 14:43:56 -0500 Subject: [PATCH] fix(reddit): restore free path via keyless RSS + shreddit scrape (.json is dead) (#457) * test(reddit): add live RSS + shreddit comment fixtures Captured from reddit.com on 2026-05-29 (search.rss listing + the /svc/shreddit/comments partial), trimmed to a representative subset plus two synthetic edge cases (deleted author, negative score) for offline parser tests. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(http): add keyless get_text helper Browser-UA text fetch for RSS/HTML endpoints; returns None on any HTTP or network failure so tiered callers fall through cleanly. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(reddit): keyless RSS discovery (search.rss + listing feeds) Replaces the now-403 search.json with keyless Atom feeds, normalized to the existing reddit_public post shape. Scores are placeholder zeros, backfilled during shreddit enrichment. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(reddit): keyless shreddit comment scraper Parses elements from /svc/shreddit/comments/r/{sub}/t3_{id} (score/author/created/permalink + thingId-anchored body) into top comments, matching reddit_enrich output. Replaces the dead {thread}.json enrichment. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(reddit): tiered keyless orchestrator Tier 0 one-shot .json (residential bonus) -> Tier 1 RSS discovery -> Tier 2 shreddit enrichment. Returns [] never raises, so the SC backup still engages when every keyless tier is empty. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(reddit): route free path through keyless pipeline (.json is dead) search_reddit_public is now a thin shim over reddit_keyless, so pipeline.py and other callers need no change. Removes the dead .json enrichment helpers; search/_parse_posts remain as the demoted Tier 0 attempt. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(reddit): request sort=top so true top comments land on page 1 Guarantees the highest-scored comments are captured even on large threads, independent of Reddit's default comment sort. Local score re-sort remains. Co-Authored-By: Claude Opus 4.8 (1M context) * feat(reddit): recover post upvote scores via keyless listing partials The shreddit community-more-posts partial server-renders each post's score and comment count (works for normal users, not IP-gated), unlike RSS or the comments endpoint. Use it as a scored discovery source and to backfill scores onto RSS-discovered posts (subreddits derived from results when not provided). Ranking now uses real upvote score. Co-Authored-By: Claude Opus 4.8 (1M context) * fix(reddit): listings backfill scores only on bare queries, not discovery Caught running the full pipeline on a bare topic: deriving subreddits from noisy RSS results and merging their top/hot listings flooded results with high-upvote off-topic posts. Now derived-subreddit listings are used only to backfill scores onto keyword-matched RSS posts; listing cards are merged as discovery only when the caller explicitly provides subreddits (on-topic). Co-Authored-By: Claude Opus 4.8 (1M context) --------- Co-authored-by: Matt Van Horn <455140+mvanhorn@users.noreply.github.com> Co-authored-by: Claude Opus 4.8 (1M context) --- fixtures/reddit_listing_cards_sample.html | 8 + fixtures/reddit_search_rss_sample.xml | 7 + fixtures/reddit_shreddit_comments_sample.html | 29 +++ skills/last30days/scripts/lib/http.py | 47 ++++ .../last30days/scripts/lib/reddit_keyless.py | 214 +++++++++++++++++ .../last30days/scripts/lib/reddit_listing.py | 183 ++++++++++++++ .../last30days/scripts/lib/reddit_public.py | 166 ++----------- skills/last30days/scripts/lib/reddit_rss.py | 224 ++++++++++++++++++ .../last30days/scripts/lib/reddit_shreddit.py | 184 ++++++++++++++ tests/test_reddit_keyless.py | 168 +++++++++++++ tests/test_reddit_listing.py | 85 +++++++ tests/test_reddit_public.py | 91 ++----- tests/test_reddit_rss.py | 96 ++++++++ tests/test_reddit_shreddit.py | 103 ++++++++ 14 files changed, 1389 insertions(+), 216 deletions(-) create mode 100644 fixtures/reddit_listing_cards_sample.html create mode 100644 fixtures/reddit_search_rss_sample.xml create mode 100644 fixtures/reddit_shreddit_comments_sample.html create mode 100644 skills/last30days/scripts/lib/reddit_keyless.py create mode 100644 skills/last30days/scripts/lib/reddit_listing.py create mode 100644 skills/last30days/scripts/lib/reddit_rss.py create mode 100644 skills/last30days/scripts/lib/reddit_shreddit.py create mode 100644 tests/test_reddit_keyless.py create mode 100644 tests/test_reddit_listing.py create mode 100644 tests/test_reddit_rss.py create mode 100644 tests/test_reddit_shreddit.py diff --git a/fixtures/reddit_listing_cards_sample.html b/fixtures/reddit_listing_cards_sample.html new file mode 100644 index 0000000..0f176b3 --- /dev/null +++ b/fixtures/reddit_listing_cards_sample.html @@ -0,0 +1,8 @@ + +
+ + + + + +
diff --git a/fixtures/reddit_search_rss_sample.xml b/fixtures/reddit_search_rss_sample.xml new file mode 100644 index 0000000..982c9cd --- /dev/null +++ b/fixtures/reddit_search_rss_sample.xml @@ -0,0 +1,7 @@ + +2026-05-29T14:14:32+00:00https://www.redditstatic.com/icon.png//r/Rakuten/top.rss?t=monthThis is an unofficial subreddit for Rakuten Rewards, the cash back website. We are not affiliated with, endorsed by, or sponsored by Rakuten or any of its subsidiaries.top scoring links : Rakuten/u/InternetUser52https://www.reddit.com/user/InternetUser52<!-- SC_OFF --><div class="md"><p>I&#39;m rich!!</p> </div><!-- SC_ON --> &#32; submitted by &#32; <a href="https://www.reddit.com/user/InternetUser52"> /u/InternetUser52 </a> <br/> <span><a href="https://i.redd.it/q8fgmxs29c2h1.jpeg">[link]</a></span> &#32; <span><a href="https://www.reddit.com/r/Rakuten/comments/1tiv013/lets_goo_002/">[comments]</a></span>t3_1tiv0132026-05-20T18:48:31+00:002026-05-20T18:48:31+00:00LETS GOO! $0.02!!! +/u/Immediate-Duck-6351https://www.reddit.com/user/Immediate-Duck-6351<!-- SC_OFF --><div class="md"><p>I don’t travel and I’m buying a house in a few weeks so cash back is amazing 🙌 hoping to keep the pace in the next quarter so I can buy new kitchen appliances lol. </p> </div><!-- SC_ON --> &#32; submitted by &#32; <a href="https://www.reddit.com/user/Immediate-Duck-6351"> /u/Immediate-Duck-6351 </a> <br/> <span><a href="https://i.redd.it/d2a4s0ipvb1h1.jpeg">[link]</a></span> &#32; <span><a href="https://www.reddit.com/r/Rakuten/comments/1te1fp8/so_excited/">[comments]</a></span>t3_1te1fp82026-05-15T16:29:28+00:002026-05-15T16:29:28+00:00So excited 🥳 +/u/gnibgnibhttps://www.reddit.com/user/gnibgnib<!-- SC_OFF --><div class="md"><p>128k for the May transfer</p> <p>41k pending for August </p> <p>Got another 9k at Asics not showing but overall pretty happy with Rakuten</p> <p>P2 was able to secure 85k for May transfer</p> </div><!-- SC_ON --> &#32; submitted by &#32; <a href="https://www.reddit.com/user/gnibgnib"> /u/gnibgnib </a> <br/> <span><a href="https://www.reddit.com/gallery/1tb8674">[link]</a></span> &#32; <span><a href="https://www.reddit.com/r/Rakuten/comments/1tb8674/had_a_great_run_so_far_this_year_thanks_to_this/">[comments]</a></span>t3_1tb86742026-05-12T17:17:19+00:002026-05-12T17:17:19+00:00Had a great run so far this year thanks to this sub! +/u/TravelVet93https://www.reddit.com/user/TravelVet93&#32; submitted by &#32; <a href="https://www.reddit.com/user/TravelVet93"> /u/TravelVet93 </a> <br/> <span><a href="https://i.redd.it/x6b9whvupb1h1.jpeg">[link]</a></span> &#32; <span><a href="https://www.reddit.com/r/Rakuten/comments/1te0hom/my_best_payout_so_far/">[comments]</a></span>t3_1te0hom2026-05-15T15:56:40+00:002026-05-15T15:56:40+00:00My best payout so far +/u/Beautiful-Piece-4252https://www.reddit.com/user/Beautiful-Piece-4252<!-- SC_OFF --><div class="md"><p>The amount of $$ available in sign up bonuses is amazing. It&#39;s kind of a part time job ensuring Rakuten captures everything, but my August and November payout should be sizeable. I&#39;m new to this and it always seemed like a lot of work for little reward. I know it&#39;s not sustainable, but wow!</p> </div><!-- SC_ON --> &#32; submitted by &#32; <a href="https://www.reddit.com/user/Beautiful-Piece-4252"> /u/Beautiful-Piece-4252 </a> <br/> <span><a href="https://i.redd.it/1vqvajsci42h1.jpeg">[link]</a></span> &#32; <span><a href="https://www.reddit.com/r/Rakuten/comments/1thsnm1/how_can_this_be_real/">[comments]</a></span>t3_1thsnm12026-05-19T16:46:17+00:002026-05-19T16:46:17+00:00How can this be real? + diff --git a/fixtures/reddit_shreddit_comments_sample.html b/fixtures/reddit_shreddit_comments_sample.html new file mode 100644 index 0000000..97308e6 --- /dev/null +++ b/fixtures/reddit_shreddit_comments_sample.html @@ -0,0 +1,29 @@ + + + + +

Where do you find $750? The highest available package for Total was $284.99 when I did the lifelock promotion. I did get the full 284.99 from Rakuten.

+
+ +

It went to pending ($712.49)

+
+ +

Hey I PM’d. can I get the screenshot ?

+
+ +

Family plan

+
+ +

Price seems to change every time I go to the page but I see only 249.99-369.99 for Total/Advanced. No where near your $750. Just saying the Total plan for 299.99 worked for me and I got 284.99 which is 95%.

+
+ +

I did that one. Let’s pray

+
+ +

[removed]

+
+ +

A downvoted but real reply with negative score for edge-case coverage.

+
+
diff --git a/skills/last30days/scripts/lib/http.py b/skills/last30days/scripts/lib/http.py index 8afd464..a8d694f 100644 --- a/skills/last30days/scripts/lib/http.py +++ b/skills/last30days/scripts/lib/http.py @@ -223,6 +223,53 @@ def post_raw(url: str, json_data: Dict[str, Any], headers: Optional[Dict[str, st return request("POST", url, headers=headers, json_data=json_data, raw=True, **kwargs) +BROWSER_USER_AGENT = ( + "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) " + "AppleWebKit/537.36 (KHTML, like Gecko) " + "Chrome/124.0.0.0 Safari/537.36" +) + + +def get_text( + url: str, + timeout: int = DEFAULT_TIMEOUT, + retries: int = 2, + accept: str = "*/*", + headers: Optional[Dict[str, str]] = None, +) -> Optional[str]: + """Fetch a URL and return decoded text, or None on any failure. + + Keyless helper for Reddit RSS and shreddit HTML endpoints — the free path + that replaced the now-403 ``.json`` endpoints. Sends a browser User-Agent + and never raises: returns None on HTTP error, network failure, or timeout + so tiered callers can fall through to the next source. + + Args: + url: Request URL + timeout: HTTP timeout per attempt in seconds + retries: Number of retries on failure (kept low — these tiers fail fast) + accept: Accept header value (e.g. "application/atom+xml", "text/html") + headers: Optional extra headers merged over the defaults + + Returns: + Decoded response body as text, or None on failure. + """ + merged = { + "User-Agent": BROWSER_USER_AGENT, + "Accept": accept, + "Accept-Language": "en-US,en;q=0.9", + } + if headers: + merged.update(headers) + try: + return request( + "GET", url, headers=merged, timeout=timeout, retries=retries, raw=True + ) + except HTTPError as e: + log(f"get_text failed ({e}): {url}") + return None + + def scrapecreators_headers(token: str) -> Dict[str, str]: """Build ScrapeCreators request headers (x-api-key + JSON content type).""" return { diff --git a/skills/last30days/scripts/lib/reddit_keyless.py b/skills/last30days/scripts/lib/reddit_keyless.py new file mode 100644 index 0000000..b1ced97 --- /dev/null +++ b/skills/last30days/scripts/lib/reddit_keyless.py @@ -0,0 +1,214 @@ +"""Keyless Reddit pipeline: tiered free search + comment enrichment. + +Replaces the dead ``.json`` free path. Discovery tiers, cheapest/most-likely +first; enrichment then runs on whatever was discovered: + + Tier 0 one-shot legacy ``.json`` search — demoted. Datacenter IPs get 403, + but a residential machine (where the skill usually runs) may still + get 200, so it is worth one cheap try. Honors the "brute-force .json" + intent without depending on it. + Tier 1 RSS discovery (reddit_rss) — keyless, robust, the load-bearing path. + Tier 2 shreddit comment + count enrichment (reddit_shreddit) for top posts. + +Returns ``[]`` (never raises) so ``pipeline.py`` can fall through to the +ScrapeCreators backup when every keyless tier comes up empty. +""" + +import concurrent.futures +import sys +from concurrent.futures import ThreadPoolExecutor +from typing import Any, Dict, List, Optional + +from collections import Counter + +from . import reddit_rss, reddit_shreddit, reddit_listing + +ENRICH_LIMITS = reddit_shreddit.ENRICH_LIMITS +ENRICH_BUDGET = 45 # seconds total across all enrichment threads +MAX_ENRICH_WORKERS = 4 +MAX_DERIVED_SUBS = 5 # subreddits derived from RSS results for score backfill + + +def _log(msg: str) -> None: + sys.stderr.write(f"[RedditKeyless] {msg}\n") + sys.stderr.flush() + + +def _tier0_json(topic: str, depth: str) -> List[Dict[str, Any]]: + """One cheap global ``.json`` discovery attempt. Returns [] on the 403 wall.""" + try: + from . import reddit_public + return reddit_public.search(topic, depth=depth) or [] + except Exception as e: # never let the demoted tier sink the run + _log(f"Tier 0 (.json) unavailable: {e}") + return [] + + +def _top_subreddits(posts: List[Dict[str, Any]], limit: int = MAX_DERIVED_SUBS) -> List[str]: + """Most frequent subreddits across discovered posts (for score backfill).""" + counts = Counter(p.get("subreddit", "") for p in posts if p.get("subreddit")) + return [sub for sub, _ in counts.most_common(limit)] + + +def _apply_scores(post: Dict[str, Any], scored: Dict[str, int]) -> None: + post["score"] = scored["score"] + post["num_comments"] = scored["num_comments"] + post.setdefault("engagement", {})["score"] = scored["score"] + post["engagement"]["num_comments"] = scored["num_comments"] + + +def _discover(topic: str, depth: str, subreddits: Optional[List[str]]) -> List[Dict[str, Any]]: + # Tier 0: demoted one-shot .json (dead for normal users too, but free to try). + posts = _tier0_json(topic, depth) + if posts: + _log(f"Tier 0 (.json) returned {len(posts)} posts") + return posts + + # Tier 1: keyless discovery. RSS gives breadth (incl. global keyword search); + # the listing partials give real upvote scores. + rss_posts = reddit_rss.search_rss(topic, depth=depth, subreddits=subreddits) + + if subreddits: + # Targeted run: the caller chose these subreddits, so their listing cards + # are on-topic — include them as scored discovery AND as a score source. + listing_posts = reddit_listing.fetch_listings(subreddits, depth=depth, query=topic) + score_source = listing_posts + else: + # Bare global run: subreddits derived from noisy RSS results are NOT + # reliably on-topic, so their listings are used ONLY to backfill scores + # onto the keyword-matched RSS posts — never merged as discovery, which + # would flood results with high-upvote but irrelevant posts. + listing_posts = [] + derived = _top_subreddits(rss_posts) + score_source = reddit_listing.fetch_listings(derived, depth=depth, query=topic) + _log( + f"Tier 1 (RSS) {len(rss_posts)} posts; " + f"{'listing discovery ' + str(len(listing_posts)) if subreddits else 'score-only'}; " + f"{len(score_source)} scored cards" + ) + + # Score lookup by post id, from the scored listing cards. + score_map: Dict[str, Dict[str, int]] = {} + for p in score_source: + pid = p.get("metadata", {}).get("post_id", "") + if pid: + score_map[pid] = {"score": p["score"], "num_comments": p["num_comments"]} + + # Merge: scored listing posts first (targeted only), then RSS breadth, + # backfilled with real scores where the post appears in a listing. + merged: List[Dict[str, Any]] = [] + seen: set = set() + for p in listing_posts: + if p["url"] not in seen: + seen.add(p["url"]) + merged.append(p) + for p in rss_posts: + if p["url"] in seen: + continue + pid = reddit_listing._post_id(p["url"]) + if pid in score_map: + _apply_scores(p, score_map[pid]) + seen.add(p["url"]) + merged.append(p) + return merged + + +def _enrich_one(post: Dict[str, Any]) -> Dict[str, Any]: + """Attach shreddit comments + real comment count. Never raises.""" + try: + data = reddit_shreddit.fetch_comments(post.get("url", "")) + if data.get("top_comments"): + post["top_comments"] = data["top_comments"] + if data.get("comment_insights"): + post["comment_insights"] = data["comment_insights"] + num = data.get("num_comments") + if num is not None: + post["num_comments"] = num + post.setdefault("engagement", {})["num_comments"] = num + except Exception: + pass # keep the post with whatever discovery gave us + return post + + +def _enrich(posts: List[Dict[str, Any]], depth: str) -> List[Dict[str, Any]]: + """Enrich the top N posts with comments under a total time budget.""" + limit = ENRICH_LIMITS.get(depth, ENRICH_LIMITS["default"]) + to_enrich = posts[:limit] + rest = posts[limit:] + if not to_enrich: + return posts + + result_map: Dict[int, Dict[str, Any]] = {} + try: + with ThreadPoolExecutor(max_workers=min(limit, MAX_ENRICH_WORKERS)) as executor: + futures = { + executor.submit(_enrich_one, post): i + for i, post in enumerate(to_enrich) + } + done, not_done = concurrent.futures.wait(futures, timeout=ENRICH_BUDGET) + for future in done: + idx = futures[future] + try: + result_map[idx] = future.result(timeout=0) + except Exception: + result_map[idx] = to_enrich[idx] + for future in not_done: + idx = futures[future] + result_map[idx] = to_enrich[idx] + future.cancel() + enriched = [result_map[i] for i in range(len(to_enrich))] + except Exception: + enriched = to_enrich + + return enriched + rest + + +def search_and_enrich( + topic: str, + from_date: str, + to_date: str, + depth: str = "default", + subreddits: Optional[List[str]] = None, +) -> List[Dict[str, Any]]: + """Full keyless Reddit pipeline: discover (Tier 0/1) then enrich (Tier 2). + + Args: + topic: Search topic + from_date: Start date (YYYY-MM-DD) + to_date: End date (YYYY-MM-DD) + depth: 'quick', 'default', or 'deep' + subreddits: Optional pre-resolved subreddit names (without r/) + + Returns: + List of normalized item dicts matching the reddit_public output shape, + with top_comments/comment_insights attached on enriched posts. + Empty list when all keyless tiers fail (so SC backup can engage). + """ + posts = _discover(topic, depth, subreddits) + if not posts: + return [] + + # Date filter: keep posts in range or with unknown dates (mirrors reddit_public). + posts = [ + p for p in posts + if p.get("date") is None or (from_date <= p["date"] <= to_date) + ] + + # Rank before enrichment by real upvote score (from listing cards / backfill), + # then query relevance, then recency. Posts without a recovered score sort by + # the latter two — same behavior as before scores were available. + posts.sort( + key=lambda p: ( + p.get("engagement", {}).get("score", 0) or 0, + p.get("relevance", 0) or 0, + p.get("date") or "", + ), + reverse=True, + ) + + posts = _enrich(posts, depth) + + for i, post in enumerate(posts): + post["id"] = f"R{i + 1}" + + return posts diff --git a/skills/last30days/scripts/lib/reddit_listing.py b/skills/last30days/scripts/lib/reddit_listing.py new file mode 100644 index 0000000..7f2fccc --- /dev/null +++ b/skills/last30days/scripts/lib/reddit_listing.py @@ -0,0 +1,183 @@ +"""Keyless Reddit listing scrape via shreddit /svc partials — with real scores. + +The subreddit listing partial +``/svc/shreddit/community-more-posts/{sort}/?name={sub}[&t={range}]`` serves +HTTP 200 with no API key and **server-renders each post's upvote score**, which +neither RSS nor the comments endpoint provides. Each post is a +```` element whose start-tag attributes carry ``score``, +``comment-count``, ``post-title``, ``permalink``, ``author``, ``subreddit-name`` +and ``created-timestamp``. + +This is the keyless source of post-level upvotes. It works for normal users on +ordinary connections (verified), so reddit_keyless uses it both as a scored +discovery source and to backfill scores onto RSS-discovered posts. +""" + +import html as _html +import re +import sys +from datetime import datetime, timezone +from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeoutError +from typing import Any, Dict, List, Optional + +from . import http +from .relevance import token_overlap_relevance + +# Listing sorts pulled per subreddit, by depth. +LISTING_SORTS = { + "quick": ["top"], + "default": ["top", "hot"], + "deep": ["top", "hot", "new"], +} +DEPTH_LIMITS = {"quick": 10, "default": 25, "deep": 50} +TIMEFRAME = "month" +MAX_WORKERS = 4 +LISTING_TIMEOUT = 15 + +_POST_CARD = re.compile(r"])[^>]*>") + + +def _log(msg: str) -> None: + sys.stderr.write(f"[RedditListing] {msg}\n") + sys.stderr.flush() + + +def _attr(tag: str, name: str) -> Optional[str]: + m = re.search(rf'\b{name}="([^"]*)"', tag) + return _html.unescape(m.group(1)) if m else None + + +def _to_date(value: Optional[str]) -> Optional[str]: + if not value: + return None + try: + return datetime.fromisoformat(value.strip()).date().isoformat() + except (ValueError, TypeError): + return None + + +def _to_epoch(value: Optional[str]) -> Optional[float]: + if not value: + return None + try: + dt = datetime.fromisoformat(value.strip()) + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + return dt.timestamp() + except (ValueError, TypeError): + return None + + +def _post_id(permalink: str) -> str: + m = re.search(r"/comments/([A-Za-z0-9]+)", permalink or "") + return m.group(1) if m else "" + + +def parse_cards(html_text: str, query: str = "") -> List[Dict[str, Any]]: + """Parse cards into normalized post dicts with real scores.""" + posts: List[Dict[str, Any]] = [] + for m in _POST_CARD.finditer(html_text or ""): + tag = m.group(0) + permalink = _attr(tag, "permalink") or "" + if "/comments/" not in permalink: + continue + try: + score = int(_attr(tag, "score") or 0) + except ValueError: + score = 0 + try: + num_comments = int(_attr(tag, "comment-count") or 0) + except ValueError: + num_comments = 0 + title = _attr(tag, "post-title") or "" + author = _attr(tag, "author") or "[deleted]" + subreddit = _attr(tag, "subreddit-name") or "" + created = _attr(tag, "created-timestamp") + url = f"https://www.reddit.com{permalink}" + + posts.append({ + "id": "", + "title": title, + "url": url, + "score": score, + "num_comments": num_comments, + "subreddit": subreddit, + "created_utc": _to_epoch(created), + "author": author if author not in ("[deleted]", "[removed]") else "[deleted]", + "selftext": "", + "date": _to_date(created), + "engagement": { + "score": score, + "num_comments": num_comments, + "upvote_ratio": None, + }, + "relevance": round(token_overlap_relevance(query, title), 3) if query else 0.0, + "why_relevant": "Reddit listing", + "metadata": {"post_id": _post_id(permalink)}, + }) + return posts + + +def _listing_url(subreddit: str, sort: str) -> str: + sub = subreddit.removeprefix("r/").strip() + url = f"https://www.reddit.com/svc/shreddit/community-more-posts/{sort}/?name={sub}" + if sort == "top": + url += f"&t={TIMEFRAME}" + return url + + +def _fetch_one(subreddit: str, sort: str, query: str) -> List[Dict[str, Any]]: + try: + text = http.get_text(_listing_url(subreddit, sort), timeout=LISTING_TIMEOUT, + accept="text/html") + return parse_cards(text, query) if text else [] + except Exception as e: + _log(f"listing fetch failed r/{subreddit} {sort}: {e}") + return [] + + +def fetch_listings( + subreddits: List[str], + depth: str = "default", + query: str = "", +) -> List[Dict[str, Any]]: + """Fetch scored post cards across subreddits × depth-appropriate sorts. + + Returns deduped normalized posts (with real scores), unranked/unsliced — + the caller merges these with other sources, ranks, and slices. + """ + if not subreddits: + return [] + sorts = LISTING_SORTS.get(depth, LISTING_SORTS["default"]) + jobs = [(sub, sort) for sub in subreddits for sort in sorts] + all_posts: List[Dict[str, Any]] = [] + with ThreadPoolExecutor(max_workers=min(MAX_WORKERS, len(jobs)) or 1) as executor: + futures = {executor.submit(_fetch_one, sub, sort, query): (sub, sort) + for sub, sort in jobs} + for future in futures: + try: + all_posts.extend(future.result(timeout=LISTING_TIMEOUT + 5)) + except (Exception, FuturesTimeoutError) as e: + _log(f"listing future failed: {e}") + + seen: set = set() + unique: List[Dict[str, Any]] = [] + for p in all_posts: + if p["url"] not in seen: + seen.add(p["url"]) + unique.append(p) + return unique + + +def score_index(subreddits: List[str], depth: str = "default") -> Dict[str, Dict[str, int]]: + """Build a {post_id: {score, num_comments}} map from subreddit listings. + + Used to backfill real scores onto posts discovered via RSS, which carries + no engagement numbers. + """ + index: Dict[str, Dict[str, int]] = {} + for p in fetch_listings(subreddits, depth=depth): + pid = p.get("metadata", {}).get("post_id") or _post_id(p["url"]) + if pid: + index[pid] = {"score": p["score"], "num_comments": p["num_comments"]} + return index diff --git a/skills/last30days/scripts/lib/reddit_public.py b/skills/last30days/scripts/lib/reddit_public.py index e8f3dc8..b6789b0 100644 --- a/skills/last30days/scripts/lib/reddit_public.py +++ b/skills/last30days/scripts/lib/reddit_public.py @@ -1,9 +1,16 @@ -"""Standalone Reddit public JSON search module. +"""Reddit public ``.json`` search module (demoted to keyless Tier 0). -Searches Reddit using the free public JSON endpoints (no API key required). -Promoted from last-resort fallback to robust primary free path. +Reddit's public ``.json`` endpoints now return HTTP 403 from most contexts +(shreddit anti-bot), so this is no longer the primary free path. The keyless +pipeline (see reddit_keyless.py) still calls ``search`` as a cheap one-shot +Tier 0 attempt — a residential machine may occasionally get a 200 — before +falling through to RSS discovery (reddit_rss.py) and shreddit comment +enrichment (reddit_shreddit.py). -Endpoints: +``search_reddit_public`` is retained as a compatibility shim that delegates to +the keyless pipeline, so existing callers (pipeline.py) need no change. + +Endpoints (Tier 0): - Global: https://www.reddit.com/search.json?q={query}&sort=relevance&t=month&limit={limit} - Subreddit: https://www.reddit.com/r/{sub}/search.json?q={query}&restrict_sr=on&sort=relevance&t=month @@ -18,7 +25,6 @@ import time import urllib.error import urllib.parse import urllib.request -from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeoutError from typing import Any, Dict, List, Optional @@ -35,13 +41,6 @@ DEPTH_LIMITS = { "deep": 50, } -# How many top posts to enrich with comments, by depth -ENRICH_LIMITS = { - "quick": 3, - "default": 5, - "deep": 8, -} - MAX_RETRIES = 3 BASE_BACKOFF = 2.0 # seconds @@ -237,78 +236,6 @@ def search( return unique[:limit] -def _enrich_post(item: Dict[str, Any], timeout: int = 10) -> Dict[str, Any]: - """Enrich a single post with top comments. Never raises.""" - try: - from . import reddit_enrich - thread_data = reddit_enrich.fetch_thread_data(item["url"], timeout=timeout) - if not thread_data: - return item - parsed = reddit_enrich.parse_thread_data(thread_data) - comments = parsed.get("comments", []) - top = reddit_enrich.get_top_comments(comments) - item["top_comments"] = [ - { - "score": c.get("score", 0), - "excerpt": (c.get("body") or "")[:200], - "author": c.get("author", ""), - } - for c in top[:10] - ] - except Exception: - # Never discard — keep post with empty metadata - pass - return item - - -def _enrich_posts(posts: List[Dict[str, Any]], depth: str = "default") -> List[Dict[str, Any]]: - """Enrich top N posts with comment data using threads. Total budget 45s.""" - limit = ENRICH_LIMITS.get(depth, ENRICH_LIMITS["default"]) - to_enrich = posts[:limit] - rest = posts[limit:] - - if not to_enrich: - return posts - - enriched = [] - try: - with ThreadPoolExecutor(max_workers=min(limit, 4)) as executor: - futures = { - executor.submit(_enrich_post, post, 10): i - for i, post in enumerate(to_enrich) - } - # Collect results with 45s total budget - import concurrent.futures - done, not_done = concurrent.futures.wait(futures, timeout=45) - # Build result list preserving order - result_map: Dict[int, Dict[str, Any]] = {} - for future in done: - idx = futures[future] - try: - result_map[idx] = future.result(timeout=0) - except Exception: - result_map[idx] = to_enrich[idx] - # Any not-done futures: keep original post - for future in not_done: - idx = futures[future] - result_map[idx] = to_enrich[idx] - future.cancel() - enriched = [result_map[i] for i in range(len(to_enrich))] - except Exception: - enriched = to_enrich - - return enriched + rest - - -def _search_subreddit(sub: str, topic: str, depth: str, timeout: int = 15) -> List[Dict[str, Any]]: - """Search a single subreddit. Never raises.""" - try: - return search(topic, depth=depth, subreddit=sub, timeout=timeout) - except Exception as e: - _log(f"Subreddit search failed for r/{sub}: {e}") - return [] - - def search_reddit_public( topic: str, from_date: str, @@ -316,12 +243,17 @@ def search_reddit_public( depth: str = "default", subreddits: Optional[List[str]] = None, ) -> List[Dict[str, Any]]: - """High-level Reddit public search matching the openai_reddit interface. + """High-level free Reddit search + enrichment (keyless). - When subreddits are provided (from agent planning), searches each targeted - sub first, then does global search, and deduplicates across both. This - mirrors the SC search_and_enrich() flow where pre-resolved subreddits get - priority. + Thin compatibility shim over the tiered keyless pipeline: the legacy + ``.json`` search/enrichment endpoints now return HTTP 403, so this delegates + to ``reddit_keyless.search_and_enrich`` (Tier 0 one-shot ``.json`` → + Tier 1 RSS discovery → Tier 2 shreddit comment enrichment). The name and + signature are preserved so ``pipeline.py`` and other callers need no change + and the ScrapeCreators backup still engages when this returns empty. + + The module-level ``search`` / ``_parse_posts`` helpers remain in use as the + keyless pipeline's demoted Tier 0 ``.json`` attempt. Args: topic: Search topic @@ -332,57 +264,9 @@ def search_reddit_public( Returns: List of normalized item dicts matching ScrapeCreators output format. + Empty list on total failure (so SC backup can engage). """ - all_posts: List[Dict[str, Any]] = [] - - # Phase 1: Search targeted subreddits in parallel (if provided) - if subreddits: - _log(f"Searching {len(subreddits)} targeted subreddits: {subreddits}") - workers = min(4, len(subreddits)) - with ThreadPoolExecutor(max_workers=workers) as executor: - futures = { - executor.submit(_search_subreddit, sub, topic, depth): sub - for sub in subreddits - } - for future in futures: - sub = futures[future] - try: - sub_posts = future.result(timeout=30) - _log(f" -> {len(sub_posts)} results from r/{sub}") - all_posts.extend(sub_posts) - except (Exception, FuturesTimeoutError) as e: - _log(f" -> r/{sub} failed: {e}") - - # Phase 2: Global search - global_posts = search(topic, depth=depth) - all_posts.extend(global_posts) - - # Deduplicate by URL (targeted results keep priority since they come first) - seen_urls: set = set() - results: List[Dict[str, Any]] = [] - for post in all_posts: - if post["url"] not in seen_urls: - seen_urls.add(post["url"]) - results.append(post) - - # Date filter: keep posts in range or with unknown dates - filtered = [] - for item in results: - d = item.get("date") - if d is None or (from_date <= d <= to_date): - filtered.append(item) - - # Sort by engagement (score desc) - filtered.sort( - key=lambda x: x.get("engagement", {}).get("score", 0), - reverse=True, + from . import reddit_keyless + return reddit_keyless.search_and_enrich( + topic, from_date, to_date, depth=depth, subreddits=subreddits ) - - # Enrich top posts with comments - filtered = _enrich_posts(filtered, depth=depth) - - # Re-index IDs - for i, item in enumerate(filtered): - item["id"] = f"R{i + 1}" - - return filtered diff --git a/skills/last30days/scripts/lib/reddit_rss.py b/skills/last30days/scripts/lib/reddit_rss.py new file mode 100644 index 0000000..b7c3576 --- /dev/null +++ b/skills/last30days/scripts/lib/reddit_rss.py @@ -0,0 +1,224 @@ +"""Keyless Reddit discovery via public RSS/Atom feeds. + +Reddit's ``.json`` search endpoints now return HTTP 403 (shreddit anti-bot). +RSS feeds still serve HTTP 200 with no API key, so this module uses them for +post discovery, replacing ``reddit_public.search`` as the free search path. + +Two feed families are combined and deduped: +- search: /search.rss?q=... and /r/{sub}/search.rss?q=...&restrict_sr=on +- listing: /r/{sub}/{top,hot}.rss?t=month + +RSS entries carry no engagement score, so ``score``/``num_comments`` start at 0 +and are backfilled during shreddit enrichment (see reddit_shreddit.py). Output +dicts match the normalized shape emitted by ``reddit_public._parse_posts`` so +downstream code (pipeline, renderer) is unaffected. +""" + +import sys +import xml.etree.ElementTree as ET +from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeoutError +from datetime import datetime, timezone +from typing import Any, Dict, List, Optional +from urllib.parse import quote_plus + +from . import http +from .relevance import token_overlap_relevance + +ATOM = "{http://www.w3.org/2005/Atom}" + +# Mirror reddit_public depth-aware limits so the two free paths behave alike. +DEPTH_LIMITS = { + "quick": 10, + "default": 25, + "deep": 50, +} + +# Listing sorts pulled per subreddit (in addition to search), for volume. +LISTING_SORTS = { + "quick": ["top"], + "default": ["top", "hot"], + "deep": ["top", "hot", "new"], +} + +MAX_WORKERS = 4 +FEED_TIMEOUT = 15 + + +def _log(msg: str) -> None: + sys.stderr.write(f"[RedditRSS] {msg}\n") + sys.stderr.flush() + + +def _iso_to_date(value: Optional[str]) -> Optional[str]: + """Parse an ISO-8601 timestamp (e.g. 2026-05-20T18:48:31+00:00) to YYYY-MM-DD.""" + if not value: + return None + try: + dt = datetime.fromisoformat(value.strip()) + return dt.date().isoformat() + except (ValueError, TypeError): + return None + + +def _iso_to_epoch(value: Optional[str]) -> Optional[float]: + if not value: + return None + try: + dt = datetime.fromisoformat(value.strip()) + if dt.tzinfo is None: + dt = dt.replace(tzinfo=timezone.utc) + return dt.timestamp() + except (ValueError, TypeError): + return None + + +def _subreddit_from(category: str, url: str) -> str: + """Derive subreddit name from the entry category or, failing that, the URL.""" + if category: + return category + # URL form: https://www.reddit.com/r/{sub}/comments/{id}/... + parts = url.split("/r/", 1) + if len(parts) == 2: + return parts[1].split("/", 1)[0] + return "" + + +def _parse_feed(xml_text: str, query: str = "") -> List[Dict[str, Any]]: + """Parse an Atom feed string into normalized post dicts. Never raises.""" + if not xml_text: + return [] + try: + root = ET.fromstring(xml_text) + except ET.ParseError as e: + _log(f"feed parse error: {e}") + return [] + + posts: List[Dict[str, Any]] = [] + for entry in root.iter(f"{ATOM}entry"): + link_el = entry.find(f"{ATOM}link") + url = link_el.get("href", "").strip() if link_el is not None else "" + if not url or "/comments/" not in url: + continue + + title_el = entry.find(f"{ATOM}title") + title = (title_el.text or "").strip() if title_el is not None else "" + + author = "" + author_el = entry.find(f"{ATOM}author/{ATOM}name") + if author_el is not None and author_el.text: + author = author_el.text.strip().removeprefix("/u/").removeprefix("u/") + if author in ("[deleted]", "[removed]", ""): + author = "[deleted]" + + cat_el = entry.find(f"{ATOM}category") + category = cat_el.get("term", "").strip() if cat_el is not None else "" + subreddit = _subreddit_from(category, url) + + updated_el = entry.find(f"{ATOM}updated") + updated = (updated_el.text or "").strip() if updated_el is not None else "" + + content_el = entry.find(f"{ATOM}content") + selftext = "" + if content_el is not None and content_el.text: + # Strip the simplest HTML; renderer only needs an excerpt. + import re as _re + selftext = _re.sub(r"<[^>]+>", " ", content_el.text) + selftext = _re.sub(r"\s+", " ", selftext).strip()[:500] + + relevance = round(token_overlap_relevance(query, title), 3) if query else 0.0 + + posts.append({ + "id": "", # assigned after dedup + "title": title, + "url": url, + "score": 0, # backfilled by shreddit enrichment + "num_comments": 0, # backfilled by shreddit enrichment + "subreddit": subreddit, + "created_utc": _iso_to_epoch(updated), + "author": author, + "selftext": selftext, + "date": _iso_to_date(updated), + "engagement": { + "score": 0, + "num_comments": 0, + "upvote_ratio": None, + }, + "relevance": relevance, + "why_relevant": "Reddit RSS", + "metadata": {}, + }) + + return posts + + +def _build_urls(query: str, depth: str, subreddits: Optional[List[str]]) -> List[str]: + """Build the keyless RSS feed URLs to fan out across.""" + q = quote_plus(query) + urls: List[str] = [ + f"https://www.reddit.com/search.rss?q={q}&sort=relevance&t=month" + ] + for raw_sub in (subreddits or []): + sub = raw_sub.removeprefix("r/").strip() + if not sub: + continue + urls.append( + f"https://www.reddit.com/r/{sub}/search.rss" + f"?q={q}&restrict_sr=on&sort=relevance&t=month" + ) + for sort in LISTING_SORTS.get(depth, LISTING_SORTS["default"]): + urls.append(f"https://www.reddit.com/r/{sub}/{sort}.rss?t=month") + return urls + + +def _fetch_feed(url: str, query: str) -> List[Dict[str, Any]]: + """Fetch and parse one feed. Never raises.""" + try: + text = http.get_text(url, timeout=FEED_TIMEOUT, accept="application/atom+xml") + return _parse_feed(text, query) if text else [] + except Exception as e: # defensive: a single bad feed must not sink the run + _log(f"feed fetch failed for {url}: {e}") + return [] + + +def search_rss( + query: str, + depth: str = "default", + subreddits: Optional[List[str]] = None, +) -> List[Dict[str, Any]]: + """Discover Reddit posts for a query via keyless RSS feeds. + + Args: + query: Search query string + depth: 'quick', 'default', or 'deep' — controls result limit and feeds + subreddits: Optional pre-resolved subreddit names (without r/) to target + + Returns: + List of normalized post dicts (deduped by URL, capped by depth), + with placeholder scores to be backfilled during enrichment. + Empty list on any failure. + """ + limit = DEPTH_LIMITS.get(depth, DEPTH_LIMITS["default"]) + urls = _build_urls(query, depth, subreddits) + + all_posts: List[Dict[str, Any]] = [] + workers = min(MAX_WORKERS, len(urls)) or 1 + with ThreadPoolExecutor(max_workers=workers) as executor: + futures = {executor.submit(_fetch_feed, url, query): url for url in urls} + for future in futures: + try: + all_posts.extend(future.result(timeout=FEED_TIMEOUT + 5)) + except (Exception, FuturesTimeoutError) as e: + _log(f"feed future failed: {e}") + + # Dedupe by URL (first occurrence wins). + seen: set = set() + unique: List[Dict[str, Any]] = [] + for post in all_posts: + if post["url"] not in seen: + seen.add(post["url"]) + unique.append(post) + + for i, post in enumerate(unique): + post["id"] = f"R{i + 1}" + + return unique[:limit] diff --git a/skills/last30days/scripts/lib/reddit_shreddit.py b/skills/last30days/scripts/lib/reddit_shreddit.py new file mode 100644 index 0000000..d222ac0 --- /dev/null +++ b/skills/last30days/scripts/lib/reddit_shreddit.py @@ -0,0 +1,184 @@ +"""Keyless Reddit comment enrichment via shreddit /svc endpoints. + +Reddit's ``{thread}.json`` endpoint now returns HTTP 403. The shreddit partial +endpoint ``/svc/shreddit/comments/r/{sub}/t3_{id}`` still serves HTTP 200 HTML +with no API key, embedding each comment as a ```` custom +element whose start-tag attributes carry ``score`` / ``author`` / ``created`` / +``permalink``, and whose body lives in a ``
`` +block. This module parses that markup into top comments, matching the +``top_comments`` / ``comment_insights`` shape produced by ``reddit_enrich`` so +the renderer is unaffected. + +Limitation: the comments endpoint carries the real comment count +(``total-comments``) but not the post's upvote score, so post-level ``score`` +cannot be recovered keylessly here (ScrapeCreators backup still provides it). +""" + +import html as _html +import re +import sys +from datetime import datetime +from typing import Any, Dict, List, Optional + +from . import http +from . import reddit_enrich + +# Up to N posts enriched per run, by depth (mirrors reddit_public.ENRICH_LIMITS). +ENRICH_LIMITS = { + "quick": 3, + "default": 5, + "deep": 8, +} + +# Max comments returned per post (independent of how many posts get enriched). +MAX_COMMENTS = 10 + +SVC_TIMEOUT = 12 + +# Match the exact element start tag, not +# or (lookahead requires whitespace or '>'). +_COMMENT_START = re.compile(r"])[^>]*>") +_TOTAL_COMMENTS = re.compile(r'total-comments="(\d+)"') +_PARA = re.compile(r"]*>(.*?)

", re.S) +_TAG = re.compile(r"<[^>]+>") +_WS = re.compile(r"\s+") +_NEXT_RTJSON = re.compile(r'id="t1_[A-Za-z0-9]+-(?:comment|post)-rtjson-content"') + + +def _log(msg: str) -> None: + sys.stderr.write(f"[RedditShreddit] {msg}\n") + sys.stderr.flush() + + +def extract_post_ref(url: str) -> Optional[tuple]: + """Return (subreddit, post_id) from a Reddit thread URL, or None.""" + m = re.search(r"/r/([^/]+)/comments/([A-Za-z0-9]+)", url or "") + if not m: + return None + return m.group(1), m.group(2) + + +def _svc_url(subreddit: str, post_id: str) -> str: + # sort=top guarantees Reddit front-loads the highest-scored comments on the + # first page, so the true top comments are captured even on huge threads + # (we still re-sort by score locally as a backstop). + return ( + f"https://www.reddit.com/svc/shreddit/comments/r/{subreddit}/t3_{post_id}" + f"?sort=top" + ) + + +def _attr(tag: str, name: str) -> str: + m = re.search(rf'\b{name}="([^"]*)"', tag) + return _html.unescape(m.group(1)) if m else "" + + +def _iso_to_date(value: str) -> Optional[str]: + if not value: + return None + try: + return datetime.fromisoformat(value.strip()).date().isoformat() + except (ValueError, TypeError): + return None + + +def _body_for(html_text: str, thing_id: str) -> str: + """Extract a comment's text body, anchored on its unique thingId. + + The body div id embeds the comment's thingId, so this assigns body→comment + correctly even for nested replies. The slice is bounded by the next + comment's rtjson anchor to avoid swallowing child-comment text. + """ + if not thing_id: + return "" + anchor = f'id="{thing_id}-post-rtjson-content"' + idx = html_text.find(anchor) + if idx == -1: + return "" + window = html_text[idx + len(anchor): idx + len(anchor) + 8000] + nxt = _NEXT_RTJSON.search(window) + if nxt: + window = window[: nxt.start()] + paras = _PARA.findall(window) + if not paras: + return "" + text = " ".join(_TAG.sub("", p) for p in paras) + return _WS.sub(" ", _html.unescape(text)).strip() + + +def parse_comments(html_text: str, limit: int = MAX_COMMENTS) -> List[Dict[str, Any]]: + """Parse elements into scored comment dicts (sorted desc).""" + comments: List[Dict[str, Any]] = [] + for m in _COMMENT_START.finditer(html_text or ""): + tag = m.group(0) + author = _attr(tag, "author") or "[deleted]" + if author in ("[deleted]", "[removed]"): + continue + thing_id = _attr(tag, "thingId") + body = _body_for(html_text, thing_id) + if not body or body in ("[deleted]", "[removed]"): + continue + try: + score = int(_attr(tag, "score") or 0) + except ValueError: + score = 0 + permalink = _attr(tag, "permalink") + comments.append({ + "score": score, + "author": author, + "body": body[:300], + "excerpt": body[:200], + "permalink": permalink, + "date": _iso_to_date(_attr(tag, "created")), + "url": f"https://reddit.com{permalink}" if permalink else "", + }) + + comments.sort(key=lambda c: c.get("score", 0), reverse=True) + return comments[:limit] + + +def _total_comments(html_text: str) -> Optional[int]: + m = _TOTAL_COMMENTS.search(html_text or "") + return int(m.group(1)) if m else None + + +def fetch_comments( + post_url: str, + timeout: int = SVC_TIMEOUT, +) -> Dict[str, Any]: + """Fetch and parse top comments for a Reddit post via the shreddit endpoint. + + Args: + post_url: Reddit thread URL (…/r/{sub}/comments/{id}/…) + timeout: HTTP timeout in seconds + + Returns: + Dict with 'top_comments' (list, reddit_enrich shape), 'comment_insights' + (list[str]), and 'num_comments' (int or None). Empty/None on any + failure — never raises, so the caller can fall through to SC backup. + """ + ref = extract_post_ref(post_url) + if not ref: + return {"top_comments": [], "comment_insights": [], "num_comments": None} + sub, post_id = ref + + html_text = http.get_text(_svc_url(sub, post_id), timeout=timeout, accept="text/html") + if not html_text: + return {"top_comments": [], "comment_insights": [], "num_comments": None} + + comments = parse_comments(html_text, limit=MAX_COMMENTS) + insights = reddit_enrich.extract_comment_insights(comments) + return { + "top_comments": [ + { + "score": c["score"], + "date": c["date"], + "author": c["author"], + "excerpt": c["excerpt"], + "url": c["url"], + } + for c in comments + ], + "comment_insights": insights, + "num_comments": _total_comments(html_text), + } diff --git a/tests/test_reddit_keyless.py b/tests/test_reddit_keyless.py new file mode 100644 index 0000000..26de05c --- /dev/null +++ b/tests/test_reddit_keyless.py @@ -0,0 +1,168 @@ +"""Tests for scripts/lib/reddit_keyless.py — tiered keyless Reddit pipeline.""" + +from unittest import mock + +from lib import reddit_keyless + + +def _post(i, date="2026-05-20", rel=0.0): + url = f"https://www.reddit.com/r/test/comments/{i:06d}/post_{i}/" + return { + "id": "", "title": f"Post {i}", "url": url, "score": 0, "num_comments": 0, + "subreddit": "test", "created_utc": None, "author": "u", "selftext": "", + "date": date, "engagement": {"score": 0, "num_comments": 0, "upvote_ratio": None}, + "relevance": rel, "why_relevant": "Reddit RSS", "metadata": {}, + } + + +def _scored(i, score, ncmt=0): + p = _post(i) + p["score"] = score + p["num_comments"] = ncmt + p["engagement"]["score"] = score + p["engagement"]["num_comments"] = ncmt + p["why_relevant"] = "Reddit listing" + p["metadata"] = {"post_id": f"{i:06d}"} + return p + + +class TestDiscoveryTierOrder: + """Tier 0 (.json) is tried first; RSS + scored listings are the keyless path.""" + + def test_tier0_success_skips_keyless(self): + with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[_post(1)]) as t0, \ + mock.patch.object(reddit_keyless.reddit_rss, "search_rss") as rss, \ + mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings") as lst: + out = reddit_keyless._discover("topic", "default", None) + assert len(out) == 1 + t0.assert_called_once() + rss.assert_not_called() + lst.assert_not_called() + + def test_tier0_empty_falls_to_keyless(self): + with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \ + mock.patch.object(reddit_keyless.reddit_rss, "search_rss", + return_value=[_post(1), _post(2)]) as rss, \ + mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", + return_value=[]): + out = reddit_keyless._discover("topic", "default", ["test"]) + assert len(out) == 2 + rss.assert_called_once() + + def test_listing_scores_backfill_rss_posts(self): + # RSS finds post 1 (no score); listing card for post 1 carries the score. + rss_post = _post(1) + listing_post = _scored(1, score=52692, ncmt=1743) + with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \ + mock.patch.object(reddit_keyless.reddit_rss, "search_rss", + return_value=[rss_post]), \ + mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", + return_value=[listing_post]): + out = reddit_keyless._discover("topic", "default", ["test"]) + # listing post (scored) is kept; RSS dup of same url is dropped + assert len(out) == 1 + assert out[0]["engagement"]["score"] == 52692 + assert out[0]["num_comments"] == 1743 + + def test_scores_flow_to_distinct_rss_posts(self): + # Distinct RSS post whose id matches a listing card gets backfilled. + rss_post = _post(7) # url .../000007/... + listing_post = _scored(7, score=999) + listing_post["url"] = "https://www.reddit.com/r/test/comments/zzzzzz/other/" + with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \ + mock.patch.object(reddit_keyless.reddit_rss, "search_rss", + return_value=[rss_post]), \ + mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", + return_value=[listing_post]): + out = reddit_keyless._discover("topic", "default", ["test"]) + backfilled = [p for p in out if p["url"] == rss_post["url"]][0] + assert backfilled["engagement"]["score"] == 999 + + def test_bare_query_does_not_merge_listing_discovery(self): + # No subreddits provided: derived-subreddit listings must NOT be added as + # results (avoids flooding with off-topic high-upvote posts) — only used + # to backfill scores onto the keyword-matched RSS posts. + rss_post = _post(1) # on-topic keyword match + offtopic_listing = _scored(99, score=88888) # high score, unrelated sub + offtopic_listing["url"] = "https://www.reddit.com/r/random/comments/zzz999/x/" + with mock.patch.object(reddit_keyless, "_tier0_json", return_value=[]), \ + mock.patch.object(reddit_keyless.reddit_rss, "search_rss", + return_value=[rss_post]), \ + mock.patch.object(reddit_keyless, "_top_subreddits", return_value=["random"]), \ + mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", + return_value=[offtopic_listing]): + out = reddit_keyless._discover("topic", "default", None) + urls = [p["url"] for p in out] + assert rss_post["url"] in urls + assert offtopic_listing["url"] not in urls # not merged as discovery + + def test_tier0_never_raises(self): + with mock.patch("lib.reddit_public.search", side_effect=Exception("boom")), \ + mock.patch.object(reddit_keyless.reddit_rss, "search_rss", return_value=[]), \ + mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", return_value=[]): + assert reddit_keyless._discover("t", "default", None) == [] + + +class TestSearchAndEnrich: + """Full pipeline: discover -> date filter -> rank -> enrich -> reindex.""" + + def _patch_enrich_passthrough(self): + return mock.patch.object( + reddit_keyless.reddit_shreddit, "fetch_comments", + return_value={"top_comments": [], "comment_insights": [], "num_comments": None}, + ) + + def test_returns_empty_when_no_discovery(self): + with mock.patch.object(reddit_keyless, "_discover", return_value=[]): + assert reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") == [] + + def test_date_filter_keeps_in_range_and_unknown(self): + posts = [_post(1, date="2026-05-10"), _post(2, date="2020-01-01"), + _post(3, date=None)] + with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ + self._patch_enrich_passthrough(): + out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") + titles = {p["title"] for p in out} + assert "Post 1" in titles and "Post 3" in titles + assert "Post 2" not in titles + + def test_reindexes_ids(self): + posts = [_post(1), _post(2), _post(3)] + with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ + self._patch_enrich_passthrough(): + out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") + assert [p["id"] for p in out] == ["R1", "R2", "R3"] + + def test_enrichment_attaches_comments(self): + posts = [_post(1)] + enriched = { + "top_comments": [{"score": 9, "date": "2026-05-19", "author": "a", + "excerpt": "great", "url": "https://reddit.com/x"}], + "comment_insights": ["great point about X"], + "num_comments": 14, + } + with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ + mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", + return_value=enriched): + out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") + assert out[0]["top_comments"][0]["score"] == 9 + assert out[0]["num_comments"] == 14 + assert out[0]["engagement"]["num_comments"] == 14 + + def test_enrichment_failure_keeps_posts(self): + posts = [_post(i) for i in range(8)] + with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ + mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", + side_effect=Exception("svc down")): + out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") + assert len(out) == 8 # all posts retained despite enrichment failure + + def test_only_top_n_enriched_by_depth(self): + posts = [_post(i, rel=1.0 - i / 100) for i in range(10)] + with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ + mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", + return_value={"top_comments": [], "comment_insights": [], + "num_comments": None}) as fc: + reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31", depth="quick") + # quick depth enriches only top 3 posts + assert fc.call_count == reddit_keyless.ENRICH_LIMITS["quick"] diff --git a/tests/test_reddit_listing.py b/tests/test_reddit_listing.py new file mode 100644 index 0000000..fdb0e2a --- /dev/null +++ b/tests/test_reddit_listing.py @@ -0,0 +1,85 @@ +"""Tests for scripts/lib/reddit_listing.py — keyless scored listing scrape.""" + +from pathlib import Path +from unittest import mock + +from lib import reddit_listing as rl + +FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "reddit_listing_cards_sample.html" + + +def _html(): + return FIXTURE.read_text(encoding="utf-8") + + +class TestParseCards: + """parse_cards reads cards into scored post dicts.""" + + def test_parses_five_cards(self): + posts = rl.parse_cards(_html(), query="netherlands") + assert len(posts) == 5 + + def test_real_score_and_count(self): + posts = rl.parse_cards(_html()) + top = posts[0] + assert top["score"] == 52692 # the real upvote count + assert top["engagement"]["score"] == 52692 + assert top["num_comments"] == 1743 + assert top["engagement"]["num_comments"] == 1743 + + def test_normalized_shape(self): + post = rl.parse_cards(_html())[0] + required = {"id", "title", "url", "score", "num_comments", "subreddit", + "created_utc", "author", "selftext", "date", + "engagement", "relevance", "why_relevant", "metadata"} + assert required.issubset(set(post.keys())) + assert post["why_relevant"] == "Reddit listing" + assert post["metadata"]["post_id"] # post id captured for backfill + + def test_fields_populated(self): + post = rl.parse_cards(_html())[0] + assert post["title"] + assert post["author"] == "AdSpecialist6598" + assert post["subreddit"] == "technology" + assert "/comments/" in post["url"] + assert post["date"] and len(post["date"]) == 10 + + def test_empty_html_returns_empty(self): + assert rl.parse_cards("") == [] + assert rl.parse_cards("
no cards
") == [] + + +class TestListingUrl: + def test_top_includes_timeframe(self): + u = rl._listing_url("technology", "top") + assert "community-more-posts/top/" in u and "name=technology" in u and "t=month" in u + + def test_hot_no_timeframe(self): + u = rl._listing_url("r/technology", "hot") + assert "community-more-posts/hot/" in u and "name=technology" in u and "t=" not in u + assert ".json" not in u + + +class TestFetchListings: + def test_dedupes_across_sorts(self): + with mock.patch.object(rl.http, "get_text", return_value=_html()): + posts = rl.fetch_listings(["technology"], depth="default") + urls = [p["url"] for p in posts] + assert len(urls) == len(set(urls)) # top + hot return same cards -> deduped + + def test_no_subreddits_returns_empty(self): + assert rl.fetch_listings([], depth="default") == [] + + def test_all_fetches_fail_returns_empty(self): + with mock.patch.object(rl.http, "get_text", return_value=None): + assert rl.fetch_listings(["technology"]) == [] + + +class TestScoreIndex: + def test_builds_post_id_to_score_map(self): + with mock.patch.object(rl.http, "get_text", return_value=_html()): + idx = rl.score_index(["technology"], depth="quick") + assert idx # non-empty + first = next(iter(idx.values())) + assert set(first.keys()) == {"score", "num_comments"} + assert any(v["score"] == 52692 for v in idx.values()) diff --git a/tests/test_reddit_public.py b/tests/test_reddit_public.py index d8574d5..10b9194 100644 --- a/tests/test_reddit_public.py +++ b/tests/test_reddit_public.py @@ -326,83 +326,24 @@ class TestMissingSubreddit: assert results == [] # --------------------------------------------------------------------------- -# Tests for comment enrichment (Unit 2) +# search_reddit_public is now a thin shim over the keyless pipeline. +# Full discovery + enrichment behavior is covered in test_reddit_keyless.py. # --------------------------------------------------------------------------- -class TestEnrichmentIntegration: - """search_reddit_public enriches top posts with comments.""" +class TestSearchRedditPublicDelegatesToKeyless: + """search_reddit_public delegates to reddit_keyless.search_and_enrich.""" - @mock.patch("lib.reddit_public._enrich_post") - @mock.patch("lib.reddit_public.urllib.request.urlopen") - def test_search_enriches_top_5_by_default(self, mock_urlopen, mock_enrich): - listing = _make_reddit_listing([ - {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/", - "score": 100 - i, "created_utc": 1711670400} - for i in range(10) - ]) - mock_urlopen.return_value = _mock_urlopen_ok(listing) - mock_enrich.side_effect = lambda item, timeout=10: item # pass-through + def test_delegates_with_all_args(self): + with mock.patch("lib.reddit_keyless.search_and_enrich") as mock_keyless: + mock_keyless.return_value = [{"id": "R1", "title": "x"}] + results = reddit_public.search_reddit_public( + "test", "2024-03-01", "2024-03-31", + depth="quick", subreddits=["ClaudeAI"], + ) - results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31") - - assert len(results) == 10 - # Default depth enriches top 5 - assert mock_enrich.call_count == 5 - - @mock.patch("lib.reddit_public._enrich_post") - @mock.patch("lib.reddit_public.urllib.request.urlopen") - def test_enrichment_timeout_keeps_posts(self, mock_urlopen, mock_enrich): - listing = _make_reddit_listing([ - {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/", - "score": 100 - i, "created_utc": 1711670400} - for i in range(10) - ]) - mock_urlopen.return_value = _mock_urlopen_ok(listing) - - # Some enrichments raise, some succeed - call_count = {"n": 0} - def _side_effect(item, timeout=10): - call_count["n"] += 1 - if call_count["n"] % 2 == 0: - raise TimeoutError("enrichment timed out") - return item - mock_enrich.side_effect = _side_effect - - results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31") - - # All 10 posts should still be returned - assert len(results) == 10 - - @mock.patch("lib.reddit_public._enrich_post") - @mock.patch("lib.reddit_public.urllib.request.urlopen") - def test_all_enrichment_fails_all_posts_returned(self, mock_urlopen, mock_enrich): - listing = _make_reddit_listing([ - {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/", - "score": 100 - i, "created_utc": 1711670400} - for i in range(10) - ]) - mock_urlopen.return_value = _mock_urlopen_ok(listing) - mock_enrich.side_effect = Exception("total failure") - - results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31") - - # All posts returned despite enrichment failure - assert len(results) == 10 - - @mock.patch("lib.reddit_public._enrich_post") - @mock.patch("lib.reddit_public.urllib.request.urlopen") - def test_quick_depth_enriches_top_3(self, mock_urlopen, mock_enrich): - listing = _make_reddit_listing([ - {"title": f"Post {i}", "permalink": f"/r/test/comments/{i:06d}/post_{i}/", - "score": 100 - i, "created_utc": 1711670400} - for i in range(10) - ]) - mock_urlopen.return_value = _mock_urlopen_ok(listing) - mock_enrich.side_effect = lambda item, timeout=10: item - - results = reddit_public.search_reddit_public("test", "2024-03-01", "2024-03-31", depth="quick") - - assert len(results) == 10 - # Quick depth enriches only top 3 - assert mock_enrich.call_count == 3 + assert results == [{"id": "R1", "title": "x"}] + mock_keyless.assert_called_once_with( + "test", "2024-03-01", "2024-03-31", + depth="quick", subreddits=["ClaudeAI"], + ) diff --git a/tests/test_reddit_rss.py b/tests/test_reddit_rss.py new file mode 100644 index 0000000..cb8377c --- /dev/null +++ b/tests/test_reddit_rss.py @@ -0,0 +1,96 @@ +"""Tests for scripts/lib/reddit_rss.py — keyless Reddit RSS discovery.""" + +from pathlib import Path +from unittest import mock + +from lib import reddit_rss + +FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "reddit_search_rss_sample.xml" + + +def _feed_text(): + return FIXTURE.read_text(encoding="utf-8") + + +class TestParseFeed: + """_parse_feed turns Atom entries into normalized post dicts.""" + + def test_parses_entries(self): + posts = reddit_rss._parse_feed(_feed_text(), query="lifelock") + assert len(posts) == 5 + for p in posts: + assert p["title"] + assert "/comments/" in p["url"] + assert p["url"].startswith("https://www.reddit.com/") + + def test_normalized_shape_matches_scrapecreators(self): + post = reddit_rss._parse_feed(_feed_text(), query="x")[0] + required = {"id", "title", "url", "score", "num_comments", "subreddit", + "created_utc", "author", "selftext", "date", + "engagement", "relevance", "why_relevant", "metadata"} + assert required.issubset(set(post.keys())) + assert set(post["engagement"].keys()) == {"score", "num_comments", "upvote_ratio"} + assert post["why_relevant"] == "Reddit RSS" + + def test_score_is_placeholder_zero(self): + # RSS carries no engagement score; it is backfilled during enrichment. + for p in reddit_rss._parse_feed(_feed_text(), query="x"): + assert p["score"] == 0 + assert p["engagement"]["score"] == 0 + + def test_subreddit_derivation(self): + post = reddit_rss._parse_feed(_feed_text(), query="x")[0] + assert post["subreddit"] == "Rakuten" + + def test_date_parsed_to_iso(self): + post = reddit_rss._parse_feed(_feed_text(), query="x")[0] + assert post["date"] and len(post["date"]) == 10 # YYYY-MM-DD + assert isinstance(post["created_utc"], float) + + def test_author_strips_u_prefix(self): + authors = [p["author"] for p in reddit_rss._parse_feed(_feed_text(), query="x")] + assert all(not a.startswith("/u/") and not a.startswith("u/") for a in authors) + + def test_empty_and_malformed_feed_never_raises(self): + assert reddit_rss._parse_feed("", query="x") == [] + assert reddit_rss._parse_feed("", query="x") == [] + + def test_entry_without_comments_link_skipped(self): + feed = ( + '' + 'Subreddit itself' + '' + '2026-05-20T00:00:00+00:00' + ) + assert reddit_rss._parse_feed(feed, query="x") == [] + + +class TestSearchRss: + """search_rss fans out, dedupes, assigns IDs, and honors depth limits.""" + + def test_dedupe_and_ids(self): + # Same feed returned for every URL -> deduped to 5 unique posts. + with mock.patch.object(reddit_rss.http, "get_text", return_value=_feed_text()): + posts = reddit_rss.search_rss("lifelock", depth="default", + subreddits=["Rakuten", "ConsumerAdvice"]) + urls = [p["url"] for p in posts] + assert len(urls) == len(set(urls)) # no duplicates + assert [p["id"] for p in posts] == [f"R{i+1}" for i in range(len(posts))] + + def test_depth_limit_quick(self): + with mock.patch.object(reddit_rss.http, "get_text", return_value=_feed_text()): + posts = reddit_rss.search_rss("lifelock", depth="quick") + assert len(posts) <= reddit_rss.DEPTH_LIMITS["quick"] + + def test_all_feeds_fail_returns_empty(self): + with mock.patch.object(reddit_rss.http, "get_text", return_value=None): + posts = reddit_rss.search_rss("lifelock", subreddits=["Rakuten"]) + assert posts == [] + + def test_builds_keyless_rss_urls(self): + urls = reddit_rss._build_urls("life lock", "default", ["Rakuten"]) + assert any("search.rss?q=life+lock" in u and "/r/" not in u.split("?")[0] for u in urls) + assert any("/r/Rakuten/search.rss" in u and "restrict_sr=on" in u for u in urls) + assert any("/r/Rakuten/top.rss" in u for u in urls) + assert all(".json" not in u for u in urls) # never the dead endpoint diff --git a/tests/test_reddit_shreddit.py b/tests/test_reddit_shreddit.py new file mode 100644 index 0000000..3bd423e --- /dev/null +++ b/tests/test_reddit_shreddit.py @@ -0,0 +1,103 @@ +"""Tests for scripts/lib/reddit_shreddit.py — keyless shreddit comment scrape.""" + +from pathlib import Path +from unittest import mock + +from lib import reddit_shreddit as rs + +FIXTURE = Path(__file__).resolve().parent.parent / "fixtures" / "reddit_shreddit_comments_sample.html" + + +def _html(): + return FIXTURE.read_text(encoding="utf-8") + + +class TestExtractPostRef: + def test_extracts_sub_and_id(self): + ref = rs.extract_post_ref("https://www.reddit.com/r/Rakuten/comments/1taeiw0/title/") + assert ref == ("Rakuten", "1taeiw0") + + def test_non_thread_url_returns_none(self): + assert rs.extract_post_ref("https://www.reddit.com/r/Rakuten/") is None + assert rs.extract_post_ref("") is None + + def test_svc_url_shape(self): + # sort=top guarantees the highest-scored comments land on page 1. + assert rs._svc_url("Rakuten", "1taeiw0") == ( + "https://www.reddit.com/svc/shreddit/comments/r/Rakuten/t3_1taeiw0?sort=top" + ) + + +class TestParseComments: + """parse_comments reads elements into scored dicts.""" + + def test_happy_path(self): + comments = rs.parse_comments(_html()) + assert len(comments) >= 1 + for c in comments: + assert isinstance(c["score"], int) + assert c["author"] and c["author"] not in ("[deleted]", "[removed]") + assert c["body"] + + def test_sorted_by_score_desc(self): + scores = [c["score"] for c in rs.parse_comments(_html())] + assert scores == sorted(scores, reverse=True) + + def test_deleted_and_removed_filtered(self): + authors = [c["author"] for c in rs.parse_comments(_html())] + assert "[deleted]" not in authors and "[removed]" not in authors + + def test_negative_score_retained(self): + scores = [c["score"] for c in rs.parse_comments(_html())] + assert -7 in scores # synthetic downvoted-but-real comment + + def test_limit_honored(self): + assert len(rs.parse_comments(_html(), limit=2)) == 2 + + def test_body_text_extracted(self): + bodies = [c["body"] for c in rs.parse_comments(_html())] + assert any("$750" in b or "pending" in b for b in bodies) + + def test_comment_url_built(self): + for c in rs.parse_comments(_html()): + if c["url"]: + assert c["url"].startswith("https://reddit.com/r/") + + def test_empty_html_returns_empty(self): + assert rs.parse_comments("") == [] + assert rs.parse_comments("no comments here") == [] + + +class TestTotalComments: + def test_reads_total(self): + assert rs._total_comments(_html()) == 14 + + def test_missing_returns_none(self): + assert rs._total_comments("") is None + + +class TestFetchComments: + """fetch_comments wires URL -> svc fetch -> parse, never raising.""" + + def test_happy_path(self): + url = "https://www.reddit.com/r/Rakuten/comments/1taeiw0/title/" + with mock.patch.object(rs.http, "get_text", return_value=_html()) as m: + out = rs.fetch_comments(url) + # svc endpoint, not .json + assert "/svc/shreddit/comments/" in m.call_args[0][0] + assert ".json" not in m.call_args[0][0] + assert out["num_comments"] == 14 + assert len(out["top_comments"]) >= 1 + first = out["top_comments"][0] + assert {"score", "date", "author", "excerpt", "url"} <= set(first.keys()) + assert isinstance(out["comment_insights"], list) + + def test_bad_url_returns_empty(self): + out = rs.fetch_comments("https://www.reddit.com/r/Rakuten/") + assert out["top_comments"] == [] and out["num_comments"] is None + + def test_fetch_failure_returns_empty(self): + url = "https://www.reddit.com/r/Rakuten/comments/1taeiw0/title/" + with mock.patch.object(rs.http, "get_text", return_value=None): + out = rs.fetch_comments(url) + assert out["top_comments"] == [] and out["num_comments"] is None