fix(reddit): use browser-like headers to fix HTTP 403 from urllib

Reddit's public JSON endpoint returns 403 to requests carrying the
generic User-Agent and minimal header set urllib defaults to, while
matching curl requests succeed. Switch to a current-Chrome User-Agent
and add Accept-Language / Accept-Encoding / Connection headers so the
fingerprint matches a normal browser. Reddit now serves gzip when
Accept-Encoding includes it, so decompress the body before JSON parse.
Update the user-agent assertion in tests/test_reddit_public.py to match
the new browser-like string. Closes #199.

Co-authored-by: Franco Carballar <francocarballar@gmail.com>
This commit is contained in:
Trevin Chow
2026-05-17 01:29:26 -07:00
parent a4f1f94802
commit 4bae05e7fa
2 changed files with 14 additions and 3 deletions
+13 -2
View File
@@ -11,6 +11,7 @@ Handles 429 rate limits with exponential backoff, HTML anti-bot responses,
network timeouts, and missing subreddits. network timeouts, and missing subreddits.
""" """
import gzip
import json import json
import sys import sys
import time import time
@@ -21,7 +22,11 @@ from concurrent.futures import ThreadPoolExecutor, TimeoutError as FuturesTimeou
from typing import Any, Dict, List, Optional from typing import Any, Dict, List, Optional
USER_AGENT = "last30days/3.0 (research tool)" USER_AGENT = (
"Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) "
"AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/124.0.0.0 Safari/537.36"
)
# Depth-aware limits for thread counts # Depth-aware limits for thread counts
DEPTH_LIMITS = { DEPTH_LIMITS = {
@@ -60,6 +65,9 @@ def _fetch_json(url: str, timeout: int = 15) -> Optional[Dict[str, Any]]:
headers = { headers = {
"User-Agent": USER_AGENT, "User-Agent": USER_AGENT,
"Accept": "application/json", "Accept": "application/json",
"Accept-Language": "en-US,en;q=0.9",
"Accept-Encoding": "gzip, deflate",
"Connection": "keep-alive",
} }
req = urllib.request.Request(url, headers=headers) req = urllib.request.Request(url, headers=headers)
@@ -71,7 +79,10 @@ def _fetch_json(url: str, timeout: int = 15) -> Optional[Dict[str, Any]]:
_log(f"Anti-bot HTML response (Content-Type: {content_type})") _log(f"Anti-bot HTML response (Content-Type: {content_type})")
return None return None
body = resp.read().decode("utf-8") raw = resp.read()
if resp.headers.get("Content-Encoding", "").lower() == "gzip":
raw = gzip.decompress(raw)
body = raw.decode("utf-8")
return json.loads(body) return json.loads(body)
except urllib.error.HTTPError as e: except urllib.error.HTTPError as e:
+1 -1
View File
@@ -313,7 +313,7 @@ class TestSearchRedditPublicHighLevel:
reddit_public.search("test") reddit_public.search("test")
req = mock_urlopen.call_args[0][0] req = mock_urlopen.call_args[0][0]
assert req.get_header("User-agent") == "last30days/3.0 (research tool)" assert "Mozilla/5.0" in req.get_header("User-agent")
class TestMissingSubreddit: class TestMissingSubreddit: