mirror of
https://github.com/TauricResearch/TradingAgents.git
synced 2026-09-19 19:25:24 +03:00
fix(dataflows): search all subreddits in one Reddit request
- anonymous RSS allows about one request per minute per IP, so fetching each subreddit separately hit a 429 and a 60s back-off on nearly every run - search the subreddits as one combined feed at Reddit's full page size and group posts by the subreddit each entry names; each subreddit keeps its own limit, and a full page is not taken as evidence of absence - drop the unused JSON search path, still blocked with a 403, and the per-subreddit pacing; arguments after subreddits are keyword-only - the sentiment prompt no longer asks for vote and comment counts, which the RSS feed does not carry
This commit is contained in:
@@ -1,5 +1,5 @@
|
|||||||
"""Tests for the RSS-first Reddit fetcher, its 429 backoff, the opt-in JSON
|
"""Tests for the Reddit RSS fetcher: one combined request, its 429 backoff, and
|
||||||
path's degradation (#862), and chunked-transfer error handling (#1024)."""
|
chunked-transfer error handling (#1024)."""
|
||||||
|
|
||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
@@ -80,48 +80,15 @@ class TestRssParsing:
|
|||||||
posts = reddit._fetch_subreddit_rss("NVDA", "stocks", limit=5, timeout=5.0)
|
posts = reddit._fetch_subreddit_rss("NVDA", "stocks", limit=5, timeout=5.0)
|
||||||
assert len(posts) == 2
|
assert len(posts) == 2
|
||||||
assert posts[0]["title"] == "NVDA earnings beat, stock pops"
|
assert posts[0]["title"] == "NVDA earnings beat, stock pops"
|
||||||
assert posts[0]["source"] == "rss"
|
|
||||||
assert posts[0]["score"] is None
|
|
||||||
assert posts[0]["num_comments"] is None
|
|
||||||
assert posts[0]["created_utc"] > 0
|
assert posts[0]["created_utc"] > 0
|
||||||
assert "datacenter unit" in posts[0]["selftext"]
|
assert "datacenter unit" in posts[0]["selftext"]
|
||||||
|
assert posts[0]["subreddit"] == "stocks"
|
||||||
|
|
||||||
def test_malformed_xml_reports_unavailable(self):
|
def test_malformed_xml_reports_unavailable(self):
|
||||||
with patch.object(reddit, "urlopen", return_value=_resp(lambda: b"<<not xml>>")):
|
with patch.object(reddit, "urlopen", return_value=_resp(lambda: b"<<not xml>>")):
|
||||||
assert reddit._fetch_subreddit_rss("NVDA", "stocks", 5, 5.0) is None
|
assert reddit._fetch_subreddit_rss("NVDA", "stocks", 5, 5.0) is None
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
|
||||||
class TestFetchSubredditIsRssFirst:
|
|
||||||
"""The default per-subreddit fetch goes straight to RSS — it must not hit
|
|
||||||
the WAF-blocked JSON endpoint, which only burned rate-limit budget."""
|
|
||||||
|
|
||||||
def test_delegates_to_rss_without_touching_json(self):
|
|
||||||
sentinel = [{"title": "x", "source": "rss", "score": None,
|
|
||||||
"num_comments": None, "created_utc": None, "selftext": ""}]
|
|
||||||
with patch.object(reddit, "_fetch_subreddit_rss", return_value=sentinel) as rss, \
|
|
||||||
patch.object(reddit, "urlopen",
|
|
||||||
side_effect=AssertionError("JSON endpoint must not be called")):
|
|
||||||
out = reddit._fetch_subreddit("NVDA", "stocks", 5, 5.0)
|
|
||||||
rss.assert_called_once()
|
|
||||||
assert out is sentinel
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
|
||||||
class TestJsonPathFallsBackToRss:
|
|
||||||
"""The opt-in JSON path still degrades to RSS on a 403 (kept for #862)."""
|
|
||||||
|
|
||||||
def test_403_triggers_rss(self):
|
|
||||||
err = HTTPError("url", 403, "Blocked", {}, None)
|
|
||||||
rss_posts = [{"title": "x", "source": "rss", "score": None,
|
|
||||||
"num_comments": None, "created_utc": None, "selftext": ""}]
|
|
||||||
with patch.object(reddit, "urlopen", side_effect=err), \
|
|
||||||
patch.object(reddit, "_fetch_subreddit_rss", return_value=rss_posts) as rss:
|
|
||||||
out = reddit._fetch_subreddit_json("NVDA", "stocks", 5, 5.0)
|
|
||||||
rss.assert_called_once()
|
|
||||||
assert out and out[0]["source"] == "rss"
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
class TestRss429Backoff:
|
class TestRss429Backoff:
|
||||||
def test_429_then_success_retries_once(self):
|
def test_429_then_success_retries_once(self):
|
||||||
@@ -178,12 +145,6 @@ class TestChunkedTransferErrorsHandled:
|
|||||||
with patch.object(reddit, "urlopen", return_value=_raise(http.client.IncompleteRead(b""))):
|
with patch.object(reddit, "urlopen", return_value=_raise(http.client.IncompleteRead(b""))):
|
||||||
assert reddit._fetch_subreddit_rss("NVDA", "stocks", 5, 5.0) is None
|
assert reddit._fetch_subreddit_rss("NVDA", "stocks", 5, 5.0) is None
|
||||||
|
|
||||||
def test_json_incomplete_read_falls_back_to_rss(self):
|
|
||||||
with patch.object(reddit, "urlopen", return_value=_raise(http.client.IncompleteRead(b""))), \
|
|
||||||
patch.object(reddit, "_fetch_subreddit_rss", return_value=[]) as rss:
|
|
||||||
reddit._fetch_subreddit_json("NVDA", "stocks", 5, 5.0)
|
|
||||||
rss.assert_called_once()
|
|
||||||
|
|
||||||
def test_oversized_rss_feed_is_refused_not_parsed(self):
|
def test_oversized_rss_feed_is_refused_not_parsed(self):
|
||||||
# A hostile/misbehaving endpoint streaming an unbounded body must not be
|
# A hostile/misbehaving endpoint streaming an unbounded body must not be
|
||||||
# read into memory before parsing; overflow degrades to an empty feed.
|
# read into memory before parsing; overflow degrades to an empty feed.
|
||||||
@@ -201,25 +162,12 @@ class TestFormatterHandlesRssPosts:
|
|||||||
"created_utc": reddit._iso_to_timestamp("2026-05-20T14:30:00Z"),
|
"created_utc": reddit._iso_to_timestamp("2026-05-20T14:30:00Z"),
|
||||||
"selftext": "great quarter", "source": "rss",
|
"selftext": "great quarter", "source": "rss",
|
||||||
}]
|
}]
|
||||||
with patch.object(reddit, "_fetch_subreddit", return_value=rss_posts):
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=rss_posts):
|
||||||
out = reddit.fetch_reddit_posts("NVDA", subreddits=("stocks",), inter_request_delay=0)
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("stocks",))
|
||||||
assert "via RSS feed" in out
|
assert "↑" not in out # RSS has no scores; none are invented
|
||||||
assert "↑" not in out # no fake score arrow
|
|
||||||
assert "NVDA pops" in out
|
assert "NVDA pops" in out
|
||||||
assert "great quarter" in out
|
assert "great quarter" in out
|
||||||
|
|
||||||
def test_json_posts_still_show_counts(self):
|
|
||||||
json_posts = [{
|
|
||||||
"title": "NVDA pops", "score": 1234, "num_comments": 56,
|
|
||||||
"created_utc": reddit._iso_to_timestamp("2026-05-20T14:30:00Z"),
|
|
||||||
"selftext": "",
|
|
||||||
}]
|
|
||||||
with patch.object(reddit, "_fetch_subreddit", return_value=json_posts):
|
|
||||||
out = reddit.fetch_reddit_posts("NVDA", subreddits=("stocks",), inter_request_delay=0)
|
|
||||||
assert "1234↑" in out
|
|
||||||
assert "56c" in out
|
|
||||||
assert "via RSS" not in out
|
|
||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
class TestCryptoSearchTerm:
|
class TestCryptoSearchTerm:
|
||||||
@@ -228,12 +176,12 @@ class TestCryptoSearchTerm:
|
|||||||
def _captured_ticker(self, ticker):
|
def _captured_ticker(self, ticker):
|
||||||
seen = {}
|
seen = {}
|
||||||
|
|
||||||
def fake_fetch(t, sub, limit, timeout, **kwargs):
|
def fake_fetch(t, subs, limit, timeout, **kwargs):
|
||||||
seen["ticker"] = t
|
seen["ticker"] = t
|
||||||
return []
|
return []
|
||||||
|
|
||||||
with patch.object(reddit, "_fetch_subreddit", side_effect=fake_fetch):
|
with patch.object(reddit, "_fetch_subreddit_rss", side_effect=fake_fetch):
|
||||||
reddit.fetch_reddit_posts(ticker, subreddits=("stocks",), inter_request_delay=0)
|
reddit.fetch_reddit_posts(ticker, subreddits=("stocks",))
|
||||||
return seen["ticker"]
|
return seen["ticker"]
|
||||||
|
|
||||||
def test_crypto_pair_searches_base(self):
|
def test_crypto_pair_searches_base(self):
|
||||||
@@ -244,61 +192,83 @@ class TestCryptoSearchTerm:
|
|||||||
|
|
||||||
|
|
||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
class TestFailedFetchIsNotSilence:
|
class TestOneRequestForAllSubreddits:
|
||||||
"""A throttled fetch must not be rendered as "no posts found" (#1295).
|
"""Reddit's anonymous RSS allows about one request per minute per IP, so a
|
||||||
|
request per subreddit spent a back-off on nearly every run. One combined
|
||||||
|
feed (``r/a+b+c``) carries each entry's subreddit, so nothing is lost."""
|
||||||
|
|
||||||
Returning [] for both a failed request and a genuinely empty search made the
|
def _post(self, sub, title="NVDA pops"):
|
||||||
sentiment analyst read rate limiting as real silence ("r/stocks and
|
return {"title": title, "score": None, "num_comments": None,
|
||||||
r/investing are silent"), which is a signal that was never observed.
|
"created_utc": reddit._iso_to_timestamp("2026-05-20T14:30:00Z"),
|
||||||
"""
|
"selftext": "", "source": "rss", "subreddit": sub}
|
||||||
|
|
||||||
_POST = {
|
def test_all_subreddits_share_one_request(self):
|
||||||
"title": "NVDA pops", "score": None, "num_comments": None,
|
calls = []
|
||||||
"created_utc": reddit._iso_to_timestamp("2026-05-20T14:30:00Z"),
|
|
||||||
"selftext": "", "source": "rss",
|
|
||||||
}
|
|
||||||
|
|
||||||
def _run(self, results):
|
def record(t, subs, limit, timeout):
|
||||||
"""Drive fetch_reddit_posts with a per-subreddit result sequence."""
|
calls.append((subs, limit))
|
||||||
subs = tuple(f"s{i}" for i in range(len(results)))
|
return []
|
||||||
with patch.object(reddit, "_fetch_subreddit", side_effect=list(results)):
|
|
||||||
return reddit.fetch_reddit_posts(
|
|
||||||
"NVDA", subreddits=subs, inter_request_delay=0
|
|
||||||
)
|
|
||||||
|
|
||||||
def test_failed_subreddit_is_marked_unavailable_not_empty(self):
|
with patch.object(reddit, "_fetch_subreddit_rss", side_effect=record):
|
||||||
out = self._run([None, [self._POST]])
|
reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b", "c"), limit_per_sub=5)
|
||||||
assert "unavailable" in out
|
# One full page, so a busy subreddit cannot crowd the others out.
|
||||||
assert "no posts found" not in out.split("unavailable")[0]
|
assert calls == [("a+b+c", reddit._FEED_PAGE)]
|
||||||
|
|
||||||
def test_all_sources_failing_does_not_claim_no_posts(self):
|
def test_posts_are_grouped_back_by_subreddit(self):
|
||||||
out = self._run([None, None])
|
posts = [self._post("b", "FROM B"), self._post("a", "FROM A")]
|
||||||
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=posts):
|
||||||
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b"))
|
||||||
|
assert out.index("r/a") < out.index("FROM A") < out.index("r/b") < out.index("FROM B")
|
||||||
|
|
||||||
|
def test_failed_request_is_unavailable_not_silence(self):
|
||||||
|
# #1295: a throttled fetch must not read as "no posts found".
|
||||||
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=None):
|
||||||
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b"))
|
||||||
assert "Reddit unavailable" in out
|
assert "Reddit unavailable" in out
|
||||||
assert "no Reddit posts found" not in out
|
assert "no Reddit posts found" not in out
|
||||||
|
|
||||||
def test_mixed_failure_and_empty_only_claims_silence_for_searched_subs(self):
|
|
||||||
# s0 failed, s1 genuinely returned nothing: the "no posts" claim must
|
|
||||||
# cover only s1, with s0 reported separately as unavailable.
|
|
||||||
out = self._run([None, []])
|
|
||||||
assert "r/s1" in out.split("unavailable (fetch failed)")[0]
|
|
||||||
assert "unavailable (fetch failed): r/s0" in out
|
|
||||||
|
|
||||||
def test_genuine_empty_still_reports_no_posts(self):
|
def test_genuine_empty_still_reports_no_posts(self):
|
||||||
out = self._run([[], []])
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=[]):
|
||||||
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b"))
|
||||||
assert "no Reddit posts found" in out
|
assert "no Reddit posts found" in out
|
||||||
assert "unavailable" not in out
|
assert "unavailable" not in out
|
||||||
|
|
||||||
def test_retry_is_not_spent_again_after_a_failure(self):
|
def test_subreddit_with_no_posts_is_listed_when_others_have_some(self):
|
||||||
# The 60s back-off must be paid at most once per run, so subsequent
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=[self._post("a")]):
|
||||||
# subreddits are fetched with retry disabled rather than stalling.
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b"))
|
||||||
seen = []
|
assert "r/b: <no posts found" in out
|
||||||
|
|
||||||
def record(t, sub, limit, timeout, _retry=True):
|
|
||||||
seen.append(_retry)
|
|
||||||
return None
|
|
||||||
|
|
||||||
with patch.object(reddit, "_fetch_subreddit", side_effect=record):
|
@pytest.mark.unit
|
||||||
reddit.fetch_reddit_posts(
|
def test_posts_from_an_unrequested_or_unnamed_subreddit_are_not_dropped():
|
||||||
"NVDA", subreddits=("a", "b", "c"), inter_request_delay=0
|
posts = [
|
||||||
)
|
{"title": "ELSEWHERE", "created_utc": None, "selftext": "", "subreddit": "options"},
|
||||||
assert seen == [True, False, False]
|
{"title": "NO LABEL", "created_utc": None, "selftext": "", "subreddit": ""},
|
||||||
|
]
|
||||||
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=posts):
|
||||||
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b"))
|
||||||
|
assert "ELSEWHERE" in out and "r/options" in out
|
||||||
|
assert "NO LABEL" in out
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.unit
|
||||||
|
def test_each_subreddit_keeps_its_own_quota():
|
||||||
|
busy = [{"title": f"A{i}", "created_utc": None, "selftext": "", "subreddit": "a"} for i in range(9)]
|
||||||
|
quiet = [{"title": "B0", "created_utc": None, "selftext": "", "subreddit": "b"}]
|
||||||
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=busy + quiet):
|
||||||
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b"), limit_per_sub=3)
|
||||||
|
assert "A0" in out and "A2" in out and "A3" not in out # capped per subreddit
|
||||||
|
assert "B0" in out # not crowded out
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.unit
|
||||||
|
def test_empty_subreddit_on_a_full_page_is_not_called_empty():
|
||||||
|
# A full page may have cut a quieter subreddit's posts off, so its absence
|
||||||
|
# from the page is not evidence of no posts.
|
||||||
|
full = [{"title": f"A{i}", "created_utc": None, "selftext": "", "subreddit": "a"}
|
||||||
|
for i in range(reddit._FEED_PAGE)]
|
||||||
|
with patch.object(reddit, "_fetch_subreddit_rss", return_value=full):
|
||||||
|
out = reddit.fetch_reddit_posts("NVDA", subreddits=("a", "b"))
|
||||||
|
assert "r/b: <no posts found" not in out
|
||||||
|
assert f"newest {reddit._FEED_PAGE}" in out
|
||||||
|
|||||||
@@ -102,9 +102,9 @@ def _epoch(date_str):
|
|||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
def test_reddit_historical_window_excludes_recent(monkeypatch):
|
def test_reddit_historical_window_excludes_recent(monkeypatch):
|
||||||
posts = [{"title": "NOW", "created_utc": _epoch("2026-08-30"), "source": "rss"}]
|
posts = [{"title": "NOW", "created_utc": _epoch("2026-08-30"), "source": "rss"}]
|
||||||
monkeypatch.setattr(reddit, "_fetch_subreddit", lambda *a, **k: posts)
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: posts)
|
||||||
out = reddit.fetch_reddit_posts(
|
out = reddit.fetch_reddit_posts(
|
||||||
"AAPL", subreddits=("stocks",), inter_request_delay=0,
|
"AAPL", subreddits=("stocks",),
|
||||||
start_date="2026-05-01", end_date="2026-05-08",
|
start_date="2026-05-01", end_date="2026-05-08",
|
||||||
)
|
)
|
||||||
assert "NOW" not in out
|
assert "NOW" not in out
|
||||||
@@ -114,9 +114,9 @@ def test_reddit_historical_window_excludes_recent(monkeypatch):
|
|||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
def test_reddit_live_window_keeps_in_range(monkeypatch):
|
def test_reddit_live_window_keeps_in_range(monkeypatch):
|
||||||
posts = [{"title": "INRANGE", "created_utc": _epoch("2026-05-05"), "source": "rss"}]
|
posts = [{"title": "INRANGE", "created_utc": _epoch("2026-05-05"), "source": "rss"}]
|
||||||
monkeypatch.setattr(reddit, "_fetch_subreddit", lambda *a, **k: posts)
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: posts)
|
||||||
out = reddit.fetch_reddit_posts(
|
out = reddit.fetch_reddit_posts(
|
||||||
"AAPL", subreddits=("stocks",), inter_request_delay=0,
|
"AAPL", subreddits=("stocks",),
|
||||||
start_date="2026-05-01", end_date="2026-05-08",
|
start_date="2026-05-01", end_date="2026-05-08",
|
||||||
)
|
)
|
||||||
assert "INRANGE" in out
|
assert "INRANGE" in out
|
||||||
@@ -142,9 +142,9 @@ def test_stocktwits_covered_but_empty_window_is_a_real_absence(monkeypatch):
|
|||||||
def test_reddit_covered_but_empty_window_is_a_real_absence(monkeypatch):
|
def test_reddit_covered_but_empty_window_is_a_real_absence(monkeypatch):
|
||||||
posts = [{"title": "NOW", "created_utc": _epoch("2026-08-30"), "source": "rss"},
|
posts = [{"title": "NOW", "created_utc": _epoch("2026-08-30"), "source": "rss"},
|
||||||
{"title": "OLD", "created_utc": _epoch("2026-04-20"), "source": "rss"}]
|
{"title": "OLD", "created_utc": _epoch("2026-04-20"), "source": "rss"}]
|
||||||
monkeypatch.setattr(reddit, "_fetch_subreddit", lambda *a, **k: posts)
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: posts)
|
||||||
out = reddit.fetch_reddit_posts(
|
out = reddit.fetch_reddit_posts(
|
||||||
"AAPL", subreddits=("stocks",), inter_request_delay=0,
|
"AAPL", subreddits=("stocks",),
|
||||||
start_date="2026-05-01", end_date="2026-05-08",
|
start_date="2026-05-01", end_date="2026-05-08",
|
||||||
)
|
)
|
||||||
assert "no reddit posts" in out.lower()
|
assert "no reddit posts" in out.lower()
|
||||||
@@ -156,9 +156,9 @@ def test_reddit_empty_feed_for_an_old_window_is_unavailable(monkeypatch):
|
|||||||
# Search is limited to the last week, so an empty response says nothing
|
# Search is limited to the last week, so an empty response says nothing
|
||||||
# about a window from months ago: there are no timestamps to go on, and the
|
# about a window from months ago: there are no timestamps to go on, and the
|
||||||
# lookback bound alone must decide.
|
# lookback bound alone must decide.
|
||||||
monkeypatch.setattr(reddit, "_fetch_subreddit", lambda *a, **k: [])
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: [])
|
||||||
out = reddit.fetch_reddit_posts(
|
out = reddit.fetch_reddit_posts(
|
||||||
"AAPL", subreddits=("stocks",), inter_request_delay=0,
|
"AAPL", subreddits=("stocks",),
|
||||||
start_date="2024-05-01", end_date="2024-05-08",
|
start_date="2024-05-01", end_date="2024-05-08",
|
||||||
)
|
)
|
||||||
assert "unavailable" in out and "not an absence" in out
|
assert "unavailable" in out and "not an absence" in out
|
||||||
@@ -166,8 +166,8 @@ def test_reddit_empty_feed_for_an_old_window_is_unavailable(monkeypatch):
|
|||||||
|
|
||||||
@pytest.mark.unit
|
@pytest.mark.unit
|
||||||
def test_reddit_live_empty_feed_is_a_real_absence(monkeypatch):
|
def test_reddit_live_empty_feed_is_a_real_absence(monkeypatch):
|
||||||
monkeypatch.setattr(reddit, "_fetch_subreddit", lambda *a, **k: [])
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: [])
|
||||||
out = reddit.fetch_reddit_posts("AAPL", subreddits=("stocks",), inter_request_delay=0)
|
out = reddit.fetch_reddit_posts("AAPL", subreddits=("stocks",))
|
||||||
assert "no reddit posts" in out.lower() and "past 7 days" in out
|
assert "no reddit posts" in out.lower() and "past 7 days" in out
|
||||||
assert "unavailable" not in out
|
assert "unavailable" not in out
|
||||||
|
|
||||||
@@ -187,9 +187,9 @@ def test_reddit_window_straddling_the_lookback_is_unavailable(monkeypatch):
|
|||||||
# first three days, so an empty result cannot stand for the whole window.
|
# first three days, so an empty result cannot stand for the whole window.
|
||||||
from datetime import timedelta
|
from datetime import timedelta
|
||||||
today = datetime.now(timezone.utc).date()
|
today = datetime.now(timezone.utc).date()
|
||||||
monkeypatch.setattr(reddit, "_fetch_subreddit", lambda *a, **k: [])
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: [])
|
||||||
out = reddit.fetch_reddit_posts(
|
out = reddit.fetch_reddit_posts(
|
||||||
"AAPL", subreddits=("stocks",), inter_request_delay=0,
|
"AAPL", subreddits=("stocks",),
|
||||||
start_date=str(today - timedelta(days=10)), end_date=str(today - timedelta(days=5)),
|
start_date=str(today - timedelta(days=10)), end_date=str(today - timedelta(days=5)),
|
||||||
)
|
)
|
||||||
assert "unavailable" in out
|
assert "unavailable" in out
|
||||||
@@ -201,9 +201,25 @@ def test_reddit_standard_week_window_empty_is_a_real_absence(monkeypatch):
|
|||||||
# covers it, so an empty result is genuine silence.
|
# covers it, so an empty result is genuine silence.
|
||||||
from datetime import timedelta
|
from datetime import timedelta
|
||||||
today = datetime.now(timezone.utc).date()
|
today = datetime.now(timezone.utc).date()
|
||||||
monkeypatch.setattr(reddit, "_fetch_subreddit", lambda *a, **k: [])
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: [])
|
||||||
out = reddit.fetch_reddit_posts(
|
out = reddit.fetch_reddit_posts(
|
||||||
"AAPL", subreddits=("stocks",), inter_request_delay=0,
|
"AAPL", subreddits=("stocks",),
|
||||||
start_date=str(today - timedelta(days=7)), end_date=str(today),
|
start_date=str(today - timedelta(days=7)), end_date=str(today),
|
||||||
)
|
)
|
||||||
assert "no reddit posts" in out.lower() and "unavailable" not in out
|
assert "no reddit posts" in out.lower() and "unavailable" not in out
|
||||||
|
|
||||||
|
|
||||||
|
@pytest.mark.unit
|
||||||
|
def test_reddit_full_page_does_not_vouch_for_older_days(monkeypatch):
|
||||||
|
# 100 posts from today say nothing about five days ago: the page may have
|
||||||
|
# cut older matches off, so the window stays unavailable.
|
||||||
|
from datetime import timedelta
|
||||||
|
today = datetime.now(timezone.utc).date()
|
||||||
|
ts = _epoch(str(today))
|
||||||
|
page = [{"title": f"T{i}", "created_utc": ts, "subreddit": "stocks"} for i in range(reddit._FEED_PAGE)]
|
||||||
|
monkeypatch.setattr(reddit, "_fetch_subreddit_rss", lambda *a, **k: page)
|
||||||
|
out = reddit.fetch_reddit_posts(
|
||||||
|
"AAPL", subreddits=("stocks",),
|
||||||
|
start_date=str(today - timedelta(days=6)), end_date=str(today - timedelta(days=5)),
|
||||||
|
)
|
||||||
|
assert "unavailable" in out and "no reddit posts" not in out.lower()
|
||||||
|
|||||||
@@ -160,7 +160,7 @@ Fast-moving signal. Each message carries a user-labeled sentiment tag (Bullish /
|
|||||||
<end_of_stocktwits>
|
<end_of_stocktwits>
|
||||||
|
|
||||||
### Reddit posts — r/wallstreetbets, r/stocks, r/investing (past 7 days)
|
### Reddit posts — r/wallstreetbets, r/stocks, r/investing (past 7 days)
|
||||||
Community discussion. Engagement signal via upvote score and comment count. Subreddit character matters (r/wallstreetbets is often contrarian/exuberant; r/stocks more measured; r/investing longer-term).
|
Community discussion, without vote or comment counts. Subreddit character matters (r/wallstreetbets is often contrarian/exuberant; r/stocks more measured; r/investing longer-term).
|
||||||
|
|
||||||
<start_of_reddit>
|
<start_of_reddit>
|
||||||
{reddit_block}
|
{reddit_block}
|
||||||
@@ -172,7 +172,7 @@ Community discussion. Engagement signal via upvote score and comment count. Subr
|
|||||||
|
|
||||||
2. **Look for cross-source divergences.** If news framing is bearish but StockTwits is overwhelmingly bullish, that mismatch is itself a signal — it can mean retail is leaning into a thesis the news flow hasn't caught up to (or vice versa, that retail is chasing while institutions are cautious).
|
2. **Look for cross-source divergences.** If news framing is bearish but StockTwits is overwhelmingly bullish, that mismatch is itself a signal — it can mean retail is leaning into a thesis the news flow hasn't caught up to (or vice versa, that retail is chasing while institutions are cautious).
|
||||||
|
|
||||||
3. **Weight Reddit posts by engagement.** A 400-upvote / 200-comment thread reflects community attention; a 3-upvote post is noise. Read the body excerpts for context — the title alone often misleads.
|
3. **Read Reddit posts for substance.** The feed carries no vote or comment counts, so judge a post by its body excerpt, not its title alone, and do not infer engagement.
|
||||||
|
|
||||||
4. **Distinguish opinion from event.** A news headline ("Nvidia announces $500M Corning deal") is an event; a StockTwits post ("buying NVDA, this is going to moon") is opinion. Both are inputs but should be weighted differently in your conclusions.
|
4. **Distinguish opinion from event.** A news headline ("Nvidia announces $500M Corning deal") is an event; a StockTwits post ("buying NVDA, this is going to moon") is opinion. Both are inputs but should be weighted differently in your conclusions.
|
||||||
|
|
||||||
|
|||||||
@@ -1,14 +1,9 @@
|
|||||||
"""Reddit search fetcher for ticker-specific discussion posts.
|
"""Reddit search fetcher for ticker-specific discussion posts.
|
||||||
|
|
||||||
Default path is Reddit's public Atom/RSS search feed
|
Reads Reddit's public Atom/RSS search feed, searching all subreddits in one
|
||||||
(``reddit.com/r/{sub}/search.rss``). The richer JSON search endpoint
|
combined request. The JSON search endpoint is WAF-blocked (``HTTP 403``) for
|
||||||
(``/search.json``) is reliably WAF-blocked (``HTTP 403``) for public clients
|
anonymous clients (#862), so RSS is the only path; it carries no score or comment
|
||||||
(issue #862), and probing it on every call only doubled our request volume
|
counts. On a 429 we back off once, honouring ``Retry-After``.
|
||||||
against Reddit's per-IP rate limit — tripping ``429`` on the RSS fallback — so
|
|
||||||
it is kept (``_fetch_subreddit_json``) but not used by default. On a 429 we back
|
|
||||||
off once (honouring ``Retry-After``). RSS lacks score / comment counts, so those
|
|
||||||
posts are marked and the formatter omits the metrics rather than printing fake
|
|
||||||
zeros.
|
|
||||||
|
|
||||||
A fetch that fails is reported as ``<unavailable>``, never as "no posts found":
|
A fetch that fails is reported as ``<unavailable>``, never as "no posts found":
|
||||||
the two are different claims, and passing a rate-limited fetch off as silence
|
the two are different claims, and passing a rate-limited fetch off as silence
|
||||||
@@ -23,7 +18,6 @@ from __future__ import annotations
|
|||||||
|
|
||||||
import html
|
import html
|
||||||
import http.client
|
import http.client
|
||||||
import json
|
|
||||||
import logging
|
import logging
|
||||||
import random
|
import random
|
||||||
import re
|
import re
|
||||||
@@ -61,12 +55,15 @@ def _posted_at(post) -> datetime | None:
|
|||||||
|
|
||||||
|
|
||||||
def _coverage_dates(posts) -> list:
|
def _coverage_dates(posts) -> list:
|
||||||
"""Post dates plus the search lookback start: the query is limited to the
|
"""Dates that bound the feed's coverage. The search is limited to the last
|
||||||
last week (``t=week``), so a window older than that is out of reach even
|
week (``t=week``), so the lookback start bounds it even when nothing came
|
||||||
when the feed returns nothing."""
|
back; a full page may have cut older matches off, so then only the posts
|
||||||
return [_posted_at(p) for p in posts] + [datetime.now(timezone.utc) - _SEARCH_LOOKBACK]
|
themselves do."""
|
||||||
|
dates = [_posted_at(p) for p in posts]
|
||||||
|
if len(posts) < _FEED_PAGE:
|
||||||
|
dates.append(datetime.now(timezone.utc) - _SEARCH_LOOKBACK)
|
||||||
|
return dates
|
||||||
|
|
||||||
_API = "https://www.reddit.com/r/{sub}/search.json?{qs}"
|
|
||||||
_RSS = "https://www.reddit.com/r/{sub}/search.rss?{qs}"
|
_RSS = "https://www.reddit.com/r/{sub}/search.rss?{qs}"
|
||||||
# A descriptive, identified User-Agent (per Reddit's API etiquette). Reddit
|
# A descriptive, identified User-Agent (per Reddit's API etiquette). Reddit
|
||||||
# blocks generic/anonymous tokens like bare "Mozilla/5.0" or "curl/…" but
|
# blocks generic/anonymous tokens like bare "Mozilla/5.0" or "curl/…" but
|
||||||
@@ -80,6 +77,11 @@ _ATOM_NS = {"atom": "http://www.w3.org/2005/Atom"}
|
|||||||
# investing trend more measured. Caller can override.
|
# investing trend more measured. Caller can override.
|
||||||
DEFAULT_SUBREDDITS = ("wallstreetbets", "stocks", "investing")
|
DEFAULT_SUBREDDITS = ("wallstreetbets", "stocks", "investing")
|
||||||
|
|
||||||
|
# Reddit's maximum page size. A week of posts for a ticker across the default
|
||||||
|
# subreddits fits comfortably (a busy symbol measured 12), so one full page keeps
|
||||||
|
# a high-volume subreddit from crowding the others out of a combined search.
|
||||||
|
_FEED_PAGE = 100
|
||||||
|
|
||||||
|
|
||||||
_SEARCH_LOOKBACK = timedelta(days=7) # matches t=week below
|
_SEARCH_LOOKBACK = timedelta(days=7) # matches t=week below
|
||||||
|
|
||||||
@@ -173,8 +175,7 @@ def _fetch_subreddit_rss(
|
|||||||
) -> list[dict] | None:
|
) -> list[dict] | None:
|
||||||
"""Default path: parse the public Atom search feed for a subreddit.
|
"""Default path: parse the public Atom search feed for a subreddit.
|
||||||
|
|
||||||
Carries no score / comment counts, so those fields are left None and the
|
``sub`` may be one subreddit or several joined with ``+``. On a 429 (Reddit's
|
||||||
post is tagged ``source="rss"`` for honest display. On a 429 (Reddit's
|
|
||||||
per-IP rate limit) we back off once — honouring ``Retry-After`` when
|
per-IP rate limit) we back off once — honouring ``Retry-After`` when
|
||||||
present — before giving up, so a transient burst doesn't blank the feed.
|
present — before giving up, so a transient burst doesn't blank the feed.
|
||||||
|
|
||||||
@@ -213,79 +214,37 @@ def _fetch_subreddit_rss(
|
|||||||
title_el = entry.find("atom:title", _ATOM_NS)
|
title_el = entry.find("atom:title", _ATOM_NS)
|
||||||
published_el = entry.find("atom:published", _ATOM_NS)
|
published_el = entry.find("atom:published", _ATOM_NS)
|
||||||
content_el = entry.find("atom:content", _ATOM_NS)
|
content_el = entry.find("atom:content", _ATOM_NS)
|
||||||
|
category_el = entry.find("atom:category", _ATOM_NS)
|
||||||
posts.append({
|
posts.append({
|
||||||
"title": (title_el.text if title_el is not None else "") or "",
|
"title": (title_el.text if title_el is not None else "") or "",
|
||||||
"score": None,
|
|
||||||
"num_comments": None,
|
|
||||||
"created_utc": _iso_to_timestamp(
|
"created_utc": _iso_to_timestamp(
|
||||||
published_el.text if published_el is not None else None
|
published_el.text if published_el is not None else None
|
||||||
),
|
),
|
||||||
"selftext": _strip_html(content_el.text if content_el is not None else ""),
|
"selftext": _strip_html(content_el.text if content_el is not None else ""),
|
||||||
"source": "rss",
|
# A combined feed names each entry's subreddit; a single-subreddit
|
||||||
|
# feed may omit it, and then it can only be that one.
|
||||||
|
"subreddit": category_el.get("term") if category_el is not None
|
||||||
|
else (sub if "+" not in sub else ""),
|
||||||
})
|
})
|
||||||
return posts
|
return posts
|
||||||
|
|
||||||
|
|
||||||
def _fetch_subreddit_json(
|
|
||||||
ticker: str,
|
|
||||||
sub: str,
|
|
||||||
limit: int,
|
|
||||||
timeout: float,
|
|
||||||
) -> list[dict]:
|
|
||||||
"""Richer JSON search path (carries score / comment counts).
|
|
||||||
|
|
||||||
Reddit's WAF currently returns ``403 Blocked`` on this endpoint for
|
|
||||||
non-OAuth clients (issue #862), so it is NOT used by default — calling it on
|
|
||||||
every request only doubled our volume against the per-IP rate limit and
|
|
||||||
triggered 429s on the RSS fallback. Kept for the day the WAF relaxes or an
|
|
||||||
OAuth token is wired in; degrades to RSS on failure.
|
|
||||||
"""
|
|
||||||
url = _API.format(sub=sub, qs=_search_qs(ticker, limit))
|
|
||||||
req = Request(url, headers={"User-Agent": _UA, "Accept": "application/json"})
|
|
||||||
try:
|
|
||||||
with urlopen(req, timeout=timeout) as resp:
|
|
||||||
payload = json.loads(_read_capped(resp))
|
|
||||||
children = (payload.get("data") or {}).get("children") or []
|
|
||||||
return [c.get("data", {}) for c in children if isinstance(c, dict)]
|
|
||||||
except (OSError, http.client.HTTPException, json.JSONDecodeError) as exc:
|
|
||||||
logger.warning(
|
|
||||||
"Reddit JSON fetch failed for r/%s · %s: %s — falling back to RSS feed.",
|
|
||||||
sub, ticker, exc,
|
|
||||||
)
|
|
||||||
return _fetch_subreddit_rss(ticker, sub, limit, timeout)
|
|
||||||
|
|
||||||
|
|
||||||
def _fetch_subreddit(
|
|
||||||
ticker: str,
|
|
||||||
sub: str,
|
|
||||||
limit: int,
|
|
||||||
timeout: float,
|
|
||||||
_retry: bool = True,
|
|
||||||
) -> list[dict] | None:
|
|
||||||
"""Fetch one subreddit, RSS-first. ``None`` means the fetch failed.
|
|
||||||
|
|
||||||
The JSON search endpoint is reliably WAF-blocked (403) for public clients,
|
|
||||||
so we go straight to the RSS feed — which serves our identified User-Agent
|
|
||||||
reliably — halving our request volume against Reddit's per-IP rate limit.
|
|
||||||
"""
|
|
||||||
return _fetch_subreddit_rss(ticker, sub, limit, timeout, _retry=_retry)
|
|
||||||
|
|
||||||
|
|
||||||
def fetch_reddit_posts(
|
def fetch_reddit_posts(
|
||||||
ticker: str,
|
ticker: str,
|
||||||
subreddits: Iterable[str] = DEFAULT_SUBREDDITS,
|
subreddits: Iterable[str] = DEFAULT_SUBREDDITS,
|
||||||
|
*,
|
||||||
limit_per_sub: int = 5,
|
limit_per_sub: int = 5,
|
||||||
timeout: float = 10.0,
|
timeout: float = 10.0,
|
||||||
inter_request_delay: float = 1.0,
|
|
||||||
start_date: str | None = None,
|
start_date: str | None = None,
|
||||||
end_date: str | None = None,
|
end_date: str | None = None,
|
||||||
) -> str:
|
) -> str:
|
||||||
"""Fetch recent Reddit posts mentioning ``ticker`` across finance
|
"""Fetch recent Reddit posts mentioning ``ticker`` across finance
|
||||||
subreddits and return them as a formatted plaintext block.
|
subreddits and return them as a formatted plaintext block.
|
||||||
|
|
||||||
``inter_request_delay`` paces the (now RSS-only) per-subreddit requests to
|
All subreddits are searched in one combined feed (``r/a+b+c``): anonymous
|
||||||
stay under Reddit's public per-IP rate limit; combined with the RSS-first
|
RSS allows about one request per minute per IP, so a request per subreddit
|
||||||
path it makes 429s rare even when several analyses run back-to-back.
|
spent a back-off on almost every run. Each entry names its subreddit, and
|
||||||
|
posts are grouped back by it.
|
||||||
|
|
||||||
When ``start_date``/``end_date`` (yyyy-mm-dd) are given, posts are trimmed to
|
When ``start_date``/``end_date`` (yyyy-mm-dd) are given, posts are trimmed to
|
||||||
that window so a historical run does not leak current discussion into a
|
that window so a historical run does not leak current discussion into a
|
||||||
@@ -295,86 +254,50 @@ def fetch_reddit_posts(
|
|||||||
# ("BTC") so the query actually matches discussion instead of near-nothing.
|
# ("BTC") so the query actually matches discussion instead of near-nothing.
|
||||||
ticker = crypto_base(ticker) or ticker
|
ticker = crypto_base(ticker) or ticker
|
||||||
subreddits = list(subreddits)
|
subreddits = list(subreddits)
|
||||||
blocks = []
|
label = ", ".join(f"r/{s}" for s in subreddits)
|
||||||
total_posts = 0
|
fetched = _fetch_subreddit_rss(ticker, "+".join(subreddits), _FEED_PAGE, timeout)
|
||||||
unavailable = []
|
if fetched is None:
|
||||||
fetched_posts = []
|
return f"<Reddit unavailable: fetch failed ({label}); this is not an absence of discussion>"
|
||||||
allow_retry = True
|
|
||||||
for i, sub in enumerate(subreddits):
|
|
||||||
if i > 0 and inter_request_delay:
|
|
||||||
time.sleep(_jitter(inter_request_delay))
|
|
||||||
fetched = _fetch_subreddit(ticker, sub, limit_per_sub, timeout, _retry=allow_retry)
|
|
||||||
if fetched is None:
|
|
||||||
# A failed fetch is not an absence of discussion, so it must not be
|
|
||||||
# rendered as "no posts found" (#1295). One failure also means the
|
|
||||||
# per-IP budget is likely gone, so skip the (now 60s) back-off on
|
|
||||||
# the remaining subreddits rather than stalling the run on retries
|
|
||||||
# that cannot succeed; #1286 tracks coordinating this properly.
|
|
||||||
allow_retry = False
|
|
||||||
unavailable.append(sub)
|
|
||||||
blocks.append(f"r/{sub}: <unavailable: fetch failed, not an absence of posts>")
|
|
||||||
continue
|
|
||||||
posts = _within_window(fetched, start_date, end_date)
|
|
||||||
total_posts += len(posts)
|
|
||||||
fetched_posts.extend(fetched)
|
|
||||||
if not posts:
|
|
||||||
gap = start_date and end_date and coverage_gap(
|
|
||||||
_coverage_dates(fetched), start_date, end_date,
|
|
||||||
f"r/{sub}", f"discussion of {ticker.upper()}",
|
|
||||||
)
|
|
||||||
period = f"within {start_date}..{end_date}" if start_date and end_date else "in the past 7 days"
|
|
||||||
blocks.append(f"r/{sub}: {gap or f'<no posts found mentioning {ticker.upper()} {period}>'}")
|
|
||||||
continue
|
|
||||||
|
|
||||||
via_rss = any(p.get("source") == "rss" for p in posts)
|
window = bool(start_date and end_date)
|
||||||
header = f"r/{sub} — {len(posts)} recent posts mentioning {ticker.upper()}"
|
posts = _within_window(fetched, start_date, end_date)
|
||||||
header += " (via RSS feed; scores/comments unavailable):" if via_rss else ":"
|
if not posts:
|
||||||
lines = [header]
|
gap = window and coverage_gap(
|
||||||
for p in posts:
|
_coverage_dates(fetched), start_date, end_date,
|
||||||
title = (p.get("title") or "").replace("\n", " ").strip()
|
"Reddit search", f"discussion of {ticker.upper()}",
|
||||||
score = p.get("score")
|
)
|
||||||
comments = p.get("num_comments")
|
period = f"within {start_date}..{end_date}" if window else "in the past 7 days"
|
||||||
created = p.get("created_utc")
|
return gap or f"<no Reddit posts found mentioning {ticker.upper()} across {label} {period}>"
|
||||||
created_str = (
|
|
||||||
time.strftime("%Y-%m-%d", time.gmtime(created)) if created else "?"
|
# Group by the subreddit each entry names, in the requested order. Nothing
|
||||||
|
# is dropped: an unlabelled post from a one-subreddit request belongs to it,
|
||||||
|
# and any other name gets its own block.
|
||||||
|
by_sub = {s.lower(): (s, []) for s in subreddits}
|
||||||
|
for p in posts:
|
||||||
|
name = p.get("subreddit") or (subreddits[0] if len(subreddits) == 1 else "unknown")
|
||||||
|
by_sub.setdefault(name.lower(), (name, []))[1].append(p)
|
||||||
|
|
||||||
|
page_full = len(fetched) >= _FEED_PAGE
|
||||||
|
blocks = []
|
||||||
|
for sub, sub_posts in by_sub.values():
|
||||||
|
if not sub_posts:
|
||||||
|
blocks.append(
|
||||||
|
f"r/{sub}: <not among the newest {_FEED_PAGE} matches across {label}>"
|
||||||
|
if page_full else f"r/{sub}: <no posts found mentioning {ticker.upper()}>"
|
||||||
)
|
)
|
||||||
# Score / comment counts are absent on the RSS fallback path —
|
continue
|
||||||
# show them only when present rather than printing fake zeros.
|
sub_posts = sub_posts[:limit_per_sub] # the feed is newest-first
|
||||||
meta = created_str
|
lines = [f"r/{sub} — {len(sub_posts)} recent posts mentioning {ticker.upper()}:"]
|
||||||
if score is not None and comments is not None:
|
for p in sub_posts:
|
||||||
meta += f" · {score:>4}↑ · {comments:>3}c"
|
title = (p.get("title") or "").replace("\n", " ").strip()
|
||||||
|
created = p.get("created_utc")
|
||||||
|
created_str = time.strftime("%Y-%m-%d", time.gmtime(created)) if created else "?"
|
||||||
selftext = (p.get("selftext") or "").replace("\n", " ").strip()
|
selftext = (p.get("selftext") or "").replace("\n", " ").strip()
|
||||||
if len(selftext) > 240:
|
if len(selftext) > 240:
|
||||||
selftext = selftext[:240] + "…"
|
selftext = selftext[:240] + "…"
|
||||||
lines.append(
|
lines.append(
|
||||||
f" [{meta}] {title}"
|
f" [{created_str}] {title}"
|
||||||
+ (f"\n body excerpt: {selftext}" if selftext else "")
|
+ (f"\n body excerpt: {selftext}" if selftext else "")
|
||||||
)
|
)
|
||||||
blocks.append("\n".join(lines))
|
blocks.append("\n".join(lines))
|
||||||
|
|
||||||
if total_posts == 0:
|
|
||||||
searched = [s for s in subreddits if s not in unavailable]
|
|
||||||
if not searched:
|
|
||||||
# Every source failed: claiming "no posts" here would assert a
|
|
||||||
# silence we never observed.
|
|
||||||
return (
|
|
||||||
f"<Reddit unavailable: every source failed to fetch "
|
|
||||||
f"({', '.join(f'r/{s}' for s in unavailable)}); this is not an "
|
|
||||||
f"absence of discussion>"
|
|
||||||
)
|
|
||||||
gap = start_date and end_date and coverage_gap(
|
|
||||||
_coverage_dates(fetched_posts), start_date, end_date,
|
|
||||||
"Reddit search", f"discussion of {ticker.upper()}",
|
|
||||||
)
|
|
||||||
period = f"within {start_date}..{end_date}" if start_date and end_date else "in the past 7 days"
|
|
||||||
summary = gap or (
|
|
||||||
f"<no Reddit posts found mentioning {ticker.upper()} across "
|
|
||||||
f"{', '.join(f'r/{s}' for s in searched)} {period}>"
|
|
||||||
)
|
|
||||||
if unavailable:
|
|
||||||
summary += (
|
|
||||||
f"\n<unavailable (fetch failed): "
|
|
||||||
f"{', '.join(f'r/{s}' for s in unavailable)}>"
|
|
||||||
)
|
|
||||||
return summary
|
|
||||||
return "\n\n".join(blocks)
|
return "\n\n".join(blocks)
|
||||||
|
|||||||
Reference in New Issue
Block a user