| 1 | """Tests for scripts/lib/reddit_keyless.py: tiered keyless Reddit pipeline.""" |
| 2 | |
| 3 | from unittest import mock |
| 4 | |
| 5 | from lib import reddit_keyless |
| 6 | |
| 7 | |
| 8 | def _post(i, date="2026-05-20", rel=0.0): |
| 9 | url = f"https://www.reddit.com/r/test/comments/{i:06d}/post_{i}/" |
| 10 | return { |
| 11 | "id": "", "title": f"Post {i}", "url": url, "score": 0, "num_comments": 0, |
| 12 | "subreddit": "test", "created_utc": None, "author": "u", "selftext": "", |
| 13 | "date": date, "engagement": {"score": 0, "num_comments": 0, "upvote_ratio": None}, |
| 14 | "relevance": rel, "why_relevant": "Reddit search", "metadata": {}, |
| 15 | } |
| 16 | |
| 17 | |
| 18 | def _scored(i, score, ncmt=0): |
| 19 | p = _post(i) |
| 20 | p["score"] = score |
| 21 | p["num_comments"] = ncmt |
| 22 | p["engagement"]["score"] = score |
| 23 | p["engagement"]["num_comments"] = ncmt |
| 24 | p["why_relevant"] = "Reddit listing" |
| 25 | p["metadata"] = {"post_id": f"{i:06d}"} |
| 26 | return p |
| 27 | |
| 28 | |
| 29 | def _searched(i, score, ncmt=0, rel=0.5): |
| 30 | """A site-search result: dated and scored straight from the search page.""" |
| 31 | p = _scored(i, score, ncmt) |
| 32 | p["relevance"] = rel |
| 33 | p["why_relevant"] = "Reddit search" |
| 34 | return p |
| 35 | |
| 36 | |
| 37 | def _lanes(search=(), listing=(), arctic_listing=(), arctic_scores=None): |
| 38 | """Patch every discovery lane at its module boundary; return the mocks.""" |
| 39 | return ( |
| 40 | mock.patch.object(reddit_keyless.reddit_search, "search", return_value=list(search)), |
| 41 | mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", |
| 42 | return_value=list(listing)), |
| 43 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_listings", |
| 44 | return_value=list(arctic_listing)), |
| 45 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_scores", |
| 46 | return_value=dict(arctic_scores or {})), |
| 47 | ) |
| 48 | |
| 49 | |
| 50 | class TestDiscovery: |
| 51 | """Reddit site search + scored listings are the keyless discovery path.""" |
| 52 | |
| 53 | def test_bare_run_returns_search_posts_without_listing_requests(self): |
| 54 | hits = [_searched(1, score=412, ncmt=38), _searched(2, score=77, ncmt=5)] |
| 55 | p_search, p_listing, p_arctic_listing, p_scores = _lanes(search=hits) |
| 56 | with p_search as search, p_listing as listing, \ |
| 57 | p_arctic_listing as arctic_listing, p_scores as scores: |
| 58 | out = reddit_keyless._discover("topic", "default", None) |
| 59 | search.assert_called_once() |
| 60 | assert search.call_args.kwargs["subreddits"] is None |
| 61 | listing.assert_not_called() |
| 62 | arctic_listing.assert_not_called() |
| 63 | scores.assert_not_called() # real scores, nothing to backfill |
| 64 | assert [p["url"] for p in out] == [h["url"] for h in hits] |
| 65 | assert [p["engagement"]["score"] for p in out] == [412, 77] |
| 66 | assert [p["num_comments"] for p in out] == [38, 5] |
| 67 | |
| 68 | def test_targeted_run_merges_search_and_listing_first_writer_wins(self): |
| 69 | listing_post = _scored(1, score=52692, ncmt=1743) |
| 70 | listing_only = _scored(2, score=10) |
| 71 | search_dup = _searched(1, score=50000, ncmt=1700) # same url as listing_post |
| 72 | search_only = _searched(3, score=9) |
| 73 | p_search, p_listing, p_arctic_listing, p_scores = _lanes( |
| 74 | search=[search_dup, search_only], listing=[listing_post, listing_only]) |
| 75 | with p_search as search, p_listing as listing, p_arctic_listing, p_scores: |
| 76 | out = reddit_keyless._discover("topic", "default", ["test"]) |
| 77 | assert search.call_args.kwargs["subreddits"] == ["test"] |
| 78 | assert listing.call_args.args[0] == ["test"] |
| 79 | urls = [p["url"] for p in out] |
| 80 | assert urls == [listing_post["url"], listing_only["url"], search_only["url"]] |
| 81 | assert out[0]["why_relevant"] == "Reddit listing" # one copy, listing kept |
| 82 | assert out[0]["engagement"]["score"] == 52692 |
| 83 | |
| 84 | def test_targeted_listing_score_fills_distinct_search_post(self): |
| 85 | # A search post whose id matches a listing card under another url takes |
| 86 | # the listing's live score. |
| 87 | search_post = _searched(7, score=0) |
| 88 | listing_post = _scored(7, score=999) |
| 89 | listing_post["url"] = "https://www.reddit.com/r/test/comments/zzzzzz/other/" |
| 90 | p_search, p_listing, p_arctic_listing, p_scores = _lanes( |
| 91 | search=[search_post], listing=[listing_post]) |
| 92 | with p_search, p_listing, p_arctic_listing, p_scores: |
| 93 | out = reddit_keyless._discover("topic", "default", ["test"]) |
| 94 | filled = [p for p in out if p["url"] == search_post["url"]][0] |
| 95 | assert filled["engagement"]["score"] == 999 |
| 96 | |
| 97 | def test_zero_score_search_post_gets_arctic_fill(self): |
| 98 | unscored = _searched(4, score=0) |
| 99 | scored = _searched(5, score=120, ncmt=3) |
| 100 | p_search, p_listing, p_arctic_listing, p_scores = _lanes( |
| 101 | search=[unscored, scored], |
| 102 | arctic_scores={"000004": {"score": 31, "num_comments": 6}}) |
| 103 | with p_search, p_listing, p_arctic_listing, p_scores as scores: |
| 104 | out = reddit_keyless._discover("topic", "default", None) |
| 105 | scores.assert_called_once_with(["000004"]) |
| 106 | by_url = {p["url"]: p for p in out} |
| 107 | assert by_url[unscored["url"]]["engagement"]["score"] == 31 |
| 108 | assert by_url[unscored["url"]]["num_comments"] == 6 |
| 109 | assert by_url[scored["url"]]["engagement"]["score"] == 120 |
| 110 | |
| 111 | def test_bare_query_does_not_merge_listing_discovery(self): |
| 112 | # No subreddits provided: no listing is fetched, so high-upvote |
| 113 | # off-topic listing posts can never flood the keyword-matched results. |
| 114 | on_topic = _searched(1, score=15) |
| 115 | offtopic_listing = _scored(99, score=88888) |
| 116 | offtopic_listing["url"] = "https://www.reddit.com/r/random/comments/zzz999/x/" |
| 117 | p_search, p_listing, p_arctic_listing, p_scores = _lanes( |
| 118 | search=[on_topic], listing=[offtopic_listing], arctic_listing=[offtopic_listing]) |
| 119 | with p_search, p_listing as listing, p_arctic_listing as arctic_listing, p_scores: |
| 120 | out = reddit_keyless._discover("topic", "default", None) |
| 121 | urls = [p["url"] for p in out] |
| 122 | assert urls == [on_topic["url"]] |
| 123 | listing.assert_not_called() |
| 124 | arctic_listing.assert_not_called() |
| 125 | |
| 126 | def test_discover_never_raises_returns_empty(self): |
| 127 | p_search, p_listing, p_arctic_listing, p_scores = _lanes() |
| 128 | with p_search, p_listing, p_arctic_listing, p_scores: |
| 129 | assert reddit_keyless._discover("t", "default", None) == [] |
| 130 | |
| 131 | def test_empty_search_makes_keyless_path_return_empty(self): |
| 132 | p_search, p_listing, p_arctic_listing, p_scores = _lanes() |
| 133 | with p_search, p_listing, p_arctic_listing, p_scores: |
| 134 | assert reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") == [] |
| 135 | |
| 136 | def test_search_window_follows_lookback(self): |
| 137 | p_search, p_listing, p_arctic_listing, p_scores = _lanes() |
| 138 | with p_search as search, p_listing, p_arctic_listing, p_scores: |
| 139 | reddit_keyless.search_and_enrich("t", "2026-05-24", "2026-05-31", depth="quick") |
| 140 | kwargs = search.call_args.kwargs |
| 141 | assert kwargs["from_date"] == "2026-05-24" |
| 142 | assert kwargs["to_date"] == "2026-05-31" |
| 143 | assert kwargs["depth"] == "quick" |
| 144 | |
| 145 | |
| 146 | class TestSearchAndEnrich: |
| 147 | """Full pipeline: discover -> date filter -> rank -> enrich -> reindex.""" |
| 148 | |
| 149 | def _patch_enrich_passthrough(self): |
| 150 | return mock.patch.object( |
| 151 | reddit_keyless.reddit_shreddit, "fetch_comments", |
| 152 | return_value={"top_comments": [], "comment_insights": [], "num_comments": None}, |
| 153 | ) |
| 154 | |
| 155 | def test_returns_empty_when_no_discovery(self): |
| 156 | with mock.patch.object(reddit_keyless, "_discover", return_value=[]): |
| 157 | assert reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") == [] |
| 158 | |
| 159 | def test_date_filter_keeps_in_range_and_unknown(self): |
| 160 | posts = [_post(1, date="2026-05-10"), _post(2, date="2020-01-01"), |
| 161 | _post(3, date=None)] |
| 162 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 163 | self._patch_enrich_passthrough(): |
| 164 | out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") |
| 165 | titles = {p["title"] for p in out} |
| 166 | assert "Post 1" in titles and "Post 3" in titles |
| 167 | assert "Post 2" not in titles |
| 168 | |
| 169 | def test_reindexes_ids(self): |
| 170 | posts = [_post(1), _post(2), _post(3)] |
| 171 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 172 | self._patch_enrich_passthrough(): |
| 173 | out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") |
| 174 | assert [p["id"] for p in out] == ["R1", "R2", "R3"] |
| 175 | |
| 176 | def test_enrichment_attaches_comments(self): |
| 177 | posts = [_post(1)] |
| 178 | enriched = { |
| 179 | "top_comments": [{"score": 9, "date": "2026-05-19", "author": "a", |
| 180 | "excerpt": "great", "url": "https://reddit.com/x"}], |
| 181 | "comment_insights": ["great point about X"], |
| 182 | "num_comments": 14, |
| 183 | } |
| 184 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 185 | mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", |
| 186 | return_value=enriched): |
| 187 | out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") |
| 188 | assert out[0]["top_comments"][0]["score"] == 9 |
| 189 | assert out[0]["num_comments"] == 14 |
| 190 | assert out[0]["engagement"]["num_comments"] == 14 |
| 191 | |
| 192 | def test_enrichment_failure_keeps_posts(self): |
| 193 | posts = [_post(i) for i in range(8)] |
| 194 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 195 | mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", |
| 196 | side_effect=Exception("svc down")): |
| 197 | out = reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31") |
| 198 | assert len(out) == 8 # all posts retained despite enrichment failure |
| 199 | |
| 200 | def test_only_top_n_enriched_by_depth(self): |
| 201 | posts = [_post(i, rel=1.0 - i / 100) for i in range(10)] |
| 202 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 203 | mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", |
| 204 | return_value={"top_comments": [], "comment_insights": [], |
| 205 | "num_comments": None}) as fc: |
| 206 | reddit_keyless.search_and_enrich("t", "2026-05-01", "2026-05-31", depth="quick") |
| 207 | # quick depth enriches only top 3 posts |
| 208 | assert fc.call_count == reddit_keyless.ENRICH_LIMITS["quick"] |
| 209 | |
| 210 | |
| 211 | class TestSlotPriority: |
| 212 | """Enrichment slot selection prefers entity-matching posts (R1-R3).""" |
| 213 | |
| 214 | @staticmethod |
| 215 | def _titled(i, title, score=0, selftext=""): |
| 216 | p = _post(i) |
| 217 | p["title"] = title |
| 218 | p["selftext"] = selftext |
| 219 | p["score"] = score |
| 220 | p["engagement"]["score"] = score |
| 221 | return p |
| 222 | |
| 223 | def test_on_topic_low_score_beats_off_topic_high_score(self): |
| 224 | # 3 off-topic monsters + 2 on-topic small threads; quick depth = 3 slots. |
| 225 | posts = [ |
| 226 | self._titled(1, "Stop asking what model to run", score=2662), |
| 227 | self._titled(2, "RTX 4090 PSA", score=2068), |
| 228 | self._titled(3, "Gemma 4 release", score=997), |
| 229 | self._titled(4, "My OpenClaw self-migrated", score=73), |
| 230 | self._titled(5, "Using openclaw with Claude API key is so expensive", score=47), |
| 231 | ] |
| 232 | enriched_urls = [] |
| 233 | |
| 234 | def _capture(url): |
| 235 | enriched_urls.append(url) |
| 236 | return {"top_comments": [], "comment_insights": [], "num_comments": None} |
| 237 | |
| 238 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 239 | mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", |
| 240 | side_effect=_capture): |
| 241 | reddit_keyless.search_and_enrich( |
| 242 | "openclaw", "2026-05-01", "2026-05-31", depth="quick") |
| 243 | assert posts[3]["url"] in enriched_urls |
| 244 | assert posts[4]["url"] in enriched_urls |
| 245 | assert len(enriched_urls) == reddit_keyless.ENRICH_LIMITS["quick"] |
| 246 | |
| 247 | def test_slot_priority_grounds_on_head_token_not_full_phrase(self): |
| 248 | # Mirrors rerank's head-token grounding: a post naming the brand head |
| 249 | # ("Stripe") lands in the match tier even without the trailing search |
| 250 | # descriptor ("payments"), so it is not buried under an unrelated |
| 251 | # high-upvote post that never names the brand. |
| 252 | head_only = self._titled(1, "Stripe is friendly to 'friendly fraud'", score=5) |
| 253 | off_topic = self._titled(2, "PayPal raises dispute fees again", score=900) |
| 254 | out = reddit_keyless._slot_priority("Stripe payments", [off_topic, head_only]) |
| 255 | assert out[0] is head_only |
| 256 | assert out[1] is off_topic |
| 257 | |
| 258 | def test_intent_modifier_topic_prioritizes_head_token_match(self): |
| 259 | # Intent-modifier topics still partition by the brand head token: the |
| 260 | # on-entity post wins over a high-upvote post that never names the brand. |
| 261 | on_topic = self._titled(1, "Hermes Agent v0.13 is great", score=1) |
| 262 | off_topic = self._titled(2, "LangGraph tutorial walkthrough", score=900) |
| 263 | out = reddit_keyless._slot_priority("Hermes Agent review", [off_topic, on_topic]) |
| 264 | assert out[0] is on_topic |
| 265 | |
| 266 | def test_all_miss_keeps_score_order_and_full_slots(self): |
| 267 | posts = [self._titled(i, f"Gemma thread {i}", score=1000 - i) for i in range(5)] |
| 268 | out = reddit_keyless._slot_priority("openclaw", posts) |
| 269 | assert out == posts # order unchanged |
| 270 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 271 | mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", |
| 272 | return_value={"top_comments": [], "comment_insights": [], |
| 273 | "num_comments": None}) as fc: |
| 274 | reddit_keyless.search_and_enrich( |
| 275 | "openclaw", "2026-05-01", "2026-05-31", depth="quick") |
| 276 | assert fc.call_count == reddit_keyless.ENRICH_LIMITS["quick"] |
| 277 | |
| 278 | def test_same_tier_order_preserved(self): |
| 279 | posts = [self._titled(i, f"openclaw thread {i}", score=100 - i) for i in range(4)] |
| 280 | out = reddit_keyless._slot_priority("openclaw", posts) |
| 281 | assert out == posts |
| 282 | |
| 283 | def test_empty_entity_falls_back_to_token_overlap(self): |
| 284 | # Pure intent-modifier topic yields no primary entity; fallback path |
| 285 | # must not raise and must keep every post. |
| 286 | posts = [self._titled(1, "Post one"), self._titled(2, "review of things")] |
| 287 | out = reddit_keyless._slot_priority("review", posts) |
| 288 | assert len(out) == 2 |
| 289 | assert {p["url"] for p in out} == {p["url"] for p in posts} |
| 290 | |
| 291 | def test_selftext_match_lands_in_match_tier(self): |
| 292 | body_match = self._titled(1, "Need help with my setup", score=2, |
| 293 | selftext="my openclaw agent keeps asking for ssh keys") |
| 294 | off_topic = self._titled(2, "Gemma 4 with QAT", score=700) |
| 295 | out = reddit_keyless._slot_priority("openclaw", [off_topic, body_match]) |
| 296 | assert out[0] is body_match |
| 297 | |
| 298 | def test_none_score_posts_do_not_break_partition(self): |
| 299 | p1 = self._titled(1, "openclaw tips") |
| 300 | p1["engagement"]["score"] = None |
| 301 | p2 = self._titled(2, "Gemma news") |
| 302 | p2["engagement"]["score"] = None |
| 303 | out = reddit_keyless._slot_priority("openclaw", [p2, p1]) |
| 304 | assert out[0] is p1 |
| 305 | |
| 306 | def test_partition_never_raises(self): |
| 307 | posts = [self._titled(1, "openclaw tips", score=1)] |
| 308 | with mock.patch("lib.rerank._primary_entity", side_effect=Exception("boom")): |
| 309 | out = reddit_keyless._slot_priority("openclaw", posts) |
| 310 | assert out == posts |
| 311 | |
| 312 | @staticmethod |
| 313 | def _titled_nc(i, title, score=0, ncmt=0, selftext=""): |
| 314 | """_titled variant that also sets a real comment count (both surfaces).""" |
| 315 | p = TestSlotPriority._titled(i, title, score=score, selftext=selftext) |
| 316 | p["num_comments"] = ncmt |
| 317 | p["engagement"]["num_comments"] = ncmt |
| 318 | return p |
| 319 | |
| 320 | def test_comment_count_orders_within_match_tier(self): |
| 321 | # Two entity-matching posts: the low-score high-comment thread wins the slot. |
| 322 | high_comments = self._titled_nc(1, "openclaw thread with lots of discussion", score=1, ncmt=45) |
| 323 | low_comments = self._titled_nc(2, "openclaw thread, quiet", score=900, ncmt=3) |
| 324 | out = reddit_keyless._slot_priority("openclaw", [low_comments, high_comments]) |
| 325 | assert out[0] is high_comments |
| 326 | assert out[1] is low_comments |
| 327 | |
| 328 | def test_entity_match_tier_beats_comment_count(self): |
| 329 | # Entity priority is preserved: a miss with 100 comments still follows a |
| 330 | # match with 1 comment, regardless of discussion volume. |
| 331 | match = self._titled_nc(1, "openclaw tips", score=10, ncmt=1) |
| 332 | miss = self._titled_nc(2, "Gemma news", score=100, ncmt=100) |
| 333 | out = reddit_keyless._slot_priority("openclaw", [miss, match]) |
| 334 | assert out[0] is match |
| 335 | assert out[1] is miss |
| 336 | |
| 337 | def test_equal_comment_counts_preserve_incoming_order_stable(self): |
| 338 | # Stable tiebreak: equal comment counts preserve the incoming order. The |
| 339 | # score-first order is established by search_and_enrich's provisional |
| 340 | # sort before _slot_priority runs; _slot_priority must not re-sort ties. |
| 341 | p1 = self._titled_nc(1, "openclaw thread a", score=100, ncmt=5) |
| 342 | p2 = self._titled_nc(2, "openclaw thread b", score=50, ncmt=5) |
| 343 | out = reddit_keyless._slot_priority("openclaw", [p2, p1]) |
| 344 | assert out[0] is p2 |
| 345 | assert out[1] is p1 |
| 346 | |
| 347 | def test_unknown_comment_count_ties_with_zero(self): |
| 348 | # Missing/None comment count is treated as 0: it ties with a known-zero |
| 349 | # post (stable) and sorts below any positive-count post in its tier. |
| 350 | unknown = self._titled_nc(1, "openclaw unknown", score=100, ncmt=None) |
| 351 | positive = self._titled_nc(2, "openclaw positive", score=10, ncmt=3) |
| 352 | known_zero = self._titled_nc(3, "openclaw zero", score=5, ncmt=0) |
| 353 | out = reddit_keyless._slot_priority("openclaw", [known_zero, unknown, positive]) |
| 354 | assert out[0] is positive |
| 355 | assert out[1:] == [known_zero, unknown] |
| 356 | |
| 357 | def test_richest_thread_gets_slot_when_score_ranked_low(self): |
| 358 | # Issue #906 regression: a 45-comment thread ranked last by score must |
| 359 | # get an enrichment slot at default depth (limit 8) while a 4-comment |
| 360 | # thread above it in score order does not. All posts are in the same |
| 361 | # entity tier; there are more posts than slots so ordering matters. |
| 362 | posts = [ |
| 363 | self._titled_nc(1, "openclaw thread one", score=1000, ncmt=4), |
| 364 | self._titled_nc(2, "openclaw thread two", score=900, ncmt=4), |
| 365 | self._titled_nc(3, "openclaw thread three", score=800, ncmt=4), |
| 366 | self._titled_nc(4, "openclaw thread four", score=700, ncmt=4), |
| 367 | self._titled_nc(5, "openclaw thread five", score=600, ncmt=6), |
| 368 | self._titled_nc(6, "openclaw thread six", score=500, ncmt=5), |
| 369 | self._titled_nc(7, "openclaw thread seven", score=300, ncmt=4), |
| 370 | self._titled_nc(9, "openclaw thread nine", score=250, ncmt=7), |
| 371 | self._titled_nc(10, "openclaw thread ten", score=200, ncmt=8), |
| 372 | self._titled_nc(11, "openclaw thread eleven", score=150, ncmt=9), |
| 373 | self._titled_nc(8, "openclaw thread eight", score=77, ncmt=45), |
| 374 | ] |
| 375 | enriched_urls = [] |
| 376 | |
| 377 | def _capture(url): |
| 378 | enriched_urls.append(url) |
| 379 | return {"top_comments": [], "comment_insights": [], "num_comments": None} |
| 380 | |
| 381 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 382 | mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", |
| 383 | side_effect=_capture): |
| 384 | reddit_keyless.search_and_enrich( |
| 385 | "openclaw", "2026-05-01", "2026-05-31", depth="default") |
| 386 | assert posts[10]["url"] in enriched_urls # 45-comment thread enriched |
| 387 | assert posts[6]["url"] not in enriched_urls # 4-comment thread above it skipped |
| 388 | assert len(enriched_urls) == reddit_keyless.ENRICH_LIMITS["default"] |
| 389 | |
| 390 | def test_miss_tier_orders_by_comments_for_leftover_slots(self): |
| 391 | # Review finding #1 (validated): when the entity-match tier is smaller |
| 392 | # than ENRICH_LIMITS, leftover slots are filled from the miss tier in |
| 393 | # comment-count order. 1 match + 4 misses at quick depth (limit 4): the |
| 394 | # three most-commented misses get slots, the least-commented miss does not. |
| 395 | # Score order deliberately differs from comment order so this test |
| 396 | # discriminates the miss-tier sort from the old score-first order. |
| 397 | posts = [ |
| 398 | self._titled_nc(1, "openclaw thread", score=100, ncmt=2), |
| 399 | self._titled_nc(2, "Gemma thread A", score=5, ncmt=30), |
| 400 | self._titled_nc(3, "Gemma thread B", score=40, ncmt=9), |
| 401 | self._titled_nc(4, "Gemma thread C", score=30, ncmt=2), |
| 402 | self._titled_nc(5, "Gemma thread D", score=20, ncmt=1), |
| 403 | ] |
| 404 | enriched_urls = [] |
| 405 | |
| 406 | def _capture(url): |
| 407 | enriched_urls.append(url) |
| 408 | return {"top_comments": [], "comment_insights": [], "num_comments": None} |
| 409 | |
| 410 | with mock.patch.object(reddit_keyless, "_discover", return_value=posts), \ |
| 411 | mock.patch.object(reddit_keyless.reddit_shreddit, "fetch_comments", |
| 412 | side_effect=_capture): |
| 413 | reddit_keyless.search_and_enrich( |
| 414 | "openclaw", "2026-05-01", "2026-05-31", depth="quick") |
| 415 | assert posts[0]["url"] in enriched_urls # entity match always slotted |
| 416 | assert posts[1]["url"] in enriched_urls # 30-comment miss (top miss) |
| 417 | assert posts[2]["url"] in enriched_urls # 9-comment miss |
| 418 | assert posts[3]["url"] in enriched_urls # 2-comment miss takes the last slot |
| 419 | assert posts[4]["url"] not in enriched_urls # 1-comment miss below the cut |
| 420 | assert len(enriched_urls) == reddit_keyless.ENRICH_LIMITS["quick"] |
| 421 | |
| 422 | |
| 423 | class TestScoredListingsFallback: |
| 424 | """_scored_listings falls back to the arctic-shift archive when the |
| 425 | shreddit listing partials return nothing (datacenter egress 403).""" |
| 426 | |
| 427 | def test_arctic_fallback_when_shreddit_empty(self): |
| 428 | arctic_post = _scored(1, score=406) |
| 429 | with mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", |
| 430 | return_value=[]), \ |
| 431 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_listings", |
| 432 | return_value=[arctic_post]) as arctic: |
| 433 | out = reddit_keyless._scored_listings(["tea"], depth="quick", query="matcha") |
| 434 | assert out == [arctic_post] |
| 435 | arctic.assert_called_once_with(["tea"], depth="quick", query="matcha", sorts=None) |
| 436 | |
| 437 | def test_shreddit_and_arctic_both_called_deduped(self): |
| 438 | """Shreddit and arctic are both called; arctic supplements missing posts.""" |
| 439 | shreddit_post = _scored(1, score=42) |
| 440 | shreddit_post["subreddit"] = "tea" |
| 441 | arctic_post = _scored(2, score=100) |
| 442 | arctic_post["subreddit"] = "tea" |
| 443 | with mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", |
| 444 | return_value=[shreddit_post]), \ |
| 445 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_listings", |
| 446 | return_value=[arctic_post]) as arctic: |
| 447 | out = reddit_keyless._scored_listings(["tea"], depth="quick", query="matcha") |
| 448 | # Both shreddit and arctic posts should be in the result (deduped by URL). |
| 449 | assert len(out) == 2 |
| 450 | urls = {p["url"] for p in out} |
| 451 | assert shreddit_post["url"] in urls |
| 452 | assert arctic_post["url"] in urls |
| 453 | arctic.assert_called_once() |
| 454 | |
| 455 | def test_both_empty_returns_empty(self): |
| 456 | with mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", |
| 457 | return_value=[]), \ |
| 458 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_listings", |
| 459 | return_value=[]): |
| 460 | out = reddit_keyless._scored_listings(["tea"], depth="quick", query="matcha") |
| 461 | assert out == [] |
| 462 | |
| 463 | def test_never_raises_when_arctic_fails(self): |
| 464 | with mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", |
| 465 | return_value=[]), \ |
| 466 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_listings", |
| 467 | side_effect=Exception("boom")): |
| 468 | out = reddit_keyless._scored_listings(["tea"], depth="quick", query="matcha") |
| 469 | assert out == [] |
| 470 | |
| 471 | def test_dedicated_sorts_passed_through(self): |
| 472 | with mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", |
| 473 | return_value=[]), \ |
| 474 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_listings", |
| 475 | return_value=[]) as arctic: |
| 476 | reddit_keyless._scored_listings( |
| 477 | ["Kanye"], depth="default", query="Kanye", sorts=["top", "hot", "new"] |
| 478 | ) |
| 479 | arctic.assert_called_once_with( |
| 480 | ["Kanye"], depth="default", query="Kanye", sorts=["top", "hot", "new"] |
| 481 | ) |
| 482 | |
| 483 | def test_arctic_supplements_all_subreddits(self): |
| 484 | """Arctic is called for all subreddits to supplement any failed sort lanes.""" |
| 485 | shreddit_post = _scored(1, score=100) |
| 486 | shreddit_post["subreddit"] = "tea" |
| 487 | arctic_post_tea = _scored(2, score=200) |
| 488 | arctic_post_tea["subreddit"] = "tea" |
| 489 | arctic_post_coffee = _scored(3, score=150) |
| 490 | arctic_post_coffee["subreddit"] = "coffee" |
| 491 | |
| 492 | def shreddit_side_effect(subs, **kwargs): |
| 493 | # Shreddit only returns posts for "tea", not "coffee". |
| 494 | return [shreddit_post] if "tea" in subs else [] |
| 495 | |
| 496 | with mock.patch.object(reddit_keyless.reddit_listing, "fetch_listings", |
| 497 | side_effect=shreddit_side_effect), \ |
| 498 | mock.patch.object(reddit_keyless.reddit_arctic, "fetch_listings", |
| 499 | return_value=[arctic_post_tea, arctic_post_coffee]) as arctic: |
| 500 | out = reddit_keyless._scored_listings( |
| 501 | ["tea", "coffee"], depth="quick", query="beverages" |
| 502 | ) |
| 503 | # Arctic is called for ALL requested subreddits to supplement any failed sorts. |
| 504 | arctic.assert_called_once() |
| 505 | call_args = arctic.call_args |
| 506 | assert set(call_args[0][0]) == {"tea", "coffee"}, "arctic should be called for all subs" |
| 507 | # All posts should be in the result (deduped by URL). |
| 508 | urls = [p["url"] for p in out] |
| 509 | assert shreddit_post["url"] in urls |
| 510 | assert arctic_post_tea["url"] in urls |
| 511 | assert arctic_post_coffee["url"] in urls |
| 512 |