返回 last30days-skill
bird_x.py
根目录 / skills / last30days / scripts / lib / bird_x.py
1 """Bird X search client for the v3.0.0 last30days pipeline.
2
3 Uses a vendored subset of @steipete/bird v0.8.0 (MIT License) to search X
4 via Twitter's GraphQL API. No external `bird` CLI binary needed - just Node.js.
5 See scripts/lib/vendor/bird-search/package.json for authoritative version.
6 """
7
8 import json
9 import os
10 import shutil
11 import sys
12 import time
13 from pathlib import Path
14
15 from . import env, health, http, log, subproc
16 from datetime import date, datetime, timedelta
17 from typing import Any, Dict, List, Optional, Tuple
18
19 from .relevance import token_overlap_relevance as _compute_relevance
20
21 # How many times to retry the bird-search subprocess when stdout is non-JSON
22 # (typically an HTML anti-bot interstitial from Twitter's edge).
23 MAX_JSON_DECODE_RETRIES = 2
24 JSON_DECODE_RETRY_DELAY = 5.0 # seconds between retry attempts
25
26
27 def _leading_mentions(text: str) -> list:
28 """Leading-run @mention parse, shared with other X-shaped sources (xquik).
29
30 Thin wrapper over ``query.leading_mentions`` so bird and xquik share one
31 implementation; kept here for existing call sites and tests.
32 """
33 from .query import leading_mentions
34 return leading_mentions(text)
35
36
37 def _first_of(*values):
38 """Return first value that is not None."""
39 for v in values:
40 if v is not None:
41 return v
42 return None
43
44 # Path to the vendored bird-search wrapper
45 _BIRD_SEARCH_MJS = Path(__file__).parent / "vendor" / "bird-search" / "bird-search.mjs"
46
47 # Depth configurations: number of results to request
48 DEPTH_CONFIG = {
49 "quick": 12,
50 "default": 30,
51 "deep": 60,
52 }
53
54 # Module-level credentials injected from .env config
55 _credentials: Dict[str, str] = {}
56
57 # The vendored bird-search client reads exactly this env surface, and the
58 # node subprocess it spawns needs the node-runtime env (platform, locale,
59 # TLS/proxy config) to run in every environment it runs in today. Ambient
60 # BIRD_* vars pass through as well. Everything else in os.environ - unrelated
61 # API keys, tokens, .env contents - must not reach the scan-excluded vendored
62 # client (issue #1063). Mirrors the platform-var surface grok_x keeps for its
63 # node child (grok_x._subprocess_env).
64 _SUBPROCESS_ENV_ALLOWLIST = (
65 # Runtime / platform vars the node subprocess needs (mirrors grok_x)
66 "PATH", "HOME", "LANG", "LC_ALL", "TMPDIR", "SystemRoot",
67 "USERPROFILE", "HOMEDRIVE", "HOMEPATH", "SystemDrive", "COMSPEC",
68 "PATHEXT", "TEMP", "TMP",
69 # Node TLS / proxy / CA config for custom-CA and proxied environments
70 "NODE_ENV", "NODE_OPTIONS", "NODE_EXTRA_CA_CERTS",
71 "NODE_TLS_REJECT_UNAUTHORIZED", "NODE_USE_ENV_PROXY",
72 "SSL_CERT_FILE", "SSL_CERT_DIR", "HTTP_PROXY", "HTTPS_PROXY", "NO_PROXY",
73 "http_proxy", "https_proxy", "no_proxy",
74 # X session cookies the vendored client reads from the environment
75 "AUTH_TOKEN", "CT0", "TWITTER_AUTH_TOKEN", "TWITTER_CT0",
76 # Browser-cookie disable flag the client reads (cookies.js envFlagEnabled)
77 "LAST30DAYS_DISABLE_BROWSER_COOKIES",
78 )
79
80
81 def set_credentials(auth_token: Optional[str], ct0: Optional[str]):
82 """Inject AUTH_TOKEN/CT0 from .env config so Node subprocesses can use them."""
83 if auth_token:
84 _credentials['AUTH_TOKEN'] = auth_token
85 if ct0:
86 _credentials['CT0'] = ct0
87
88
89 def _has_injected_credentials() -> bool:
90 """Return True when both X session cookies were injected from config."""
91 return bool(_credentials.get('AUTH_TOKEN') and _credentials.get('CT0'))
92
93
94 def _has_process_credentials() -> bool:
95 """Return True when AUTH_TOKEN/CT0 are present in process env."""
96 return bool(env.read_secret_env("AUTH_TOKEN") and env.read_secret_env("CT0"))
97
98
99 def _subprocess_env() -> Dict[str, str]:
100 """Build env dict for Node subprocesses, merging injected credentials.
101
102 The child env is limited to the vendored client's env surface (see
103 ``_SUBPROCESS_ENV_ALLOWLIST``) plus injected credentials, so unrelated
104 ambient secrets never reach scan-excluded vendored code (issue #1063).
105 The ambient-credential lane behaves exactly as before.
106 """
107 env = {
108 name: os.environ[name]
109 for name in _SUBPROCESS_ENV_ALLOWLIST
110 if name in os.environ
111 }
112 env.update({
113 key: value for key, value in os.environ.items()
114 if key.startswith("BIRD_")
115 })
116 env.update(_credentials)
117 # Hard-disable browser-cookie fallback so normal pipeline runs never hit
118 # Safari/Chrome Keychain prompts during source detection or search.
119 env["BIRD_DISABLE_BROWSER_COOKIES"] = "1"
120 return env
121
122
123 def _log(msg: str):
124 log.source_log("Bird", msg, tty_only=False)
125
126
127 def _scrub_credentials(text: str) -> str:
128 """Redact X session cookie values from subprocess output.
129
130 A failure reason built from bird-search's stderr reaches the run's
131 ``source_status`` detail, which is rendered in the report and returned by
132 ``--emit=json``. The vendored client receives AUTH_TOKEN/CT0 in its
133 environment, so an error message that echoes a rejected cookie would
134 otherwise carry it into user-facing output. Log lines get the same
135 treatment because stderr is captured in agent harnesses.
136 """
137 scrubbed = text
138 for name in ("AUTH_TOKEN", "CT0", "TWITTER_AUTH_TOKEN", "TWITTER_CT0"):
139 value = _credentials.get(name) or os.environ.get(name)
140 # Short values would match too much ordinary text to be worth it.
141 if value and len(value) >= 8:
142 scrubbed = scrubbed.replace(value, "<redacted>")
143 return scrubbed
144
145
146 def classify_run_failure(detail: str) -> str:
147 """Map Bird's subprocess-only failure shapes to run outcome states."""
148 text = detail.lower()
149 if any(marker in text for marker in ("interstitial", "non-json", "invalid json")):
150 return health.SCHEMA_DRIFT
151 if any(
152 marker in text
153 for marker in ("cookie expired", "expired cookie", "unauthorized", "forbidden", "login required")
154 ):
155 return health.AUTH_FAILED
156 return http.classify_failure(message=detail)
157
158
159 def _extract_core_subject(topic: str) -> str:
160 """Extract core subject from verbose query for X search.
161
162 X search is literal keyword AND matching — all words must appear.
163 Aggressively strip question/meta/research words to keep only the
164 core product/concept name (max 5 words).
165 """
166 from .query import extract_core_subject
167 return extract_core_subject(topic, max_words=5, strip_suffixes=True)
168
169
170 def _plain_query_tokens(text: str) -> list[str]:
171 """Return lexical tokens without Bird query grouping syntax.
172
173 Strips phrase quotes as well as grouping characters. Used where a flat
174 token list is wanted; use ``build_topic_query`` for the provider query,
175 which preserves quoted phrases.
176 """
177 separators = str.maketrans({char: " " for char in '\"“”()[]{}'})
178 return [
179 clean
180 for token in text.translate(separators).split()
181 if (clean := token.strip("'‘’"))
182 ]
183
184
185 # Bird/X grouping syntax that carries no lexical meaning. Double quotes are
186 # deliberately absent: X advanced search treats "..." as a phrase match, which
187 # is exactly what the planner intended when it quoted a proper noun.
188 _GROUPING_CHARS = "“”()[]{}"
189
190
191 def _date_filters(from_date: str, to_date: Optional[str] = None) -> str:
192 filters = f"since:{from_date}"
193 if to_date is not None:
194 end = date.fromisoformat(to_date)
195 if end == date.max:
196 # No supported post date lies beyond this inclusive upper bound.
197 return filters
198 # The research window includes to_date; X's until bound is exclusive.
199 until = end + timedelta(days=1)
200 filters += f" until:{until.isoformat()}"
201 return filters
202
203
204 def build_topic_query(
205 topic: str, from_date: str, to_date: Optional[str] = None
206 ) -> str:
207 """Build the X topic query, preserving quoted proper-noun phrases.
208
209 Previously the topic went through ``_plain_query_tokens``, which stripped
210 the quotes the planner had added, so an intended phrase match for
211 '"Peter Steinberger"' degraded into `peter AND steinberger` -- narrower and
212 noisier at once. X supports phrase queries natively, so the quotes are
213 passed through.
214 """
215 separators = str.maketrans({char: " " for char in _GROUPING_CHARS})
216 cleaned = topic.translate(separators)
217 # An unbalanced quote is worse than no quote: X reads the orphan as an
218 # unterminated phrase and matches nothing. Upstream trimming (core-subject
219 # extraction, retry shortening) can cut a topic mid-phrase, so verify the
220 # quotes pair up and fall back to bare tokens when they do not.
221 if cleaned.count('"') % 2:
222 cleaned = cleaned.replace('"', " ")
223 tokens = [
224 clean
225 for token in cleaned.split()
226 if (clean := token.strip("'‘’"))
227 ]
228 core = " ".join(tokens).strip()
229 return " ".join(part for part in (core, _date_filters(from_date, to_date)) if part)
230
231
232 def is_bird_installed() -> bool:
233 """Check if vendored Bird search module is available.
234
235 Returns:
236 True if bird-search.mjs exists and Node.js is in PATH.
237 """
238 if not _BIRD_SEARCH_MJS.exists():
239 return False
240 return shutil.which("node") is not None
241
242
243 def is_bird_authenticated() -> Optional[str]:
244 """Check if explicit X credentials are available.
245
246 Returns:
247 Auth source string if authenticated, None otherwise.
248 """
249 if not is_bird_installed():
250 return None
251
252 if _has_injected_credentials():
253 return "env AUTH_TOKEN"
254 if _has_process_credentials():
255 return "env AUTH_TOKEN"
256 return None
257
258
259 _probe_cache: Optional[Optional[bool]] = "unset" # "unset" | True | False | None
260
261
262 def probe_works(timeout: int = 8) -> Optional[bool]:
263 """Cheap runtime check that X auth actually returns data.
264
265 Returns True when a 1-result probe comes back without an error, False on a
266 clear failure (auth error / generic search failure), and None when the
267 result is inconclusive (network timeout) so callers can fail open and keep
268 the static credential-presence status rather than reporting a false-down.
269 Cached per process so repeated diagnose calls don't re-probe.
270 """
271 global _probe_cache
272 if _probe_cache != "unset":
273 return _probe_cache # type: ignore[return-value]
274 if not (_has_injected_credentials() or _has_process_credentials()):
275 _probe_cache = False
276 return False
277 from datetime import datetime, timedelta, timezone
278 since = (datetime.now(timezone.utc) - timedelta(days=30)).strftime("%Y-%m-%d")
279 # @x (the platform's own account) posts frequently, so a no-error response
280 # means auth works even if this particular window is quiet.
281 resp = _run_bird_search(f"from:x since:{since}", count=1, timeout=timeout)
282 if isinstance(resp, dict) and resp.get("error"):
283 err = str(resp.get("error")).lower()
284 if "timed out" in err or "timeout" in err:
285 _probe_cache = None # inconclusive — don't downgrade on a transient timeout
286 return None
287 _probe_cache = False
288 return False
289 _probe_cache = True
290 return True
291
292
293 def check_npm_available() -> bool:
294 """Check if npm is available (kept for API compatibility).
295
296 Returns:
297 True if 'npm' command is available in PATH, False otherwise.
298 """
299 return shutil.which("npm") is not None
300
301
302 def install_bird() -> Tuple[bool, str]:
303 """No-op. Bird search is vendored in v3.0.0, no installation needed.
304
305 Returns:
306 Tuple of (success, message).
307 """
308 if is_bird_installed():
309 return True, "Bird search is bundled with /last30days v3.0.0 - no installation needed."
310 if not shutil.which("node"):
311 return False, "Node.js 22+ is required for X search. Install Node.js first."
312 return False, f"Vendored bird-search.mjs not found at {_BIRD_SEARCH_MJS}"
313
314
315 def get_bird_status() -> Dict[str, Any]:
316 """Get comprehensive Bird search status.
317
318 Returns:
319 Dict with keys: installed, authenticated, username, can_install
320 """
321 installed = is_bird_installed()
322 auth_source = is_bird_authenticated() if installed else None
323
324 return {
325 "installed": installed,
326 "authenticated": auth_source is not None,
327 "username": auth_source, # Now returns auth source (e.g., "Safari", "env AUTH_TOKEN")
328 "can_install": True, # Always vendored in v3.0.0
329 }
330
331
332 def _invoke_bird_subprocess(query: str, count: int, timeout: int):
333 """Invoke the vendored bird-search.mjs subprocess once.
334
335 Returns (result, error_dict). If error_dict is non-None, treat it as the
336 final result and do not retry — those errors are terminal (timeout,
337 spawn failure). If error_dict is None, the subprocess ran to completion
338 and `result` is the SubprocResult; the caller decides whether to retry
339 based on the result.stdout content.
340 """
341 cmd = [
342 "node", str(_BIRD_SEARCH_MJS),
343 query,
344 "--count", str(count),
345 "--json",
346 ]
347
348 try:
349 result = subproc.run_with_timeout(
350 cmd,
351 timeout=timeout,
352 env=_subprocess_env(),
353 )
354 except subproc.SubprocTimeout:
355 return None, {"error": f"Search timed out after {timeout}s", "items": []}
356 except Exception as e:
357 return None, {"error": str(e), "items": []}
358
359 return result, None
360
361
362 def _run_bird_search(query: str, count: int, timeout: int) -> Dict[str, Any]:
363 """Run a search using the vendored bird-search.mjs module.
364
365 Retries the subprocess on JSON-decode failure (typically a Twitter
366 anti-bot HTML interstitial in stdout) up to MAX_JSON_DECODE_RETRIES
367 times with JSON_DECODE_RETRY_DELAY seconds between attempts. Terminal
368 errors (subprocess timeout, non-zero return code) are returned
369 immediately without retry.
370
371 Args:
372 query: Full search query string (including since: filter)
373 count: Number of results to request
374 timeout: Timeout in seconds (per attempt)
375
376 Returns:
377 Raw Bird JSON response or error dict.
378 """
379 last_decode_error: Optional[str] = None
380
381 for attempt in range(MAX_JSON_DECODE_RETRIES):
382 result, terminal_error = _invoke_bird_subprocess(query, count, timeout)
383 if terminal_error is not None:
384 return terminal_error
385
386 output = result.stdout.strip()
387 if result.returncode != 0:
388 if not output:
389 error = result.stderr.strip() or "Bird search failed"
390 return {"error": error, "items": []}
391 # Windows/Node 24: the vendored Bird CLI uses native fetch (undici),
392 # and calling process.exit() while keep-alive sockets are still
393 # closing trips a libuv assertion -> non-zero exit code AFTER it has
394 # already written a complete, valid JSON result to stdout. Trust
395 # stdout when it has content; only treat a non-zero exit as a real
396 # failure when stdout is empty.
397
398 if not output:
399 return {"items": []}
400
401 try:
402 parsed = json.loads(output)
403 except json.JSONDecodeError as e:
404 # Twitter's edge sometimes serves an HTML anti-bot interstitial
405 # in place of JSON. Tag the failure shape so it's distinguishable
406 # from "no results" in logs, then retry the subprocess.
407 looks_html = output.lstrip().lower().startswith(("<!doctype", "<html", "<"))
408 attempt_num = attempt + 1
409 log_msg = (
410 f"Bird search returned non-JSON stdout "
411 f"(looks_html={looks_html}, attempt {attempt_num}/{MAX_JSON_DECODE_RETRIES}, "
412 f"first 80 chars: {output[:80]!r})"
413 )
414 last_decode_error = str(e)
415 if attempt_num < MAX_JSON_DECODE_RETRIES:
416 log.source_log(
417 "X/bird",
418 f"{log_msg}; retrying in {JSON_DECODE_RETRY_DELAY:.0f}s",
419 tty_only=False,
420 )
421 time.sleep(JSON_DECODE_RETRY_DELAY)
422 continue
423 log.source_log("X/bird", log_msg, tty_only=False)
424 return {
425 "error": (
426 f"Invalid JSON response after {MAX_JSON_DECODE_RETRIES} attempts "
427 f"(likely Twitter anti-bot interstitial): {e}"
428 ),
429 "items": [],
430 }
431
432 if isinstance(parsed, list):
433 return {"items": parsed}
434 return parsed
435
436 # Defensive fallthrough — loop should always return above.
437 return {
438 "error": f"Bird search exhausted retries: {last_decode_error}",
439 "items": [],
440 }
441
442
443 def search_x(
444 topic: str,
445 from_date: str,
446 to_date: str,
447 depth: str = "default",
448 ) -> Dict[str, Any]:
449 """Search X using Bird CLI with automatic retry on 0 results.
450
451 Args:
452 topic: Search topic
453 from_date: Start date (YYYY-MM-DD)
454 to_date: Inclusive end date (YYYY-MM-DD)
455 depth: Research depth - "quick", "default", or "deep"
456
457 Returns:
458 Raw Bird JSON response or error dict.
459 """
460 count = DEPTH_CONFIG.get(depth, DEPTH_CONFIG["default"])
461 timeout = 30 if depth == "quick" else 45 if depth == "default" else 60
462
463 # Extract core subject - X search is literal, not semantic
464 core_subject = _extract_core_subject(topic)
465 core_words = _plain_query_tokens(core_subject)
466 core_topic = " ".join(core_words)
467 query = build_topic_query(core_subject, from_date, to_date)
468 date_filters = _date_filters(from_date, to_date)
469
470 _log(f"Searching: {query}")
471 response = _run_bird_search(query, count, timeout)
472 last_clean_response = response if not response.get("error") else None
473
474 # Check if we got results
475 items = parse_bird_response(response, query=core_topic)
476
477 # Retry with OR groups for multi-word queries (X supports OR operator)
478 if not items and len(core_words) >= 2:
479 from .query import extract_compound_terms
480 compounds = extract_compound_terms(topic)
481 if compounds:
482 # Build OR-group query: ("multi-agent" OR "agent simulation") since:DATE
483 or_parts = ' OR '.join(f'"{t}"' for t in compounds[:3])
484 _log(f"0 results for '{core_topic}', retrying with OR groups: {or_parts}")
485 query = f"({or_parts}) {date_filters}"
486 response = _run_bird_search(query, count, timeout)
487 if not response.get("error"):
488 last_clean_response = response
489 items = parse_bird_response(response, query=core_topic)
490
491 # Retry with fewer keywords if still 0 results and query has 3+ words
492 if not items and len(core_words) > 2:
493 shorter = ' '.join(core_words[:2])
494 _log(f"0 results for '{core_topic}', retrying with '{shorter}'")
495 query = f"{shorter} {date_filters}"
496 response = _run_bird_search(query, count, timeout)
497 if not response.get("error"):
498 last_clean_response = response
499 items = parse_bird_response(response, query=core_topic)
500
501 # Last-chance retry: use strongest remaining token (often the product name)
502 if not items and core_words:
503 low_signal = {
504 'trendiest', 'trending', 'hottest', 'hot', 'popular', 'viral',
505 'best', 'top', 'latest', 'new', 'plugin', 'plugins',
506 'skill', 'skills', 'tool', 'tools',
507 }
508 candidates = [w for w in core_words if w not in low_signal]
509 if candidates:
510 # Keep an entity anchor (the first distinctive topic token) in the
511 # retry so it can't collapse to a bare generic token like "compound"
512 # and flood the X pool with off-topic noise. Add the strongest
513 # (longest) distinctive token when it differs from the anchor;
514 # otherwise query the anchor alone. Better to return 0 than to
515 # over-broaden to an unanchored generic term.
516 anchor = candidates[0]
517 strongest = max(candidates, key=len)
518 retry_terms = anchor if strongest == anchor else f"{anchor} {strongest}"
519 _log(f"0 results for '{core_topic}', retrying anchored on '{retry_terms}'")
520 query = f"{retry_terms} {date_filters}"
521 response = _run_bird_search(query, count, timeout)
522 if not response.get("error"):
523 last_clean_response = response
524
525 if response.get("error") and last_clean_response is not None:
526 _log("Optional retry failed after a clean empty response; preserving no-results outcome")
527 return last_clean_response
528 return response
529
530
531 def search_handles(
532 handles: List[str],
533 topic: Optional[str],
534 from_date: str,
535 count_per: int = 5,
536 failure_out: Optional[List[str]] = None,
537 *,
538 to_date: Optional[str] = None,
539 ) -> List[Dict[str, Any]]:
540 """Search specific X handles for topic-related content.
541
542 Pulls each handle's actual timeline via `from:handle since:` — the FROM
543 lane (tweets BY the person), engagement-weighted downstream. The topic is
544 used for relevance RANKING, never AND'd into the query: X search is literal,
545 so `from:handle <their name>` only matched tweets where they wrote their own
546 name and returned ~0. Used in Phase 2 after entity extraction.
547
548 Args:
549 handles: List of X handles to search (without @)
550 topic: Search topic — used for relevance ranking only, not the query
551 from_date: Start date (YYYY-MM-DD)
552 count_per: Results to request per handle
553 failure_out: When provided, a short reason is appended for every
554 per-handle failure branch, so the caller can distinguish a
555 transport failure from a handle that genuinely posted nothing.
556 to_date: Inclusive end date (YYYY-MM-DD), when supplied
557
558 Returns:
559 List of raw item dicts (same format as parse_bird_response output).
560 """
561 core_topic = _extract_core_subject(topic) if topic else None
562 date_filters = _date_filters(from_date, to_date)
563
564 def _note(msg: str) -> None:
565 if failure_out is not None:
566 failure_out.append(msg)
567
568 def _search_one_handle(handle: str) -> List[Dict[str, Any]]:
569 handle = handle.lstrip("@")
570 # Always unfiltered: pull the timeline, rank by topic relevance below.
571 query = f"from:{handle} {date_filters}"
572
573 cmd = [
574 "node", str(_BIRD_SEARCH_MJS),
575 query,
576 "--count", str(count_per),
577 "--json",
578 ]
579
580 try:
581 result = subproc.run_with_timeout(cmd, timeout=15, env=_subprocess_env())
582 except subproc.SubprocTimeout:
583 _log(f"Handle search timed out for @{handle}")
584 _note(f"@{handle}: bird-search timed out after 15s")
585 return []
586 except OSError as e:
587 _log(f"Handle search error for @{handle}: {e}")
588 _note(f"@{handle}: could not spawn bird-search ({e})")
589 return []
590
591 output = result.stdout.strip()
592 if result.returncode != 0:
593 if not output:
594 _log(f"Handle search failed for @{handle}: {_scrub_credentials(result.stderr.strip())}")
595 _note(
596 f"@{handle}: bird-search exited {result.returncode} "
597 f"({_scrub_credentials(result.stderr.strip())[:160] or 'no stderr'})"
598 )
599 return []
600 # Windows/Node 24: benign libuv assertion can cause non-zero exit
601 # AFTER valid JSON is written to stdout. Trust stdout content.
602
603 if not output:
604 return []
605
606 try:
607 response = json.loads(output)
608 except json.JSONDecodeError:
609 _log(f"Invalid JSON from handle search for @{handle}")
610 _note(f"@{handle}: bird-search returned invalid JSON")
611 return []
612 items = parse_bird_response(response, query=core_topic)
613 # Log on success/empty too (not only on failure): a silent handle search
614 # made the from: query look like it never ran and caused wrong diagnoses.
615 _log(f"Searching: {query} -> {len(items)} results")
616 return items
617
618 from concurrent.futures import ThreadPoolExecutor, as_completed
619
620 all_items: List[Dict[str, Any]] = []
621 with ThreadPoolExecutor(max_workers=min(5, len(handles))) as executor:
622 futures = {executor.submit(_search_one_handle, h): h for h in handles}
623 for future in as_completed(futures):
624 all_items.extend(future.result())
625
626 return all_items
627
628
629 def search_mentions(
630 handles: List[str],
631 from_date: str,
632 count_per: int = 5,
633 failure_out: Optional[List[str]] = None,
634 *,
635 to_date: Optional[str] = None,
636 ) -> List[Dict[str, Any]]:
637 """Search for tweets ABOUT/TO each handle — the mention lane.
638
639 Queries `@handle since:` (tweets that mention the account) and excludes the
640 handle's OWN tweets (those belong to the FROM lane via search_handles), so
641 this surfaces what OTHERS are saying about the person. Engagement-weighted
642 downstream; deduped against the FROM lane by URL at normalize time.
643
644 Args:
645 handles: List of X handles (without @)
646 from_date: Start date (YYYY-MM-DD)
647 count_per: Results to request per handle
648 failure_out: When provided, a short reason is appended for every
649 per-handle failure branch, so the caller can distinguish a
650 transport failure from a handle nobody mentioned.
651 to_date: Inclusive end date (YYYY-MM-DD), when supplied
652
653 Returns:
654 List of raw item dicts (same format as parse_bird_response output).
655 """
656 date_filters = _date_filters(from_date, to_date)
657
658 def _note(msg: str) -> None:
659 if failure_out is not None:
660 failure_out.append(msg)
661
662 def _search_one(handle: str) -> List[Dict[str, Any]]:
663 handle = handle.lstrip("@")
664 query = f"@{handle} {date_filters}"
665 cmd = [
666 "node", str(_BIRD_SEARCH_MJS),
667 query,
668 "--count", str(count_per),
669 "--json",
670 ]
671 try:
672 result = subproc.run_with_timeout(cmd, timeout=15, env=_subprocess_env())
673 except subproc.SubprocTimeout:
674 _log(f"Mention search timed out for @{handle}")
675 _note(f"@{handle}: bird-search timed out after 15s")
676 return []
677 except OSError as e:
678 _log(f"Mention search error for @{handle}: {e}")
679 _note(f"@{handle}: could not spawn bird-search ({e})")
680 return []
681 if result.returncode != 0:
682 _log(f"Mention search failed for @{handle}: {_scrub_credentials(result.stderr.strip())}")
683 _note(
684 f"@{handle}: bird-search exited {result.returncode} "
685 f"({_scrub_credentials(result.stderr.strip())[:160] or 'no stderr'})"
686 )
687 return []
688 output = result.stdout.strip()
689 if not output:
690 return []
691 try:
692 response = json.loads(output)
693 except json.JSONDecodeError:
694 _log(f"Invalid JSON from mention search for @{handle}")
695 _note(f"@{handle}: bird-search returned invalid JSON")
696 return []
697 items = parse_bird_response(response, query=None)
698 # ABOUT lane = OTHERS mentioning the handle. Drop the handle's own tweets
699 # (the FROM lane already covers those); identify by the status URL author.
700 hl = handle.lower()
701 # The Bird API may return either x.com or twitter.com permalinks, so
702 # match both when excluding the handle's own tweets.
703 def _is_own(url):
704 u = (url or "").lower()
705 return f"x.com/{hl}/status" in u or f"twitter.com/{hl}/status" in u
706 about = [it for it in items if not _is_own(it.get("url"))]
707 _log(f"Searching: {query} -> {len(about)} mentions")
708 return about
709
710 from concurrent.futures import ThreadPoolExecutor, as_completed
711
712 all_items: List[Dict[str, Any]] = []
713 with ThreadPoolExecutor(max_workers=min(5, len(handles))) as executor:
714 futures = {executor.submit(_search_one, h): h for h in handles}
715 for future in as_completed(futures):
716 all_items.extend(future.result())
717 return all_items
718
719
720 def parse_bird_response(response: Dict[str, Any], query: str = "") -> List[Dict[str, Any]]:
721 """Parse Bird response to match xai_x output format.
722
723 Args:
724 response: Raw Bird JSON response
725 query: Original search query for relevance scoring
726
727 Returns:
728 List of normalized item dicts matching xai_x.parse_x_response() format.
729 """
730 items = []
731
732 # Check for errors
733 if "error" in response and response["error"]:
734 _log(f"Bird error: {response['error']}")
735 return items
736
737 # Bird returns a list of tweets directly or under a key
738 raw_items = response if isinstance(response, list) else response.get("items", response.get("tweets", []))
739
740 if not isinstance(raw_items, list):
741 return items
742
743 for i, tweet in enumerate(raw_items):
744 if not isinstance(tweet, dict):
745 continue
746
747 # Extract URL - Bird uses permanent_url or we construct from id
748 url = tweet.get("permanent_url") or tweet.get("url", "")
749 if not url and tweet.get("id"):
750 # Try different field structures Bird might use
751 author = tweet.get("author", {}) or tweet.get("user", {})
752 screen_name = author.get("username") or author.get("screen_name", "")
753 if screen_name:
754 url = f"https://x.com/{screen_name}/status/{tweet['id']}"
755
756 if not url:
757 continue
758
759 # Parse date from created_at/createdAt (e.g., "Wed Jan 15 14:30:00 +0000 2026")
760 date = None
761 created_at = tweet.get("createdAt") or tweet.get("created_at", "")
762 if created_at:
763 try:
764 # Try ISO format first (e.g., "2026-02-03T22:33:32Z")
765 # Check for ISO date separator, not just "T" (which appears in "Tue")
766 if len(created_at) > 10 and created_at[10] == "T":
767 dt = datetime.fromisoformat(created_at.replace("Z", "+00:00"))
768 else:
769 # Twitter format: "Wed Jan 15 14:30:00 +0000 2026"
770 dt = datetime.strptime(created_at, "%a %b %d %H:%M:%S %z %Y")
771 date = dt.strftime("%Y-%m-%d")
772 except (ValueError, TypeError):
773 pass
774
775 # Extract user info (Bird uses author.username, older format uses user.screen_name)
776 author = tweet.get("author", {}) or tweet.get("user", {})
777 author_handle = author.get("username") or author.get("screen_name", "") or tweet.get("author_handle", "")
778
779 # Build engagement dict (Bird uses camelCase: likeCount, retweetCount, etc.)
780 engagement = {
781 "likes": _first_of(tweet.get("likeCount"), tweet.get("like_count"), tweet.get("favorite_count")),
782 "reposts": _first_of(tweet.get("retweetCount"), tweet.get("retweet_count")),
783 "replies": _first_of(tweet.get("replyCount"), tweet.get("reply_count")),
784 "quotes": _first_of(tweet.get("quoteCount"), tweet.get("quote_count")),
785 }
786 # Convert to int where possible
787 for key in engagement:
788 if engagement[key] is not None:
789 try:
790 engagement[key] = int(engagement[key])
791 except (ValueError, TypeError):
792 engagement[key] = None
793
794 # Build normalized item
795 text = str(tweet.get("text", tweet.get("full_text", ""))).strip()[:500]
796 item = {
797 "id": f"X{i+1}",
798 "text": text,
799 "url": url,
800 "author_handle": author_handle.lstrip("@"),
801 # Leading @mentions parsed from the post text identify who a reply is
802 # directed at (X replies open with the target handle(s)). Used by the
803 # interaction-signal classifier in rerank.
804 "mentioned_handles": _leading_mentions(text),
805 "date": date,
806 "engagement": engagement if any(v is not None for v in engagement.values()) else None,
807 "why_relevant": "", # Bird doesn't provide relevance explanations
808 "relevance": _compute_relevance(query, str(tweet.get("text", ""))) if query else 0.7,
809 }
810
811 items.append(item)
812
813 return items
814
814 lines PYTHON