返回 last30days-skill
xquik.py
根目录 / skills / last30days / scripts / lib / xquik.py
1 """Xquik X search source for the v3.0.0 last30days pipeline.
2
3 Uses the Xquik REST API (https://xquik.com/api/v1) to search X/Twitter
4 with full engagement metrics (likes, retweets, replies, quotes, views,
5 bookmarks). Requires an API key from xquik.com.
6 """
7
8 from __future__ import annotations
9
10 from datetime import datetime
11 from typing import Any, Dict, List, Optional
12
13 from . import http, log
14 from .relevance import token_overlap_relevance as _compute_relevance
15 from .x_api import is_own_post
16
17 # Per-process probe cache: (state, reason). state is "unset" until probed, then
18 # True (funded/working) | False (auth/payment failure) | None (inconclusive).
19 _probe_cache: tuple = ("unset", "")
20
21 # Depth configurations: number of results to request per query
22 DEPTH_CONFIG = {
23 "quick": {"limit": 10, "queries": 1},
24 "default": {"limit": 20, "queries": 2},
25 "deep": {"limit": 40, "queries": 3},
26 }
27
28 _BASE_URL = "https://xquik.com/api/v1"
29
30
31 def _log(msg: str):
32 log.source_log("Xquik", msg, tty_only=False)
33
34
35 def _extract_core_subject(topic: str) -> str:
36 """Extract core subject for X search queries."""
37 from .query import extract_core_subject
38 return extract_core_subject(topic, max_words=5, strip_suffixes=True)
39
40
41 def expand_xquik_queries(topic: str, depth: str) -> List[str]:
42 """Generate query variants based on depth.
43
44 Args:
45 topic: Research topic
46 depth: "quick", "default", or "deep"
47
48 Returns:
49 List of query strings (1 for quick, 2 for default, 3 for deep).
50 """
51 core = _extract_core_subject(topic)
52 # Anti-bare-generic guard (#607): never let the core collapse to a single
53 # bare token when the topic carries more — a lone generic word floods X with
54 # off-topic collisions. Fall back to the full multi-word topic as the anchor.
55 topic_clean = topic.strip()
56 if len(core.split()) <= 1 and len(topic_clean.split()) > 1 and core.lower() != topic_clean.lower():
57 core = topic_clean
58 queries = [core]
59
60 # Add original topic if meaningfully different
61 if topic.lower().strip() != core.lower().strip():
62 queries.append(topic.strip())
63
64 # Add compound term variant for deep searches
65 if len(queries) < 3:
66 from .query import extract_compound_terms
67 compounds = extract_compound_terms(topic)
68 if compounds:
69 or_parts = " OR ".join(f'"{t}"' for t in compounds[:3])
70 queries.append(f"({or_parts})")
71
72 cap = DEPTH_CONFIG.get(depth, DEPTH_CONFIG["default"])["queries"]
73 return queries[:cap]
74
75
76 def search_xquik(
77 topic: str,
78 from_date: str,
79 to_date: str,
80 depth: str = "default",
81 token: str = "",
82 ) -> Dict[str, Any]:
83 """Search X via Xquik REST API.
84
85 Args:
86 topic: Search topic
87 from_date: Start date (YYYY-MM-DD)
88 to_date: End date (YYYY-MM-DD)
89 depth: Research depth - "quick", "default", or "deep"
90 token: Xquik API key
91
92 Returns:
93 Dict with "items" list and optional "error" string.
94 """
95 if not token:
96 return {"items": [], "error": "No XQUIK_API_KEY configured"}
97
98 cfg = DEPTH_CONFIG.get(depth, DEPTH_CONFIG["default"])
99 queries = expand_xquik_queries(topic, depth)
100 all_items: List[Dict[str, Any]] = []
101 seen_ids: set[str] = set()
102
103 for query_text in queries:
104 q = f"{query_text} since:{from_date} until:{to_date}"
105 items, auth_error = _execute_search(
106 q, cfg["limit"], token,
107 label=query_text, id_prefix="XQ",
108 seen_ids=seen_ids, relevance_query=query_text,
109 index_offset=len(all_items),
110 )
111 if auth_error:
112 # Auth/payment failure is fatal for the whole source (e.g. 401/403,
113 # and 402-unpaid surfaced via U5 diagnose) — return it so the caller
114 # settles honestly instead of silently empty.
115 return {"items": [], "error": auth_error}
116 all_items.extend(items)
117
118 return {"items": all_items}
119
120
121 def _execute_search(
122 q: str,
123 limit: int,
124 token: str,
125 *,
126 label: str,
127 id_prefix: str,
128 seen_ids: set[str],
129 relevance_query: str,
130 index_offset: int = 0,
131 failure_out: Optional[List[str]] = None,
132 ) -> tuple[List[Dict[str, Any]], str | None]:
133 """Run one Xquik search call and parse its tweets.
134
135 Returns ``(items, auth_error)``. ``auth_error`` is a non-empty string only
136 on a fatal auth/payment failure (401/403); transient/HTTP errors log and
137 return ``([], None)`` so one bad lane never discards another's results.
138
139 Non-fatal failures (429, 5xx, network error, unparseable payload) append a
140 reason to ``failure_out`` instead. They stay out of the return value
141 because the caller must keep going to the next handle, but a caller that
142 stopped there would report the lane as empty rather than as failed.
143 ``relevance_query`` (the topic) is what items are scored against — for the
144 handle lanes that differs from the search query (``from:handle``).
145 ``index_offset`` keeps item ids unique across multiple calls that share an
146 accumulator (multi-query topic search, per-handle lanes).
147 """
148 full_url = f"{_BASE_URL}/x/tweets/search?q={_url_encode(q)}&queryType=Top&limit={limit}"
149 _log(f"Searching: {label}")
150 try:
151 request_headers = {"X-Api-Key": token}
152 response = http.get(full_url, headers=request_headers, timeout=30, retries=2)
153 except http.HTTPError as exc:
154 status = getattr(exc, "status_code", None)
155 if status == 402:
156 # Unpaid key — fatal for the source, and surfaced on the real search
157 # path (not just --diagnose) so a live run reports it instead of
158 # settling silently empty. The X retrieval branch classifies by
159 # message text only, so the detail carries the "payment required"
160 # marker that http.classify_failure maps to PAYMENT_REQUIRED.
161 return [], "Xquik key unpaid: payment required (402)"
162 if status in (401, 403):
163 return [], f"Xquik auth failed ({status})"
164 _log(f"HTTP error for '{label}': {exc}")
165 if failure_out is not None:
166 failure_out.append(f"Xquik HTTP error for '{label}': {exc}")
167 return [], None
168 except Exception as exc:
169 _log(f"Error for '{label}': {exc}")
170 if failure_out is not None:
171 failure_out.append(f"Xquik request failed for '{label}': {exc}")
172 return [], None
173
174 tweets = response.get("tweets", [])
175 if not isinstance(tweets, list):
176 if failure_out is not None:
177 failure_out.append(
178 f"Xquik returned no tweets array for '{label}' "
179 f"(got {type(tweets).__name__})"
180 )
181 return [], None
182 items: List[Dict[str, Any]] = []
183 for tweet in tweets:
184 if not isinstance(tweet, dict):
185 continue
186 tweet_id = str(tweet.get("id", ""))
187 if tweet_id in seen_ids:
188 continue
189 seen_ids.add(tweet_id)
190 item = _parse_tweet(tweet, index_offset + len(items), relevance_query, id_prefix=id_prefix)
191 if item:
192 items.append(item)
193 return items, None
194
195
196 # True when a tweet URL is authored by ``handle`` (their own post). Used by
197 # the ABOUT lane to drop the subject's own tweets so only mentions *by
198 # others* remain; the implementation is ``x_api.is_own_post``.
199 _is_own = is_own_post
200
201
202 def search_handles(
203 handles: List[str],
204 topic: str,
205 from_date: str,
206 to_date: str,
207 *,
208 count_per: int = 8,
209 token: str = "",
210 failure_out: Optional[List[str]] = None,
211 ) -> List[Dict[str, Any]]:
212 """FROM lane: tweets authored BY each handle (their own timeline).
213
214 The topic is NOT AND'd into the query (that was the from:-AND bug, #610) —
215 we pull the raw timeline and use ``topic`` for relevance ranking only.
216 Returns a flat list of item dicts (mirrors ``bird_x.search_handles``).
217
218 When ``failure_out`` is provided, any failure reason is appended — the
219 fatal auth/payment one that stops the lane, and the per-handle transient
220 ones (429, 5xx, network, unparseable payload) that do not — so the caller
221 can distinguish a failed request from a handle that posted nothing.
222 """
223 if not token or not handles:
224 return []
225 items: List[Dict[str, Any]] = []
226 seen_ids: set[str] = set()
227 for raw in handles:
228 handle = str(raw).lstrip("@").strip()
229 if not handle:
230 continue
231 q = f"from:{handle} since:{from_date} until:{to_date}"
232 got, auth_error = _execute_search(
233 q, count_per, token,
234 label=f"from:{handle}", id_prefix="XF",
235 seen_ids=seen_ids, relevance_query=topic,
236 index_offset=len(items),
237 failure_out=failure_out,
238 )
239 if auth_error:
240 # Fatal auth/payment failure — stop, keep what we have. Surface the
241 # reason: an empty FROM lane is otherwise indistinguishable from a
242 # subject who simply did not post.
243 if failure_out is not None:
244 failure_out.append(auth_error)
245 break
246 items.extend(got)
247 return items
248
249
250 def search_mentions(
251 handles: List[str],
252 from_date: str,
253 to_date: str,
254 *,
255 topic: str = "",
256 count_per: int = 5,
257 token: str = "",
258 failure_out: Optional[List[str]] = None,
259 ) -> List[Dict[str, Any]]:
260 """ABOUT lane: tweets mentioning each handle, authored by OTHERS.
261
262 Queries ``@handle`` then drops the handle's own tweets (``_is_own``) so only
263 third-party mentions remain. Returns a flat list of item dicts.
264
265 When ``failure_out`` is provided, any failure reason is appended — the
266 fatal auth/payment one that stops the lane, and the per-handle transient
267 ones (429, 5xx, network, unparseable payload) that do not — so the caller
268 can distinguish a failed request from a handle nobody mentioned.
269 """
270 if not token or not handles:
271 return []
272 items: List[Dict[str, Any]] = []
273 seen_ids: set[str] = set()
274 for raw in handles:
275 handle = str(raw).lstrip("@").strip()
276 if not handle:
277 continue
278 q = f"@{handle} since:{from_date} until:{to_date}"
279 got, auth_error = _execute_search(
280 q, count_per, token,
281 label=f"@{handle}", id_prefix="XA",
282 seen_ids=seen_ids, relevance_query=topic,
283 index_offset=len(items),
284 failure_out=failure_out,
285 )
286 if auth_error:
287 if failure_out is not None:
288 failure_out.append(auth_error)
289 break
290 items.extend(it for it in got if not _is_own(it.get("url", ""), handle))
291 return items
292
293
294 def probe_works(token: str, timeout: int = 8) -> Optional[bool]:
295 """Cheap runtime check that the xquik key actually returns data.
296
297 Mirrors ``bird_x.probe_works`` for the key-based X path so ``--diagnose``
298 reflects reality instead of static key presence. Returns True
299 (funded/working), False (a clear auth/payment failure — 401/403, or 402
300 when the key is configured but unpaid), or None (inconclusive: timeout /
301 transient HTTP) so callers fail open.
302 The human-readable reason is available via ``probe_reason()``. Cached per
303 process so repeated diagnose calls don't re-probe.
304 """
305 global _probe_cache
306 if _probe_cache[0] != "unset":
307 return _probe_cache[0]
308 if not token:
309 _probe_cache = (False, "no XQUIK_API_KEY configured")
310 return False
311 from datetime import timedelta, timezone
312 since = (datetime.now(timezone.utc) - timedelta(days=30)).strftime("%Y-%m-%d")
313 # @x (the platform's own account) posts frequently, so a no-error response
314 # means the key works even if this particular window is quiet.
315 q = f"from:x since:{since}"
316 full_url = f"{_BASE_URL}/x/tweets/search?q={_url_encode(q)}&queryType=Top&limit=1"
317 try:
318 request_headers = {"X-Api-Key": token}
319 http.get(full_url, headers=request_headers, timeout=timeout, retries=0)
320 except http.HTTPError as exc:
321 status = getattr(exc, "status_code", None)
322 if status == 402:
323 _probe_cache = (False, "xquik key unpaid: payment required (402)")
324 elif status in (401, 403):
325 _probe_cache = (False, f"xquik auth failed ({status})")
326 else:
327 # 5xx / unexpected status — inconclusive, don't report a false-down.
328 _probe_cache = (None, f"xquik probe inconclusive ({status})")
329 return _probe_cache[0]
330 except Exception as exc:
331 _probe_cache = (None, f"xquik probe inconclusive ({type(exc).__name__})")
332 return None
333 _probe_cache = (True, "ok")
334 return True
335
336
337 def probe_reason() -> str:
338 """Human-readable reason for the last ``probe_works`` result (or '')."""
339 return _probe_cache[1]
340
341
342 def search_and_enrich(
343 topic: str,
344 from_date: str,
345 to_date: str,
346 depth: str = "default",
347 token: str = "",
348 ) -> Dict[str, Any]:
349 """Search X via Xquik and return results.
350
351 Xquik API returns full engagement data by default, so no separate
352 enrichment step is needed.
353 """
354 return search_xquik(topic, from_date, to_date, depth=depth, token=token)
355
356
357 def parse_xquik_response(response: Dict[str, Any]) -> List[Dict[str, Any]]:
358 """Extract items from search response.
359
360 Args:
361 response: Response dict from search_xquik()
362
363 Returns:
364 List of normalized item dicts.
365 """
366 return response.get("items", [])
367
368
369 def _parse_tweet(
370 tweet: Dict[str, Any], index: int, query: str, id_prefix: str = "XQ"
371 ) -> Dict[str, Any] | None:
372 """Parse a single tweet from the API response into the standard item format."""
373 author = tweet.get("author") or {}
374 username = str(author.get("username", "")).lstrip("@")
375 tweet_id = str(tweet.get("id", ""))
376
377 # Build URL
378 url = ""
379 if username and tweet_id:
380 url = f"https://x.com/{username}/status/{tweet_id}"
381 if not url:
382 return None
383
384 # Parse date
385 date = None
386 created_at = tweet.get("createdAt") or ""
387 if created_at:
388 try:
389 if len(created_at) > 10 and created_at[10] == "T":
390 dt = datetime.fromisoformat(created_at.replace("Z", "+00:00"))
391 else:
392 dt = datetime.strptime(created_at, "%a %b %d %H:%M:%S %z %Y")
393 date = dt.strftime("%Y-%m-%d")
394 except (ValueError, TypeError):
395 pass
396
397 text = str(tweet.get("text", "")).strip()[:500]
398
399 # Leading-run @mentions = who the post is directed at (reply target). Shared
400 # parser with bird so the first-party interaction signal fires for xquik too.
401 from .query import leading_mentions
402 mentioned_handles = leading_mentions(text)
403
404 # Build engagement dict with full metrics
405 engagement = {
406 "likes": _safe_int(tweet.get("likeCount")),
407 "reposts": _safe_int(tweet.get("retweetCount")),
408 "replies": _safe_int(tweet.get("replyCount")),
409 "quotes": _safe_int(tweet.get("quoteCount")),
410 "views": _safe_int(tweet.get("viewCount")),
411 "bookmarks": _safe_int(tweet.get("bookmarkCount")),
412 }
413
414 return {
415 "id": f"{id_prefix}{index + 1}",
416 "text": text,
417 "url": url,
418 "author_handle": username,
419 "date": date,
420 "engagement": engagement,
421 "mentioned_handles": mentioned_handles,
422 "relevance": _compute_relevance(query, text) if query else 0.7,
423 "why_relevant": "",
424 }
425
426
427 def _safe_int(value: Any) -> int | None:
428 """Convert value to int, returning None on failure."""
429 if value is None:
430 return None
431 try:
432 return int(value)
433 except (ValueError, TypeError):
434 return None
435
436
437 def _url_encode(text: str) -> str:
438 """URL-encode a string using stdlib."""
439 from urllib.parse import quote
440 return quote(text, safe="")
441
441 lines PYTHON