返回 last30days-skill
setup_wizard.py
根目录 / skills / last30days / scripts / lib / setup_wizard.py
1 """First-run setup wizard for last30days.
2
3 Detects first run, performs auto-setup (cookie extraction + yt-dlp check),
4 and writes configuration. The actual wizard UI is SKILL.md-driven (the LLM
5 presents it), but this module provides the detection and setup actions.
6 """
7
8 import json
9 import logging
10 import os
11 import re
12 import shutil
13 import subprocess
14 import time
15 from pathlib import Path
16 from typing import Any, Dict, Optional, Tuple
17 from urllib.error import HTTPError, URLError
18 from urllib.request import Request, urlopen
19
20 from . import brightdata
21
22 logger = logging.getLogger(__name__)
23
24
25 def is_first_run(config: Dict[str, Any]) -> bool:
26 """Return True if the setup wizard has not been completed.
27
28 Checks for SETUP_COMPLETE in the config dict. If it's not set
29 (None or empty string), the user hasn't gone through setup yet.
30 """
31 return not config.get("SETUP_COMPLETE")
32
33
34 _WELCOME_TEXT = """Welcome to /last30days! I research any topic across Reddit, X, YouTube, TikTok, Digg, arXiv, Techmeme, HN, Polymarket & more - what people actually said in the last 30 days. Let's get you set up (~30s).
35
36 I synthesize what people are actually saying right now across social, news, and market sources.
37
38 Auto setup gives you the core sources free in about 30 seconds:
39 - Reddit with comments - free keyless discovery (Reddit search + shreddit), no API key needed.
40 - YouTube search + transcripts - installs yt-dlp (open source, 190K+ GitHub stars).
41 - Digg - trending news, GitHub stars, and pipeline feeds - installs the free, keyless Digg CLI.
42 - arXiv (papers) + Techmeme (tech-news) - install free, keyless Printing Press CLIs and run on any topic (arXiv is relevance + recency gated to research topics).
43 - StockTwits - retail trader sentiment - auto-on when your topic is a ticker or crypto (e.g. "$NVDA earnings", "bitcoin"), off for everything else.
44 - Trustpilot - brand/company review sentiment - opt-in (add trustpilot to INCLUDE_SOURCES), off by default.
45 - Hacker News + Polymarket + GitHub (auto-on if the gh CLI is installed) - always on, zero config.
46 - X/Twitter - optional. It stays available when you already configured it, or after you explicitly approve a browser-cookie read; skipping it never blocks research.
47
48 Want TikTok and Instagram too? ScrapeCreators adds those (10,000 free calls, scrapecreators.com). No kickbacks, no affiliation.
49
50 Power users can turn on more sources in the Manual Setup guide (LinkedIn, Bluesky, Perplexity, and others) - each needs its own credential, so they are off by default."""
51
52
53 def render_welcome() -> str:
54 """Return the first-run welcome text.
55
56 Owned by the engine (single source of truth) so the model relays it rather
57 than re-authoring it -- authored prose gets skipped, relayed command output
58 does not. Mirrors the SKILL.md welcome; keep in sync if the source set
59 changes.
60 """
61 return _WELCOME_TEXT
62
63
64 # Neutral note recorded on an official-only host instead of a cookie scan.
65 # Deliberately names no cookie mechanism beyond the fact that none is used
66 # (the vocabulary rule for Grok Bot onboarding output).
67 OFFICIAL_HOST_COOKIE_NOTE = "browser sessions are not read on this host"
68
69
70 def run_auto_setup(config: Dict[str, Any], *, allow_browser_cookies: bool = False) -> Dict[str, Any]:
71 """Perform the auto-setup actions.
72
73 - Optionally runs cookie extraction for all registered domains, trying the
74 browsers from ``env.cookie_extraction_browsers()``. Browser reads are off
75 unless ``allow_browser_cookies`` is true.
76 - Checks if yt-dlp is installed
77 - Best-effort install of digg-pp-cli (Printing Press library)
78
79 Returns:
80 Dict with keys:
81 cookies_found: {source_name: browser_name} for each source where cookies were found
82 browser_cookie_scan_attempted: bool (True only after explicit consent
83 AND on a host whose X policy permits cookie discovery)
84 cookie_note: present only on an official-only host, where the scan
85 is skipped for every domain (neutral, relayable text)
86 ytdlp_installed: bool
87 ytdlp_action: already_installed | installed | install_failed | no_homebrew
88 digg_installed: bool (True when the engine can resolve digg-pp-cli on PATH)
89 digg_action: already_installed | installed | installed_off_path | install_failed | no_npx
90 env_written: bool (always False here — caller writes config separately)
91 ytdlp_stderr: present when ytdlp_action is install_failed
92 digg_stderr: present when digg_action is install_failed
93 digg_path: present when digg_action is installed_off_path (binary on disk, not on PATH)
94 """
95 from .env import COOKIE_DOMAINS, cookie_extraction_browsers, x_policy
96
97 cookies_found: Dict[str, str] = {}
98 cookie_note: Optional[str] = None
99
100 # Official-only host (LAST30DAYS_HOST=grok-bot): the consented
101 # cookie scan is skipped for EVERY domain (X and Truth Social alike) and
102 # recorded as not attempted, with a neutral note the caller can relay.
103 # The free CLI installs below still run. A LAST30DAYS_X_BACKEND=bird pin
104 # is the one path that re-enables discovery, via x_policy.
105 if allow_browser_cookies and not x_policy(config).cookie_discovery:
106 allow_browser_cookies = False
107 cookie_note = OFFICIAL_HOST_COOKIE_NOTE
108
109 if allow_browser_cookies:
110 from . import cookie_extract
111
112 cookie_config = dict(config)
113 cookie_config["BROWSER_CONSENT"] = "true"
114 if not (cookie_config.get("FROM_BROWSER") or "").strip():
115 # Chromium-first: Chrome/Brave/etc. read cookies via the Keychain
116 # with no Full Disk Access, so try them before Safari, whose
117 # binarycookies read requires FDA (the dead-end most users hit).
118 # firefox/safari stay as the silent fallbacks. Note: an explicit
119 # comma list preserves this order (cookie_extraction_browsers);
120 # "auto" would put the silent browsers first, so do not use it here.
121 cookie_config["FROM_BROWSER"] = "chrome,brave,edge,vivaldi,arc,chromium,firefox,safari"
122 browsers = cookie_extraction_browsers(cookie_config)
123
124 for source_name, spec in COOKIE_DOMAINS.items():
125 domain = spec["domain"]
126 cookie_names = spec["cookies"]
127
128 for browser in browsers:
129 try:
130 result = cookie_extract.extract_cookies_with_source(browser, domain, cookie_names)
131 except Exception as exc:
132 logger.debug("Cookie extraction failed for %s via %s: %s", source_name, browser, exc)
133 continue
134 if result is not None and cookie_extract.has_complete_pair(result[0], cookie_names):
135 cookies_found[source_name] = result[1]
136 break # Complete pair found for this service, stop trying browsers
137
138 # Check yt-dlp availability and install via Homebrew if missing. Windows
139 # has no Homebrew, and its working install path is `pip install yt-dlp`
140 # (see #904), so it gets its own no-op-install guidance branch instead of
141 # falling into the Homebrew-oriented no_homebrew outcome.
142 ytdlp_action: str
143 if shutil.which("yt-dlp") is not None:
144 ytdlp_installed = True
145 ytdlp_action = "already_installed"
146 elif os.name == "nt":
147 ytdlp_installed = False
148 ytdlp_action = "no_pip_windows"
149 elif shutil.which("brew") is not None:
150 brew_stderr = ""
151 try:
152 proc = subprocess.run(
153 ["brew", "install", "yt-dlp"],
154 capture_output=True, text=True, timeout=120,
155 )
156 if proc.returncode == 0:
157 ytdlp_installed = True
158 ytdlp_action = "installed"
159 else:
160 ytdlp_installed = False
161 ytdlp_action = "install_failed"
162 brew_stderr = proc.stderr
163 logger.warning("brew install yt-dlp failed: %s", proc.stderr)
164 except Exception as exc:
165 ytdlp_installed = False
166 ytdlp_action = "install_failed"
167 brew_stderr = str(exc)
168 logger.warning("brew install yt-dlp exception: %s", exc)
169 else:
170 ytdlp_installed = False
171 ytdlp_action = "no_homebrew"
172
173 digg_installed, digg_action, digg_stderr, digg_path = _install_digg_cli()
174 pp_sources = install_default_pp_sources()
175
176 results: Dict[str, Any] = {
177 "cookies_found": cookies_found,
178 "browser_cookie_scan_attempted": allow_browser_cookies,
179 "ytdlp_installed": ytdlp_installed,
180 "ytdlp_action": ytdlp_action,
181 "digg_installed": digg_installed,
182 "digg_action": digg_action,
183 # Per-CLI status for the additional default-on Printing Press sources
184 # (arxiv, techmeme, trustpilot): {source: {installed, action, ...}}.
185 "pp_sources": pp_sources,
186 # Reported, never installed: this CLI spends the user's own metered
187 # credits, so acquiring it stays their decision. Passing
188 # config matters: a user whose key lives in a .env file or the
189 # keychain (rather than a `brightdata login` credentials file) is
190 # active in the engine, and setup must not tell them otherwise.
191 "brightdata": brightdata_status(config),
192 "env_written": False,
193 }
194 if cookie_note:
195 results["cookie_note"] = cookie_note
196 if ytdlp_action == "install_failed":
197 results["ytdlp_stderr"] = brew_stderr
198 if digg_action == "install_failed":
199 results["digg_stderr"] = digg_stderr
200 if digg_path:
201 results["digg_path"] = digg_path
202 return results
203
204
205 # Generous timeout: the install shells out to `npx`, which may download the
206 # Printing Press package and build the Go binary over the network.
207 DIGG_INSTALL_TIMEOUT = 300
208 DIGG_CLI_BIN = "digg-pp-cli"
209 # Pin the catalog installer; matches printing-press-library npm 0.1.16 default
210 # ($HOME/.local/bin on macOS/Linux).
211 PRINTING_PRESS_NPM = "@mvanhorn/printing-press-library@0.1.16"
212 DIGG_INSTALL_CMD = f"npx -y {PRINTING_PRESS_NPM} install digg --cli-only"
213
214
215 def _digg_bin_candidate_paths() -> list[Path]:
216 """Known install locations for digg-pp-cli (Printing Press library defaults).
217
218 Order: current installer default (~/.local/bin), legacy Go bins, Windows
219 managed dir. The directory list is ``health.installer_bin_dirs()`` — the
220 shared single source — with the Digg filename variants appended (plain
221 name for Unix-style dirs, ``.exe`` in the Windows managed dir).
222 ``pipeline.available_sources()`` only activates Digg when
223 ``shutil.which`` resolves on PATH — probing these dirs is for setup
224 verification and honest off-PATH messaging, not engine activation.
225 """
226 from . import health
227
228 win_dir = health.windows_printing_press_bin_dir()
229 candidates: list[Path] = []
230 for directory in health.installer_bin_dirs():
231 if win_dir is not None and directory == win_dir:
232 candidates.append(directory / f"{DIGG_CLI_BIN}.exe")
233 else:
234 candidates.append(directory / DIGG_CLI_BIN)
235 return candidates
236
237
238 def _digg_on_path() -> Optional[str]:
239 """Return digg-pp-cli when the engine would activate Digg (PATH-resolvable)."""
240 return shutil.which(DIGG_CLI_BIN)
241
242
243 def _digg_off_path_binary() -> Optional[str]:
244 """Return digg-pp-cli path from known install dirs when not on PATH."""
245 for candidate in _digg_bin_candidate_paths():
246 if candidate.is_file() and os.access(candidate, os.X_OK):
247 return str(candidate)
248 return None
249
250
251 def _digg_bin_dir_hint(digg_path: str) -> str:
252 """Return a copy-pasteable PATH directory for the given binary path."""
253 parent = os.path.dirname(os.path.expanduser(digg_path))
254 if os.name == "nt":
255 # Windows PATH edits use absolute dirs; $HOME is a Unix shell convention.
256 return parent
257 home = str(Path.home())
258 if parent == home:
259 return "$HOME"
260 prefix = home + os.sep
261 if parent.startswith(prefix):
262 rel = parent[len(prefix):].replace(os.sep, "/")
263 return f"$HOME/{rel}" if rel else "$HOME"
264 return parent
265
266
267 def _run_npx_install(slug: str) -> Tuple[str, str]:
268 """Resolve ``npx`` and run the Printing Press catalog install for ``slug``.
269
270 Shared by ``_install_digg_cli`` and ``_install_pp_cli`` -- this is only the
271 "resolve npx, run the install, interpret no_npx/exception/nonzero-rc"
272 slice; each caller keeps its own on-path/off-path re-verification
273 (``_digg_bin_candidate_paths`` vs ``_pp_bin_candidate_paths`` already use
274 different candidate-directory sources, so merging them here would change
275 off-path detection behavior beyond this fix's scope).
276
277 Fixes the Windows PATHEXT mismatch: ``shutil.which("npx")`` resolves
278 ``npx.CMD`` via PATHEXT, but ``subprocess.run`` given the bare string
279 ``"npx"`` as argv[0] does not do that resolution and fails with
280 ``WinError 2``. Passing the resolved path is a no-op on macOS/Linux, where
281 ``shutil.which`` already returns the exact path ``CreateProcess``/``execve``
282 would resolve.
283
284 Returns ``(action, stderr)``: ``action`` is ``"no_npx"``,
285 ``"install_failed"``, or ``""`` when the subprocess ran and returned
286 ``rc=0`` (in which case ``stderr`` carries any non-fatal stderr output for
287 the caller's own off-path logging).
288 """
289 npx = shutil.which("npx")
290 if npx is None:
291 return "no_npx", ""
292 try:
293 proc = subprocess.run(
294 [npx, "-y", PRINTING_PRESS_NPM, "install", slug, "--cli-only"],
295 capture_output=True, text=True, timeout=DIGG_INSTALL_TIMEOUT,
296 )
297 except Exception as exc:
298 logger.warning("npx install %s exception: %s", slug, exc)
299 return "install_failed", str(exc)
300 if proc.returncode != 0:
301 stderr = proc.stderr or f"npx install {slug} exited {proc.returncode}"
302 logger.warning("npx install %s failed (rc=%s): %s", slug, proc.returncode, stderr)
303 return "install_failed", stderr
304 return "", (proc.stderr or "")
305
306
307 def _install_digg_cli() -> Tuple[bool, str, str, str]:
308 """Best-effort install of the digg-pp-cli binary.
309
310 Mirrors the yt-dlp/brew auto-install: it never raises, and degrades to a
311 recommend-only outcome when the installer is unavailable. Uses
312 ``@mvanhorn/printing-press-library`` (``--cli-only``) — the same catalog
313 installer as pp-digg; Hermes/OpenClaw skill wiring is irrelevant here.
314
315 Returns ``(engine_active, action, stderr, off_path_binary)`` where
316 ``engine_active`` is True only when ``shutil.which`` resolves the binary
317 (matching ``pipeline.available_sources()``). ``action`` is one of:
318 already_installed | installed | installed_off_path | install_failed | no_npx
319 ``stderr`` is populated on ``install_failed``. ``off_path_binary`` is set
320 when the binary exists on disk but is not PATH-visible to this process.
321 """
322 on_path = _digg_on_path()
323 if on_path:
324 return True, "already_installed", "", ""
325 off_path = _digg_off_path_binary()
326 if off_path:
327 return False, "installed_off_path", "", off_path
328 action, stderr = _run_npx_install("digg")
329 if action:
330 return False, action, stderr, ""
331 on_path = _digg_on_path()
332 if on_path:
333 return True, "installed", "", ""
334 off_path = _digg_off_path_binary()
335 if off_path:
336 combined = stderr.strip()
337 if combined:
338 logger.warning("digg-pp-cli installed off PATH: %s", combined)
339 return False, "installed_off_path", combined, off_path
340 stderr_msg = stderr or "install completed but digg-pp-cli was not found"
341 logger.warning("npx install digg failed verification: %s", stderr_msg)
342 return False, "install_failed", stderr_msg, ""
343
344
345 # Additional default-on Printing Press sources installed the same way as Digg:
346 # (engine source key, slug for `install <slug>`, binary name). These activate in
347 # ``pipeline.available_sources()`` when ``shutil.which`` resolves the binary.
348 # Trustpilot is intentionally NOT here: it is opt-in (INCLUDE_SOURCES=trustpilot)
349 # because of its headless-Chrome cookie harvest, so auto-installing its binary
350 # for a source that stays off by default would be wasted work. Opting in installs
351 # it on demand via `npx ... install trustpilot --cli-only` (see CONFIGURATION.md).
352 PP_DEFAULT_SOURCES: list[tuple[str, str, str]] = [
353 ("arxiv", "arxiv", "arxiv-pp-cli"),
354 ("techmeme", "techmeme", "techmeme-pp-cli"),
355 ]
356
357 # Bright Data is deliberately absent from PP_DEFAULT_SOURCES: it is not a
358 # Printing Press CLI, it is opt-in like Trustpilot, and it spends the user's
359 # own metered credits. Setup reports its state and never installs it.
360 BRIGHTDATA_BIN = "brightdata"
361
362
363 def _brightdata_off_path_binary() -> Optional[str]:
364 """Locate a brightdata binary that exists on disk but not on PATH.
365
366 Covers the common npm global prefixes. The distinction matters because
367 Hermes and OpenClaw gateways routinely run the engine with a PATH that
368 excludes the user's npm bin directory, so "installed" and "the engine
369 can see it" are different questions.
370 """
371 home = Path.home()
372 candidates = [
373 home / ".local" / "bin" / BRIGHTDATA_BIN,
374 home / ".npm-global" / "bin" / BRIGHTDATA_BIN,
375 Path("/opt/homebrew/bin") / BRIGHTDATA_BIN,
376 Path("/usr/local/bin") / BRIGHTDATA_BIN,
377 ]
378 npm_prefix = os.environ.get("NPM_CONFIG_PREFIX")
379 if npm_prefix:
380 candidates.insert(0, Path(npm_prefix) / "bin" / BRIGHTDATA_BIN)
381 for candidate in candidates:
382 try:
383 if candidate.is_file() and os.access(candidate, os.X_OK):
384 return str(candidate)
385 except OSError:
386 continue
387 return None
388
389
390 def brightdata_status(config: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
391 """Report the Bright Data install and auth state honestly.
392
393 Deliberately never claims the source is active unless the engine's own
394 gate would pass -- ``brightdata.is_available`` is the single predicate,
395 so setup and the engine cannot drift apart. Three states matter:
396
397 * ``already_installed`` -- on PATH; ``authenticated`` says whether the
398 amazon lane will actually run.
399 * ``installed_off_path`` -- on disk but invisible to the engine, which
400 is the Hermes/OpenClaw failure mode. Carries the path so the user can
401 fix their PATH.
402 * ``not_installed`` -- nothing found. No auto-install: this CLI
403 spends the user's metered credits, so acquiring it is their call.
404 """
405 installed = brightdata.is_installed()
406 authenticated = brightdata.has_credentials(config)
407 if installed:
408 action = "already_installed"
409 off_path = ""
410 else:
411 off_path = _brightdata_off_path_binary() or ""
412 action = "installed_off_path" if off_path else "not_installed"
413
414 status: Dict[str, Any] = {
415 "installed": installed,
416 "action": action,
417 "authenticated": installed and authenticated,
418 # The engine gate, verbatim. Never report active on anything else.
419 "engine_active": brightdata.is_available(config),
420 }
421 if off_path:
422 status["path"] = off_path
423 status["hint"] = (
424 f"brightdata found at {off_path} but not on PATH; add its directory "
425 "to PATH so the engine subprocess can see it"
426 )
427 elif installed and not authenticated:
428 status["hint"] = "run `brightdata login` to activate the amazon source"
429 elif not installed:
430 status["hint"] = (
431 "install with `npm i -g @brightdata/cli` then `brightdata login` "
432 "to enable the amazon source"
433 )
434 return status
435
436
437 def _pp_bin_candidate_paths(bin_name: str) -> list[Path]:
438 """Known install locations for a Printing Press CLI binary (slug-parameterized
439 mirror of ``_digg_bin_candidate_paths``)."""
440 home = Path.home()
441 candidates: list[Path] = [home / ".local" / "bin" / bin_name]
442 gopath = os.environ.get("GOPATH")
443 if gopath:
444 candidates.append(Path(gopath) / "bin" / bin_name)
445 candidates.append(home / "go" / "bin" / bin_name)
446 if os.name == "nt":
447 local_app = os.environ.get("LOCALAPPDATA") or os.environ.get("LocalAppData")
448 if local_app:
449 candidates.append(
450 Path(local_app) / "Programs" / "PrintingPress" / "bin" / f"{bin_name}.exe"
451 )
452 return candidates
453
454
455 def _pp_off_path_binary(bin_name: str) -> Optional[str]:
456 for candidate in _pp_bin_candidate_paths(bin_name):
457 if candidate.is_file() and os.access(candidate, os.X_OK):
458 return str(candidate)
459 return None
460
461
462 def _install_pp_cli(slug: str, bin_name: str) -> Tuple[bool, str, str, str]:
463 """Best-effort install of a Printing Press CLI binary.
464
465 Slug-parameterized mirror of ``_install_digg_cli``: never raises, degrades
466 to recommend-only when the installer is unavailable. Returns
467 ``(engine_active, action, stderr, off_path_binary)`` with the same action
468 taxonomy: already_installed | installed | installed_off_path |
469 install_failed | no_npx.
470 """
471 on_path = shutil.which(bin_name)
472 if on_path:
473 return True, "already_installed", "", ""
474 off_path = _pp_off_path_binary(bin_name)
475 if off_path:
476 return False, "installed_off_path", "", off_path
477 action, stderr = _run_npx_install(slug)
478 if action:
479 return False, action, stderr, ""
480 on_path = shutil.which(bin_name)
481 if on_path:
482 return True, "installed", "", ""
483 off_path = _pp_off_path_binary(bin_name)
484 if off_path:
485 combined = stderr.strip()
486 if combined:
487 logger.warning("%s installed off PATH: %s", bin_name, combined)
488 return False, "installed_off_path", combined, off_path
489 stderr_msg = stderr or f"install completed but {bin_name} was not found"
490 logger.warning("npx install %s failed verification: %s", slug, stderr_msg)
491 return False, "install_failed", stderr_msg, ""
492
493
494 def install_default_pp_sources() -> Dict[str, Dict[str, Any]]:
495 """Best-effort install of every additional default-on Printing Press source.
496
497 Returns ``{source_key: {installed, action, stderr?, path?}}`` so the wizard
498 can report per-CLI status alongside Digg without raising on any single
499 failure.
500 """
501 out: Dict[str, Dict[str, Any]] = {}
502 for source_key, slug, bin_name in PP_DEFAULT_SOURCES:
503 installed, action, stderr, off_path = _install_pp_cli(slug, bin_name)
504 entry: Dict[str, Any] = {"installed": installed, "action": action}
505 if action == "install_failed" and stderr:
506 entry["stderr"] = stderr
507 if off_path:
508 entry["path"] = off_path
509 out[source_key] = entry
510 return out
511
512
513 def _open_secret_append(path: Path):
514 """Open ``path`` for appending as a 0o600 secret file with no readable window.
515
516 ``os.open`` with ``O_CREAT|O_WRONLY|O_APPEND`` and mode ``0o600`` sets
517 restrictive permissions at creation (umask can only further restrict, never
518 widen, so the file is never world-readable even transiently). An explicit
519 ``chmod`` afterwards also tightens a pre-existing loose file. This matters
520 because the .env stores API keys, cookies, and tokens.
521 """
522 fd = os.open(path, os.O_CREAT | os.O_WRONLY | os.O_APPEND, 0o600)
523 try:
524 os.chmod(path, 0o600)
525 except OSError:
526 pass
527 return os.fdopen(fd, "a", encoding="utf-8")
528
529
530 def _replace_env_line(env_path: Path, content: str, key_name: str, value: str) -> bool:
531 """Rewrite every ``key_name=`` line of ``content`` with ``value`` as a 0o600 secret.
532
533 The new file is written to a sibling temp path opened at 0o600 and moved
534 over the original, so the secret never has a readable window and a
535 crash mid-write leaves the old file intact.
536 """
537 new_line = f"{key_name}={_format_env_value(value)}"
538 lines = []
539 replaced = False
540 for line in content.splitlines():
541 stripped = line.strip()
542 is_key = (
543 stripped and not stripped.startswith("#") and "=" in stripped
544 and stripped.split("=", 1)[0].strip() == key_name
545 )
546 if is_key:
547 if not replaced:
548 lines.append(new_line)
549 replaced = True
550 continue
551 lines.append(line)
552 if not replaced:
553 lines.append(new_line)
554 tmp_path = env_path.with_name(env_path.name + ".tmp")
555 fd = os.open(tmp_path, os.O_CREAT | os.O_WRONLY | os.O_TRUNC, 0o600)
556 with os.fdopen(fd, "w", encoding="utf-8") as f:
557 f.write("\n".join(lines) + "\n")
558 os.replace(tmp_path, env_path)
559 try:
560 os.chmod(env_path, 0o600)
561 except OSError:
562 pass
563 return True
564
565
566 def _format_env_value(value: str) -> str:
567 """Quote a value so it round-trips through env.load_env_file.
568
569 env.load_env_file strips a single layer of matching surrounding quotes but
570 does NOT process backslash escapes, so we wrap (never escape):
571 - plain tokens (no whitespace, no leading quote): returned unchanged;
572 - values with whitespace/leading quote and no double-quote: double-quoted;
573 - values containing a double-quote but no single-quote: single-quoted;
574 - values containing both quote types: returned as-is (best effort; no
575 wrapper round-trips through the loader, and tokens never hit this).
576 Newlines are not valid in a single-line env value and are stripped.
577 """
578 value = value.replace("\r", "").replace("\n", " ")
579 needs_quoting = (not value) or value[0] in ("'", '"') or any(c.isspace() for c in value)
580 if not needs_quoting:
581 return value
582 if '"' not in value:
583 return f'"{value}"'
584 if "'" not in value:
585 return f"'{value}'"
586 return value
587
588
589 def write_setup_config(
590 env_path: Path,
591 from_browser: str | None = None,
592 *,
593 browser_consent: bool | None = None,
594 ) -> bool:
595 """Write setup completion, browser selection, and consent to the .env file.
596
597 Creates the file and parent directories if needed.
598 Appends without overwriting existing keys, except for an explicit consent
599 decision.
600
601 Args:
602 env_path: Path to the .env file (e.g. ~/.config/last30days/.env)
603 from_browser: Browser or comma-separated browser list that actually
604 yielded cookies after consent (e.g. "chrome,firefox"). Pass None
605 (default) to leave FROM_BROWSER unchanged; when unset, future runs
606 do not read native browser stores. Avoid "auto", which would also probe
607 browsers that did not supply cookies during setup.
608 browser_consent: Record the user's current cookie-access decision.
609 None preserves any previous decision.
610
611 Returns:
612 True if config was written successfully, False on error.
613 """
614 try:
615 env_path = Path(env_path)
616 env_path.parent.mkdir(parents=True, exist_ok=True)
617 if browser_consent is not None:
618 if not write_api_key(
619 env_path, "true" if browser_consent else "false",
620 key_name="BROWSER_CONSENT", replace=True,
621 ):
622 return False
623
624 # Read existing content to avoid overwriting keys
625 existing_keys: set = set()
626 existing_content = ""
627 if env_path.exists():
628 existing_content = env_path.read_text(encoding="utf-8")
629 for line in existing_content.splitlines():
630 stripped = line.strip()
631 if stripped and not stripped.startswith("#") and "=" in stripped:
632 key = stripped.split("=", 1)[0].strip()
633 existing_keys.add(key)
634
635 lines_to_add = []
636 if "SETUP_COMPLETE" not in existing_keys:
637 lines_to_add.append("SETUP_COMPLETE=true")
638 if from_browser and "FROM_BROWSER" not in existing_keys:
639 lines_to_add.append(f"FROM_BROWSER={_format_env_value(from_browser)}")
640
641 if not lines_to_add:
642 return True # Nothing to write, already configured
643
644 # Create/append as a 0o600 secret file: the .env holds tokens and keys,
645 # so it must never be created world-readable.
646 with _open_secret_append(env_path) as f:
647 if existing_content and not existing_content.endswith("\n"):
648 f.write("\n")
649 f.write("\n".join(lines_to_add) + "\n")
650
651 return True
652
653 except OSError as exc:
654 logger.error("Failed to write setup config to %s: %s", env_path, exc)
655 return False
656
657
658 def write_api_key(
659 env_path: Path,
660 api_key: str,
661 key_name: str = "SCRAPECREATORS_API_KEY",
662 *,
663 replace: bool = False,
664 ) -> bool:
665 """Append an API key to the .env file as a 0o600 secret.
666
667 Reuses the same secret-safe write path as ``write_setup_config`` so the
668 value lands with restrictive permissions and round-trips through
669 ``env.load_env_file``. Idempotent by default: if ``key_name`` is already
670 present in the file, nothing is written and the existing value is
671 preserved (we never clobber a key the user may have set by hand). With
672 ``replace=True`` an existing line is rewritten in place instead, so an
673 explicit ``setup --store-key`` can rotate a rejected credential.
674
675 Args:
676 env_path: Path to the .env file (e.g. ~/.config/last30days/.env).
677 api_key: The raw key value to persist.
678 key_name: The env var name to write (default SCRAPECREATORS_API_KEY).
679 replace: Rewrite an existing ``key_name`` line instead of keeping it.
680
681 Returns:
682 True if the key was written or already present, False on error or when
683 ``api_key`` is empty.
684 """
685 if not api_key:
686 return False
687 try:
688 env_path = Path(env_path)
689 env_path.parent.mkdir(parents=True, exist_ok=True)
690
691 existing_content = ""
692 if env_path.exists():
693 existing_content = env_path.read_text(encoding="utf-8")
694 for line in existing_content.splitlines():
695 stripped = line.strip()
696 if stripped and not stripped.startswith("#") and "=" in stripped:
697 if stripped.split("=", 1)[0].strip() == key_name:
698 if replace:
699 return _replace_env_line(env_path, existing_content, key_name, api_key)
700 return True # Already configured; do not duplicate
701
702 line = f"{key_name}={_format_env_value(api_key)}\n"
703 with _open_secret_append(env_path) as f:
704 if existing_content and not existing_content.endswith("\n"):
705 f.write("\n")
706 f.write(line)
707
708 return True
709
710 except OSError as exc:
711 logger.error("Failed to write API key to %s: %s", env_path, exc)
712 return False
713
714
715 def mask_api_key(api_key: str) -> str:
716 """Return a non-secret display form of an API key (prefix + last 4).
717
718 Used so the key never appears verbatim in stdout the host model captures.
719 Short or empty keys collapse to a fixed placeholder.
720 """
721 if not api_key or len(api_key) <= 8:
722 return "sc_…"
723 return f"{api_key[:3]}…{api_key[-4:]}"
724
725
726 def get_setup_status_text(results: Dict[str, Any]) -> str:
727 """Return a human-readable summary of auto-setup results.
728
729 Args:
730 results: Dict from run_auto_setup()
731
732 Returns:
733 Multi-line status text.
734 """
735 lines = []
736 lines.append("Setup complete! Here's what I found:")
737 lines.append("")
738
739 cookies_found = results.get("cookies_found", {})
740 if results.get("browser_cookie_scan_attempted") and cookies_found:
741 for source, browser in cookies_found.items():
742 lines.append(f" - {source.upper()} cookies found in {browser}")
743
744 ytdlp_action = results.get("ytdlp_action", "")
745 if ytdlp_action == "installed":
746 lines.append(" - Installed yt-dlp via Homebrew")
747 elif ytdlp_action == "install_failed":
748 lines.append(" - yt-dlp install failed \u2014 run `brew install yt-dlp` manually")
749 elif ytdlp_action == "no_homebrew":
750 lines.append(" - yt-dlp not found. Install Homebrew first, then: brew install yt-dlp")
751 elif ytdlp_action == "no_pip_windows":
752 lines.append(
753 " - yt-dlp not found. Install with: pip install yt-dlp "
754 "(it may install to a Scripts directory not on PATH -- add it to PATH if YouTube search stays inactive)"
755 )
756 elif ytdlp_action == "already_installed":
757 lines.append(" - yt-dlp already installed")
758 elif results.get("ytdlp_installed", False):
759 lines.append(" - yt-dlp is installed (YouTube search ready)")
760 else:
761 lines.append(" - yt-dlp not found (install with: brew install yt-dlp)")
762
763 digg_action = results.get("digg_action", "")
764 if digg_action == "installed":
765 lines.append(" - Installed Digg CLI (free AI-news clusters source now active)")
766 elif digg_action == "already_installed":
767 lines.append(" - Digg CLI already installed (AI-news clusters active)")
768 elif digg_action == "installed_off_path":
769 digg_path = results.get("digg_path", "")
770 if digg_path:
771 bin_dir = _digg_bin_dir_hint(digg_path)
772 lines.append(
773 f" - Digg CLI found at {digg_path} but not on PATH — add "
774 f"{bin_dir} to PATH and restart your agent session/gateway "
775 "for Digg to activate"
776 )
777 else:
778 lines.append(
779 " - Digg CLI is installed but not on PATH — add its install "
780 "directory to PATH and restart your agent session/gateway for "
781 "Digg to activate"
782 )
783 elif digg_action == "install_failed":
784 lines.append(f" - Digg CLI install failed — run `{DIGG_INSTALL_CMD}` manually")
785 elif digg_action == "no_npx":
786 lines.append(
787 " - Digg CLI not installed (free, optional). Install Node/npx, then: "
788 f"{DIGG_INSTALL_CMD}"
789 )
790
791 pp_sources = results.get("pp_sources", {})
792 pp_name: dict[str, str] = {"arxiv": "arXiv", "techmeme": "Techmeme"}
793 for source_key, entry in sorted(pp_sources.items()):
794 name = pp_name.get(source_key, source_key.title())
795 action = entry.get("action", "")
796 if action == "installed":
797 lines.append(f" - Installed {name} CLI ({name} source now active)")
798 elif action == "already_installed":
799 lines.append(f" - {name} CLI already installed ({name} active)")
800 elif action == "installed_off_path":
801 path = entry.get("path", "")
802 if path:
803 lines.append(
804 f" - {name} CLI at {path} but not on PATH — add "
805 f"{os.path.dirname(os.path.expanduser(path))} to PATH and "
806 f"restart your agent session/gateway for {name} to activate"
807 )
808 else:
809 lines.append(
810 f" - {name} CLI installed but not on PATH — add its install "
811 "directory to PATH and restart your agent session/gateway for "
812 f"{name} to activate"
813 )
814 elif action == "install_failed":
815 lines.append(
816 f" - {name} CLI install failed — run "
817 f"`npx -y {PRINTING_PRESS_NPM} install {source_key} --cli-only` manually"
818 )
819 elif action == "no_npx":
820 lines.append(
821 f" - {name} CLI not installed (free, optional). Install Node/npx, "
822 f"then: `npx -y {PRINTING_PRESS_NPM} install {source_key} --cli-only`"
823 )
824
825 # Bright Data / Amazon. Reported but never installed (it spends the user's
826 # own metered credits), so the only useful thing setup can do is say
827 # precisely why the lane is or is not active -- the three states below are
828 # otherwise invisible, since SKILL.md tells the model not to raise the
829 # subject mid-run.
830 brightdata_status_entry = results.get("brightdata") or {}
831 bd_action = brightdata_status_entry.get("action", "")
832 if brightdata_status_entry.get("engine_active"):
833 lines.append(" - Bright Data CLI ready (Amazon buyer signals available)")
834 elif bd_action == "already_installed":
835 lines.append(
836 " - Bright Data CLI installed but not logged in — run "
837 "`brightdata login` to enable Amazon buyer signals (optional)"
838 )
839 elif bd_action == "installed_off_path":
840 bd_path = brightdata_status_entry.get("path", "")
841 lines.append(
842 f" - Bright Data CLI found at {bd_path} but not on PATH — add "
843 f"{os.path.dirname(os.path.expanduser(bd_path))} to PATH and restart "
844 "your agent session/gateway for Amazon buyer signals to activate"
845 )
846 elif bd_action == "not_installed":
847 lines.append(
848 " - Amazon buyer signals not installed (optional; 5,000 free "
849 "requests/month). Install with: npm i -g @brightdata/cli && brightdata login"
850 )
851
852 cookie_note = results.get("cookie_note")
853 if cookie_note:
854 # Official-only host: relay the neutral note; say nothing about
855 # browsers. The scan was not attempted, so nothing is "found".
856 lines.append(f" - {cookie_note}")
857
858 env_written = results.get("env_written", False)
859 if env_written:
860 lines.append("")
861 if cookie_note:
862 lines.append("Configuration saved.")
863 else:
864 lines.append("Configuration saved. Future runs will auto-detect your browsers.")
865
866 return "\n".join(lines)
867
868
869 # ---------------------------------------------------------------------------
870 # OpenClaw server-side setup (no browser, JSON output)
871 # ---------------------------------------------------------------------------
872
873 _OPENCLAW_KEY_NAMES = [
874 "SCRAPECREATORS_API_KEY",
875 "XAI_API_KEY",
876 "BRAVE_API_KEY",
877 "EXA_API_KEY",
878 "SERPER_API_KEY",
879 "OPENAI_API_KEY",
880 "AUTH_TOKEN",
881 ]
882
883
884 def run_openclaw_setup(config: Dict[str, Any]) -> Dict[str, Any]:
885 """Server-side setup probe: no cookies, tool + key availability, Digg CLI.
886
887 Best-effort installs digg-pp-cli when npx is available (same as desktop
888 ``run_auto_setup``). Returns a dict suitable for JSON output to stdout so
889 that SKILL.md can present appropriate options to the user.
890 """
891 yt_dlp = shutil.which("yt-dlp") is not None
892 node = shutil.which("node") is not None
893 python3 = shutil.which("python3") is not None
894
895 digg_installed, digg_action, digg_stderr, digg_path = _install_digg_cli()
896
897 keys: Dict[str, bool] = {}
898 for key_name in _OPENCLAW_KEY_NAMES:
899 short = key_name.lower().replace("_api_key", "").replace("_key", "").replace("_token", "")
900 # Normalize: AUTH_TOKEN -> auth, SCRAPECREATORS_API_KEY -> scrapecreators
901 keys[short] = bool(config.get(key_name))
902
903 # Determine x_method
904 if config.get("XAI_API_KEY"):
905 x_method: Optional[str] = "xai"
906 elif config.get("AUTH_TOKEN") and config.get("CT0"):
907 x_method = "cookies"
908 else:
909 x_method = None
910
911 payload: Dict[str, Any] = {
912 "yt_dlp": yt_dlp,
913 "node": node,
914 "python3": python3,
915 "digg_cli": digg_installed,
916 "digg_action": digg_action,
917 "keys": keys,
918 "x_method": x_method,
919 }
920 if digg_path:
921 payload["digg_path"] = digg_path
922 if digg_action == "install_failed" and digg_stderr:
923 payload["digg_stderr"] = digg_stderr
924 return payload
925
926
927 # ---------------------------------------------------------------------------
928 # Device auth flow (GitHub OAuth via ScrapeCreators)
929 # ---------------------------------------------------------------------------
930
931 _DEVICE_BASE = "https://api.scrapecreators.com/v1/github/device"
932
933 # A GitHub device code is always XXXX-XXXX (uppercase alphanumerics). We validate
934 # user_code against this before copying, labeling, or emitting it so a malformed
935 # or key-shaped value (e.g. a returning-account server response) is never
936 # mislabeled as a device code or leaked to stdout/clipboard.
937 _DEVICE_CODE_RE = re.compile(r"^[0-9A-Z]{4}-[0-9A-Z]{4}$")
938
939
940 def _existing_scrapecreators_key() -> Optional[str]:
941 """Return the SCRAPECREATORS_API_KEY already saved in the .env, if any."""
942 try:
943 from . import env as _env
944
945 if _env.CONFIG_FILE and _env.CONFIG_FILE.exists():
946 return _env.load_env_file(_env.CONFIG_FILE).get("SCRAPECREATORS_API_KEY") or None
947 except Exception as exc: # never let a config-read failure block auth
948 logger.debug("Could not read existing ScrapeCreators key: %s", exc)
949 return None
950
951
952 def _clamp_device_interval(interval: Any) -> int:
953 """Clamp a server-provided device-flow poll interval to [1, 30] seconds.
954
955 The ``interval`` comes from the server (device/code response or the
956 persisted poll handle), so 0 would hot-loop ``time.sleep``, a negative
957 would crash it, a huge value would sail past the poll timeout, and a
958 non-numeric value would raise. Defaults to 5 on missing/garbled input.
959 """
960 try:
961 value = int(interval or 5)
962 except (TypeError, ValueError):
963 return 5
964 return min(max(value, 1), 30)
965
966
967 def run_device_auth() -> Optional[Tuple[str, str, str, int]]:
968 """Start the device authorization flow.
969
970 POSTs to the ScrapeCreators device/code endpoint.
971
972 Returns:
973 (device_code, user_code, verification_uri, interval) on success,
974 None on failure.
975 """
976 try:
977 body = json.dumps({}).encode()
978 req = Request(f"{_DEVICE_BASE}/code", data=body, method="POST")
979 req.add_header("Content-Type", "application/json")
980 with urlopen(req, timeout=15) as resp:
981 data = json.loads(resp.read())
982 except (HTTPError, URLError, OSError) as exc:
983 logger.warning("Device auth code request failed: %s", exc)
984 return None
985
986 device_code = data.get("device_code")
987 user_code = data.get("user_code")
988 verification_uri = data.get("verification_uri")
989 interval = _clamp_device_interval(data.get("interval", 5))
990
991 if not device_code or not user_code:
992 # Log only the response's key names, never its values — a returning
993 # account's response could carry a raw API key we must not write to logs.
994 logger.warning(
995 "Device auth returned incomplete response (keys: %s)", sorted(data.keys())
996 )
997 return None
998
999 return (device_code, user_code, verification_uri or "", interval)
1000
1001
1002 def poll_device_auth(
1003 device_code: str,
1004 interval: int,
1005 timeout: int = 300,
1006 user_code: str = "",
1007 clipboard_ok: bool = False,
1008 ) -> Optional[str]:
1009 """Poll for an access token after the user authorizes the device.
1010
1011 Args:
1012 device_code: The device_code from run_device_auth().
1013 interval: Polling interval in seconds.
1014 timeout: Maximum time to poll in seconds.
1015 user_code: The user code to remind about during polling.
1016 clipboard_ok: Whether the code was copied to clipboard.
1017
1018 Returns:
1019 access_token on success, None on timeout or failure.
1020 """
1021 import sys
1022
1023 interval = _clamp_device_interval(interval)
1024 started_at = time.time()
1025 deadline = started_at + timeout
1026 last_reminder = started_at
1027 reminder_count = 0
1028 max_reminders = 4
1029 reminder_interval = 30 # seconds between reminders
1030
1031 while time.time() < deadline:
1032 time.sleep(interval)
1033
1034 # Periodic reminder of the code while waiting
1035 if (
1036 user_code
1037 and reminder_count < max_reminders
1038 and time.time() - last_reminder >= reminder_interval
1039 ):
1040 clipboard_hint = " (on your clipboard)" if clipboard_ok else ""
1041 print(
1042 f" Still waiting... Your code: {user_code}{clipboard_hint}",
1043 file=sys.stderr,
1044 flush=True,
1045 )
1046 last_reminder = time.time()
1047 reminder_count += 1
1048
1049 try:
1050 body = json.dumps({"device_code": device_code}).encode()
1051 req = Request(f"{_DEVICE_BASE}/token", data=body, method="POST")
1052 req.add_header("Content-Type", "application/json")
1053 with urlopen(req, timeout=15) as resp:
1054 data = json.loads(resp.read())
1055 except HTTPError as exc:
1056 if exc.code in (400, 403, 428):
1057 continue
1058 logger.warning("Device auth poll error: %s", exc)
1059 return None
1060 except (URLError, OSError):
1061 continue
1062
1063 if data.get("access_token"):
1064 return data["access_token"]
1065
1066 error = data.get("error")
1067 if error == "slow_down":
1068 interval = min(interval + 2, 30)
1069 continue
1070 if error == "authorization_pending":
1071 continue
1072 if error in ("expired_token", "access_denied"):
1073 logger.warning("Device auth failed: %s", error)
1074 return None
1075
1076 return None
1077
1078
1079 # Bounded retries for transient ScrapeCreators /profile 5xx (see #882).
1080 _PROFILE_FETCH_ATTEMPTS = 3
1081 _PROFILE_FETCH_RETRY_SLEEP_S = 1.0
1082 _PROFILE_ERROR_BODY_LIMIT = 200
1083
1084
1085 def _http_error_detail(exc: HTTPError, *, body_limit: int = _PROFILE_ERROR_BODY_LIMIT) -> str:
1086 """Build a diagnosable HTTPError string including a truncated body.
1087
1088 The body is capped and never treated as a secret source of truth; callers
1089 still must not log bearer tokens. Used so a 5xx is not a black box.
1090 """
1091 base = f"HTTP Error {exc.code}: {getattr(exc, 'reason', '') or ''}".rstrip(": ")
1092 try:
1093 raw = exc.read() or b""
1094 except Exception:
1095 return base
1096 if not raw:
1097 return base
1098 text = raw.decode("utf-8", errors="replace").strip()
1099 if not text:
1100 return base
1101 if len(text) > body_limit:
1102 text = text[:body_limit] + "…"
1103 return f"{base} body={text!r}"
1104
1105
1106 def fetch_api_key(access_token: str) -> Dict[str, Any]:
1107 """Fetch the ScrapeCreators API key using the GitHub access token.
1108
1109 GETs the device/profile endpoint with Bearer auth. Distinguishes outcomes
1110 so callers (and SKILL.md) do not collapse a server 5xx into the
1111 already-linked guidance (#882).
1112
1113 Returns a result dict:
1114 - ``{"ok": True, "api_key": str}`` on success
1115 - ``{"ok": False, "reason": "no_api_key"}`` when /profile is 2xx but has
1116 no ``api_key`` field (typical already-linked account)
1117 - ``{"ok": False, "reason": "upstream_error", "http_status": int,
1118 "detail": str}`` on 5xx after bounded retries
1119 - ``{"ok": False, "reason": "http_error", "http_status": int,
1120 "detail": str}`` on other HTTP errors (e.g. 401)
1121 - ``{"ok": False, "reason": "request_failed", "detail": str}`` on
1122 network/OS failures
1123 """
1124 data: Optional[Dict[str, Any]] = None
1125 last_upstream: Optional[Dict[str, Any]] = None
1126
1127 for attempt in range(_PROFILE_FETCH_ATTEMPTS):
1128 try:
1129 req = Request(f"{_DEVICE_BASE}/profile")
1130 req.add_header("Authorization", f"Bearer {access_token}")
1131 with urlopen(req, timeout=15) as resp:
1132 data = json.loads(resp.read())
1133 break
1134 except HTTPError as exc:
1135 detail = _http_error_detail(exc)
1136 if 500 <= int(exc.code) <= 599:
1137 logger.warning(
1138 "Device auth /profile upstream error (attempt %s/%s): %s",
1139 attempt + 1,
1140 _PROFILE_FETCH_ATTEMPTS,
1141 detail,
1142 )
1143 last_upstream = {
1144 "ok": False,
1145 "reason": "upstream_error",
1146 "http_status": int(exc.code),
1147 "detail": detail,
1148 }
1149 if attempt + 1 < _PROFILE_FETCH_ATTEMPTS:
1150 time.sleep(_PROFILE_FETCH_RETRY_SLEEP_S)
1151 continue
1152 return last_upstream
1153 logger.warning("Failed to fetch API key: %s", detail)
1154 return {
1155 "ok": False,
1156 "reason": "http_error",
1157 "http_status": int(exc.code),
1158 "detail": detail,
1159 }
1160 except (URLError, OSError) as exc:
1161 logger.warning("Failed to fetch API key: %s", exc)
1162 return {"ok": False, "reason": "request_failed", "detail": str(exc)}
1163
1164 if data is None:
1165 # Defensive: loop exited without success or an explicit return.
1166 return last_upstream or {
1167 "ok": False,
1168 "reason": "request_failed",
1169 "detail": "No profile response",
1170 }
1171
1172 api_key = data.get("api_key")
1173 if not api_key:
1174 # The /profile response parsed but carried no api_key — the common case
1175 # for a GitHub account already linked to a ScrapeCreators account. Log
1176 # the response's FIELD NAMES only (never values — the body may contain a
1177 # key under a different field) so the already-registered response shape
1178 # can be handled in a follow-up (see plan OQ1).
1179 logger.warning(
1180 "Device auth /profile returned no api_key (fields: %s)", sorted(data.keys())
1181 )
1182 return {"ok": False, "reason": "no_api_key"}
1183 return {"ok": True, "api_key": api_key}
1184
1185
1186 def _device_handle_path() -> Path:
1187 """Where run_github_start persists the device_code/interval for run_github_poll.
1188
1189 Kept next to the .env in the config dir; falls back to the OS temp dir when
1190 no config dir is resolvable (clean/no-config mode).
1191 """
1192 try:
1193 from . import env as _env
1194
1195 if _env.CONFIG_FILE:
1196 return _env.CONFIG_FILE.parent / ".github-device-handle.json"
1197 except Exception:
1198 pass
1199 import tempfile
1200
1201 return Path(tempfile.gettempdir()) / "last30days-github-device-handle.json"
1202
1203
1204 def _start_device_flow() -> "Tuple[Dict[str, Any], Optional[Dict[str, Any]]]":
1205 """Submit the GitHub device flow and surface the code, without polling.
1206
1207 Returns ``(public_result, handle)``. ``handle`` is None for the
1208 already-registered and error cases (nothing to poll); otherwise it carries
1209 the private poll state (``device_code``/``interval``/``user_code``/
1210 ``clipboard_ok``) that never belongs in the public, stdout-printed result.
1211 Callers either persist the handle to a file (``run_github_start``, for a
1212 separate poll process) or hand it straight to ``run_github_poll`` in-memory
1213 (``run_full_device_auth``, so a failed file write can't strand the one-shot).
1214 """
1215 import sys
1216 import webbrowser
1217
1218 # Already-registered short-circuit: a saved key means no device dance. The
1219 # key is returned raw here and masked at the CLI boundary before print.
1220 existing = _existing_scrapecreators_key()
1221 if existing:
1222 return (
1223 {
1224 "status": "already_registered",
1225 "method": "existing",
1226 "api_key": existing,
1227 "persisted": True,
1228 },
1229 None,
1230 )
1231
1232 result = run_device_auth()
1233 if result is None:
1234 return ({"status": "error", "message": "Failed to start device auth flow"}, None)
1235
1236 device_code, user_code, verification_uri, interval = result
1237
1238 # Validate the code shape BEFORE copying, labeling, or emitting it. A
1239 # non-conforming user_code (e.g. a key-shaped value) is never surfaced as a
1240 # GitHub device code; we stop rather than instruct the user to paste garbage.
1241 if not _DEVICE_CODE_RE.match(user_code):
1242 logger.warning("Device auth returned a non-device-shaped user_code; aborting.")
1243 return (
1244 {
1245 "status": "error",
1246 "message": "ScrapeCreators returned an unexpected device-code format.",
1247 },
1248 None,
1249 )
1250
1251 # Structured stdout line for machine consumers.
1252 print(
1253 json.dumps(
1254 {
1255 "event": "device_code_ready",
1256 "user_code": user_code,
1257 "verification_uri": verification_uri,
1258 }
1259 ),
1260 flush=True,
1261 )
1262
1263 # Copy the code to the clipboard BEFORE opening the browser.
1264 clipboard_ok = False
1265 if sys.platform == "darwin":
1266 try:
1267 subprocess.run(["pbcopy"], input=user_code.encode(), check=True, timeout=5)
1268 clipboard_ok = True
1269 except Exception:
1270 pass # pbcopy unavailable or failed, fall through
1271
1272 # Print the code as a plain HUMAN line on stdout too, so a foreground caller
1273 # sees it in the returned output even without reading the JSON. The clipboard
1274 # claim is only made when pbcopy actually succeeded (else: type it).
1275 if clipboard_ok:
1276 print(
1277 f"Your GitHub code: {user_code} (already on your clipboard - just paste it, Cmd+V)",
1278 flush=True,
1279 )
1280 else:
1281 print(f"Your GitHub code: {user_code} (type it on the GitHub page)", flush=True)
1282
1283 # Human box on stderr for direct-terminal users.
1284 clipboard_hint = " (copied to clipboard)" if clipboard_ok else ""
1285 code_line = f" Your code: {user_code}{clipboard_hint}"
1286 action_line = " Paste it on the GitHub page that just opened"
1287 width = max(len(code_line), len(action_line)) + 2
1288 border = "-" * width
1289 print(f"\n+{border}+", file=sys.stderr)
1290 print(f"|{code_line.ljust(width)}|", file=sys.stderr)
1291 print(f"|{action_line.ljust(width)}|", file=sys.stderr)
1292 print(f"+{border}+", file=sys.stderr)
1293
1294 if verification_uri:
1295 try:
1296 webbrowser.open(verification_uri)
1297 except Exception:
1298 print(f"Open: {verification_uri}", file=sys.stderr)
1299
1300 public = {
1301 "status": "awaiting_authorization",
1302 "user_code": user_code,
1303 "verification_uri": verification_uri,
1304 "clipboard_ok": clipboard_ok,
1305 }
1306 handle = {
1307 "device_code": device_code,
1308 "interval": interval,
1309 "user_code": user_code,
1310 "clipboard_ok": clipboard_ok,
1311 }
1312 return (public, handle)
1313
1314
1315 def run_github_start() -> Dict[str, Any]:
1316 """Start the device flow and persist the poll handle for a later
1317 ``run_github_poll`` process. Returns the public result (never the private
1318 device_code). See ``_start_device_flow`` for the returned statuses."""
1319 public, handle = _start_device_flow()
1320 if handle is not None:
1321 # Persist the poll handle (0o600) so a separate --github-poll process can
1322 # resume it. Best-effort: the in-memory one-shot path does not depend on
1323 # this write succeeding.
1324 path = _device_handle_path()
1325 try:
1326 path.parent.mkdir(parents=True, exist_ok=True)
1327 path.write_text(json.dumps(handle), encoding="utf-8")
1328 os.chmod(path, 0o600)
1329 except Exception as exc:
1330 logger.warning("Could not persist device handle: %s", exc)
1331 return public
1332
1333
1334 def run_github_poll(timeout: int = 300, *, _handle: Optional[Dict[str, Any]] = None) -> Dict[str, Any]:
1335 """Poll for authorization using the handle from start.
1336
1337 ``_handle`` (in-memory, from the one-shot) takes precedence over the
1338 persisted handle file. Returns success (with the fetched key), timeout, or
1339 the honest "Authorized but failed to fetch API key" branch. Deletes the
1340 persisted handle when the flow terminates.
1341 """
1342 import sys
1343
1344 if _handle is not None:
1345 data = _handle
1346 else:
1347 try:
1348 data = json.loads(_device_handle_path().read_text(encoding="utf-8"))
1349 except Exception:
1350 return {
1351 "status": "error",
1352 "message": "No pending GitHub device flow; run setup --github-start first.",
1353 }
1354
1355 device_code = data["device_code"]
1356 interval = _clamp_device_interval(data.get("interval", 5))
1357 user_code = data.get("user_code", "")
1358 # Read the real clipboard state so the polling reminder never falsely claims
1359 # the code is on the clipboard (non-macOS, or a failed pbcopy). Missing key
1360 # (older handle) defaults to False -- don't overstate.
1361 clipboard_ok = bool(data.get("clipboard_ok", False))
1362
1363 print("Waiting for authorization...", file=sys.stderr, flush=True)
1364 access_token = poll_device_auth(
1365 device_code, interval, timeout=timeout, user_code=user_code, clipboard_ok=clipboard_ok
1366 )
1367
1368 def _cleanup() -> None:
1369 try:
1370 _device_handle_path().unlink()
1371 except Exception:
1372 pass
1373
1374 if access_token is None:
1375 _cleanup()
1376 return {"status": "timeout", "user_code": user_code}
1377
1378 fetched = fetch_api_key(access_token)
1379 _cleanup()
1380 if fetched.get("ok") and fetched.get("api_key"):
1381 return {
1382 "status": "success",
1383 "method": "device",
1384 "api_key": fetched["api_key"],
1385 "user_code": user_code,
1386 }
1387
1388 # Discriminate failure modes so SKILL.md does not misdiagnose a 5xx as
1389 # "GitHub already linked" (#882). Keep the historical message only for the
1390 # 2xx-without-api_key / already-linked case.
1391 reason = fetched.get("reason") or "request_failed"
1392 if reason == "no_api_key":
1393 return {
1394 "status": "error",
1395 "message": "Authorized but failed to fetch API key",
1396 "reason": "no_api_key",
1397 }
1398 if reason == "upstream_error":
1399 http_status = fetched.get("http_status")
1400 return {
1401 "status": "error",
1402 "message": f"Authorized but ScrapeCreators profile failed (HTTP {http_status})",
1403 "reason": "upstream_error",
1404 "http_status": http_status,
1405 "detail": fetched.get("detail"),
1406 }
1407 if reason == "http_error":
1408 http_status = fetched.get("http_status")
1409 return {
1410 "status": "error",
1411 "message": f"Authorized but failed to fetch API key (HTTP {http_status})",
1412 "reason": "http_error",
1413 "http_status": http_status,
1414 "detail": fetched.get("detail"),
1415 }
1416 return {
1417 "status": "error",
1418 "message": "Authorized but failed to fetch API key",
1419 "reason": reason,
1420 "detail": fetched.get("detail"),
1421 }
1422
1423
1424 def run_full_device_auth(timeout: int = 300) -> Dict[str, Any]:
1425 """Back-compat one-shot: start the device flow, then poll to completion.
1426
1427 Passes the poll handle to ``run_github_poll`` IN MEMORY, so a failed handle-
1428 file write can't strand the one-shot. Kept so callers of ``setup --github`` /
1429 ``--device-auth`` still work; the model-driven wizard uses the two-command
1430 split (start then poll) instead.
1431 """
1432 public, handle = _start_device_flow()
1433 if handle is None:
1434 return public # already_registered or error
1435 return run_github_poll(timeout=timeout, _handle=handle)
1436
1437
1438 # ---------------------------------------------------------------------------
1439 # Unified GitHub auth
1440 # ---------------------------------------------------------------------------
1441
1442
1443 def run_github_auth(timeout: int = 300) -> Dict[str, Any]:
1444 """Run the --github setup path via device auth (one-shot, back-compat).
1445
1446 The existing-key short-circuit now lives in run_github_start; this delegates
1447 to the start+poll chain. This path must not read or forward local GitHub CLI
1448 tokens.
1449 """
1450 return run_full_device_auth(timeout=timeout)
1451
1451 lines PYTHON