返回 CodeWhale
catalog_models_dev.py
根目录 / scripts / catalog_models_dev.py
1 #!/usr/bin/env python3
2 """Models.dev catalog refresh / snapshot automation for CodeWhale (#4117).
3
4 Fetches the public Models.dev combined catalog, validates offline bundled seed
5 shape, supports OpenRouter public-listing inspection, and generates the offline
6 seed (#6396). It never accepts, prints, or persists API keys / auth headers.
7 The only thing it writes from fetched JSON is the seed lock: the rows the spec
8 references, projected onto allowlisted fields with credential-shaped keys
9 scrubbed. The raw document is never written.
10
11 Usage examples:
12
13 # Regenerate the offline seed (see docs/CATALOG_REFRESH.md)
14 scripts/catalog_models_dev.py seed lock --dry-run # review report only
15 scripts/catalog_models_dev.py seed lock # pin upstream rows
16 scripts/catalog_models_dev.py seed render # write the seed
17 scripts/catalog_models_dev.py seed render --check # CI: seed == render
18
19 # Dry-run: fetch + validate, print counts (no write)
20 scripts/catalog_models_dev.py refresh
21
22 # Validate the in-repo offline seed still parses as Models.dev-shaped JSON
23 scripts/catalog_models_dev.py snapshot --check \\
24 crates/config/assets/models_dev.bundled.json
25
26 # OpenRouter public /models listing (no key), dry-run only
27 scripts/catalog_models_dev.py refresh --provider openrouter \\
28 --sort newest --limit 100
29
30 Environment:
31 CODEWHALE_MODELS_DEV_URL Override Models.dev catalog URL
32 CODEWHALE_MODELS_DEV_PATH Read catalog JSON from a local file instead of network
33 """
34
35 from __future__ import annotations
36
37 import argparse
38 import difflib
39 import hashlib
40 import json
41 import os
42 import re
43 import sys
44 import urllib.error
45 import urllib.request
46 from datetime import datetime, timezone
47 from pathlib import Path
48 from typing import Any
49
50 DEFAULT_MODELS_DEV_URL = "https://models.dev/catalog.json"
51 DEFAULT_OPENROUTER_MODELS_URL = "https://openrouter.ai/api/v1/models"
52 USER_AGENT = "CodeWhale-catalog-automation/0.9.0 (+https://github.com/codewhale-hq/CodeWhale)"
53 FETCH_TIMEOUT_SECS = 60
54
55
56 def die(msg: str, code: int = 1) -> None:
57 print(f"error: {msg}", file=sys.stderr)
58 raise SystemExit(code)
59
60
61 def load_json_bytes(raw: bytes, source: str) -> Any:
62 try:
63 text = raw.decode("utf-8")
64 except UnicodeDecodeError as exc:
65 die(f"{source}: not utf-8 ({exc})")
66 try:
67 return json.loads(text)
68 except json.JSONDecodeError as exc:
69 die(f"{source}: invalid JSON ({exc})")
70
71
72 def fetch_url(url: str) -> bytes:
73 req = urllib.request.Request(
74 url,
75 headers={
76 "User-Agent": USER_AGENT,
77 "Accept": "application/json",
78 # Explicitly no Authorization header — public endpoints only.
79 },
80 method="GET",
81 )
82 try:
83 with urllib.request.urlopen(req, timeout=FETCH_TIMEOUT_SECS) as resp:
84 # Refuse to follow into non-JSON surprise payloads larger than 64 MiB.
85 data = resp.read(64 * 1024 * 1024 + 1)
86 if len(data) > 64 * 1024 * 1024:
87 die(f"{url}: response exceeds 64 MiB safety cap")
88 ctype = resp.headers.get("Content-Type", "")
89 if "json" not in ctype.lower() and not data.lstrip().startswith((b"{", b"[")):
90 die(f"{url}: unexpected Content-Type {ctype!r}")
91 return data
92 except urllib.error.HTTPError as exc:
93 die(f"{url}: HTTP {exc.code} {exc.reason}")
94 except urllib.error.URLError as exc:
95 die(f"{url}: {exc.reason}")
96
97
98 def load_models_dev_catalog() -> tuple[dict[str, Any], str, bool]:
99 """Return (document, source_label, is_local_file).
100
101 Network fetches are dry-run only for write paths: CodeQL treats remote JSON
102 as potentially sensitive, and Models.dev is large enough that maintainers
103 should stage via CODEWHALE_MODELS_DEV_PATH before writing a cache/snapshot.
104 """
105 path_override = os.environ.get("CODEWHALE_MODELS_DEV_PATH", "").strip()
106 if path_override:
107 p = Path(path_override)
108 if not p.is_file():
109 die(f"CODEWHALE_MODELS_DEV_PATH not a file: {p}")
110 raw = p.read_bytes()
111 data = load_json_bytes(raw, str(p))
112 return ensure_models_dev_shape(data, str(p)), f"file:{p}", True
113
114 url = os.environ.get("CODEWHALE_MODELS_DEV_URL", DEFAULT_MODELS_DEV_URL).strip()
115 if not url:
116 url = DEFAULT_MODELS_DEV_URL
117 raw = fetch_url(url)
118 data = load_json_bytes(raw, url)
119 return ensure_models_dev_shape(data, url), f"url:{url}", False
120
121
122 def ensure_models_dev_shape(data: Any, source: str) -> dict[str, Any]:
123 if not isinstance(data, dict):
124 die(f"{source}: expected object root")
125 # Allow optional _meta (CodeWhale offline seed) and require models+providers
126 # when present so we never write a partial secret leak document.
127 models = data.get("models")
128 providers = data.get("providers")
129 if models is None and providers is None:
130 die(f"{source}: missing both 'models' and 'providers'")
131 if models is not None and not isinstance(models, dict):
132 die(f"{source}: 'models' must be an object")
133 if providers is not None and not isinstance(providers, dict):
134 die(f"{source}: 'providers' must be an object")
135 # Rebuild a public document from allowlisted top-level keys only so we never
136 # persist credential-shaped fields even if a future Models.dev field adds them.
137 return public_models_dev_document(data)
138
139
140 def is_credential_key(key: str) -> bool:
141 banned_exact = {
142 "api_key",
143 "apikey",
144 "authorization",
145 "token",
146 "access_token",
147 "refresh_token",
148 "secret",
149 "password",
150 "client_secret",
151 }
152 lowered = key.lower()
153 return lowered in banned_exact or lowered.endswith("_api_key") or lowered.endswith("_secret")
154
155
156 def strip_sensitive_fields(node: Any) -> Any:
157 """Drop keys that look like credentials; never persist auth material."""
158 if isinstance(node, dict):
159 out: dict[str, Any] = {}
160 for key, value in node.items():
161 if not isinstance(key, str) or is_credential_key(key):
162 continue
163 out[key] = strip_sensitive_fields(value)
164 return out
165 if isinstance(node, list):
166 return [strip_sensitive_fields(item) for item in node]
167 if isinstance(node, (str, int, float, bool)) or node is None:
168 return node
169 # Drop non-JSON-scalar oddities rather than serializing them.
170 return None
171
172
173 def public_models_dev_document(data: dict[str, Any]) -> dict[str, Any]:
174 """Construct a write-safe Models.dev-shaped document (public metadata only)."""
175 out: dict[str, Any] = {}
176 if isinstance(data.get("_meta"), dict):
177 out["_meta"] = strip_sensitive_fields(data["_meta"])
178 if isinstance(data.get("models"), dict):
179 out["models"] = strip_sensitive_fields(data["models"])
180 if isinstance(data.get("_reviewed"), dict):
181 out["_reviewed"] = strip_sensitive_fields(data["_reviewed"])
182 if isinstance(data.get("providers"), dict):
183 out["providers"] = strip_sensitive_fields(data["providers"])
184 return out
185
186
187 def public_source_label(source: str) -> str:
188 """Log a catalog origin without query/fragment (tokens live there)."""
189 if source.startswith("url:"):
190 url = source[4:]
191 for sep in ("?", "#"):
192 url = url.split(sep, 1)[0]
193 return f"url:{url}"
194 return source
195
196
197 def public_limit_value(value: Any) -> str:
198 """Format a catalog limit for logs. Never print credential-shaped strings.
199
200 Remote catalog JSON is tainted for clear-text-logging rules. Only numeric
201 limits are meaningful here; anything else (including token-shaped strings)
202 is replaced with a constant so the raw value cannot reach stdout.
203 """
204 if isinstance(value, bool):
205 return "redacted"
206 if value is None:
207 return "null"
208 if isinstance(value, int):
209 return str(value)
210 if isinstance(value, float):
211 return format(value, ".6g")
212 return "redacted"
213
214
215 def catalog_stats(data: dict[str, Any]) -> str:
216 models = data.get("models") or {}
217 providers = data.get("providers") or {}
218 offerings = 0
219 if isinstance(providers, dict):
220 for prov in providers.values():
221 if isinstance(prov, dict):
222 models_map = prov.get("models") or {}
223 if isinstance(models_map, dict):
224 offerings += len(models_map)
225 return (
226 f"providers={len(providers) if isinstance(providers, dict) else 0} "
227 f"canonical_models={len(models) if isinstance(models, dict) else 0} "
228 f"provider_offerings={offerings}"
229 )
230
231
232
233 def cmd_refresh(args: argparse.Namespace) -> None:
234 if args.provider and args.provider.lower() == "openrouter":
235 refresh_openrouter(args)
236 return
237 if args.provider:
238 die(
239 f"unsupported --provider {args.provider!r} "
240 "(supported: openrouter, or omit for Models.dev)"
241 )
242
243 data, source, _is_local = load_models_dev_catalog()
244 print(f"loaded Models.dev catalog from {source}")
245 print(catalog_stats(data))
246 if args.write_cache or args.write:
247 die(
248 "disk writes are intentionally unsupported (secret-free by design); "
249 "to update the offline seed use `seed lock` then `seed render`"
250 )
251 print("dry-run complete (no secrets; no disk write)")
252
253
254 def refresh_openrouter(args: argparse.Namespace) -> None:
255 url = DEFAULT_OPENROUTER_MODELS_URL
256 raw = fetch_url(url)
257 data = load_json_bytes(raw, url)
258 if not isinstance(data, dict) or "data" not in data:
259 die(f"{url}: expected {{ data: [...] }} envelope")
260 rows = data["data"]
261 if not isinstance(rows, list):
262 die(f"{url}: data is not a list")
263
264 # Optional sort / limit for local inspection — never secrets.
265 if args.sort == "newest":
266 def created_key(row: Any) -> float:
267 if not isinstance(row, dict):
268 return 0.0
269 created = row.get("created")
270 try:
271 return float(created)
272 except (TypeError, ValueError):
273 return 0.0
274
275 rows = sorted(rows, key=created_key, reverse=True)
276 if args.limit is not None and args.limit > 0:
277 rows = rows[: args.limit]
278
279 # Project only public catalog fields — never the raw response object —
280 # so credential-shaped keys cannot reach disk even if OpenRouter adds them.
281 public_rows: list[dict[str, Any]] = []
282 allowed = {
283 "id",
284 "name",
285 "created",
286 "description",
287 "context_length",
288 "architecture",
289 "pricing",
290 "top_provider",
291 "per_request_limits",
292 "supported_parameters",
293 }
294 for row in rows:
295 if not isinstance(row, dict):
296 continue
297 projected: dict[str, Any] = {}
298 for key in allowed:
299 if key in row and not is_credential_key(key):
300 projected[key] = strip_sensitive_fields(row[key])
301 if projected.get("id"):
302 public_rows.append(projected)
303 payload = {
304 "_meta": {
305 "source": "openrouter.ai/api/v1/models",
306 "note": "Public model listing for cache dogfood; not the Models.dev SoT.",
307 "count": len(public_rows),
308 "sort": args.sort,
309 "limit": args.limit,
310 },
311 "data": public_rows,
312 }
313 print(f"loaded OpenRouter models: {len(public_rows)} rows (sort={args.sort}, limit={args.limit})")
314 if args.write_cache:
315 # OpenRouter listing is always network-sourced; avoid disk write of remote JSON.
316 die(
317 "OpenRouter refresh is dry-run only (no disk write). "
318 "Use Models.dev with CODEWHALE_MODELS_DEV_PATH for offline snapshots."
319 )
320 else:
321 print("dry-run complete (OpenRouter writes disabled; use Models.dev local path for caches)")
322 _ = payload # keep payload construction for future offline path
323
324
325 def cmd_snapshot(args: argparse.Namespace) -> None:
326 target = Path(args.path)
327 if args.check:
328 if not target.is_file():
329 die(f"--check: missing {target}")
330 raw = target.read_bytes()
331 data = load_json_bytes(raw, str(target))
332 ensure_models_dev_shape(data, str(target))
333 print(f"ok: {target} is Models.dev-shaped ({catalog_stats(data)})")
334 return
335
336 data, source, _is_local = load_models_dev_catalog()
337 print(f"loaded Models.dev catalog from {source}")
338 print(catalog_stats(data))
339 if args.write or args.force_full:
340 die(
341 "disk writes are intentionally unsupported for this automation; "
342 "the offline seed is generated: use `seed lock` then `seed render`"
343 )
344 print("dry-run complete (use --check PATH to validate an existing snapshot)")
345
346
347 # ---------------------------------------------------------------------------
348 # Offline seed generator (#6396)
349 #
350 # The offline seed (crates/config/assets/models_dev.bundled.json) is generated,
351 # never hand-edited:
352 #
353 # spec scripts/catalog/models_dev_seed.toml what to carry (reviewed)
354 # lock scripts/catalog/models_dev_seed.lock.json upstream rows, pinned
355 # seed crates/config/assets/models_dev.bundled.json = render(spec, lock)
356 #
357 # `seed lock` is the only network step and the review step: it fetches
358 # Models.dev, keeps only the rows the spec references (allowlisted fields,
359 # credential-shaped keys scrubbed), and prints what changed. `seed render` is
360 # offline and deterministic; CI runs `seed render --check`.
361 #
362 # The spec selects and maps; it never states a value that contradicts
363 # upstream. Policy (a withheld price, a clamped limit) lives in the runtime
364 # corrections file, crates/config/assets/catalog_corrections.json, so it
365 # holds online as well as offline. The one additive exception is a `curated`
366 # row for a model upstream does not list yet; `seed lock` fails once upstream
367 # lists it, so a curated row can never shadow an upstream fact.
368 # ---------------------------------------------------------------------------
369
370 SEED_SPEC = Path("scripts/catalog/models_dev_seed.toml")
371 SEED_LOCK = Path("scripts/catalog/models_dev_seed.lock.json")
372 SEED_ASSET = Path("crates/config/assets/models_dev.bundled.json")
373 CORRECTIONS_ASSET = Path("crates/config/assets/catalog_corrections.json")
374 SEED_RENDER_COMMAND = "python3 scripts/catalog_models_dev.py seed render"
375
376 # Exactly the fields crates/config/src/models_dev.rs reads, in render order.
377 PROVIDER_MODEL_FIELDS = (
378 "id",
379 "base_model",
380 "name",
381 "family",
382 "default",
383 "attachment",
384 "reasoning",
385 "reasoning_options",
386 "interleaved",
387 "tool_call",
388 "structured_output",
389 "temperature",
390 "open_weights",
391 "modalities",
392 "limit",
393 "cost",
394 )
395 CANONICAL_MODEL_FIELDS = (
396 "id",
397 "name",
398 "family",
399 "attachment",
400 "reasoning",
401 "tool_call",
402 "structured_output",
403 "temperature",
404 "open_weights",
405 "modalities",
406 "limit",
407 )
408 NESTED_FIELDS = {
409 "limit": ("context", "input", "output"),
410 "modalities": ("input", "output"),
411 "cost": ("input", "output", "cache_read", "cache_write"),
412 }
413 # Mapping-only keys a spec model entry may carry. None of them restates an
414 # upstream fact: `base_model` is Codewhale's canonical join, which upstream
415 # provider rows do not carry.
416 SPEC_MODEL_KEYS = {"id", "upstream_id", "from", "base_model", "curated"}
417 SPEC_PROVIDER_KEYS = {"id", "upstream", "name", "api", "npm", "env", "doc", "default", "models"}
418 SAFE_PUBLIC_TEXT = re.compile(r"^[A-Za-z0-9 ._:/()+\-]{0,64}$")
419
420
421 def project_fields(row: Any, fields: tuple[str, ...]) -> dict[str, Any]:
422 """Keep only allowlisted fields, in a fixed order, scrubbed."""
423 if not isinstance(row, dict):
424 return {}
425 out: dict[str, Any] = {}
426 for field in fields:
427 if field not in row or row[field] is None:
428 continue
429 value = strip_sensitive_fields(row[field])
430 nested = NESTED_FIELDS.get(field)
431 if nested is not None:
432 if not isinstance(value, dict):
433 continue
434 value = {key: value[key] for key in nested if value.get(key) is not None}
435 if not value:
436 continue
437 out[field] = value
438 return out
439
440
441 def public_value(value: Any) -> str:
442 """Format an upstream value for the review report without echoing secrets."""
443 if isinstance(value, bool):
444 return "true" if value else "false"
445 if value is None or isinstance(value, (int, float)):
446 return public_limit_value(value)
447 if isinstance(value, str):
448 if SAFE_PUBLIC_TEXT.match(value) and not value.lower().startswith(("sk-", "bearer")):
449 return value
450 return "redacted"
451 if isinstance(value, list):
452 return "[" + ", ".join(public_value(item) for item in value) + "]"
453 if isinstance(value, dict):
454 return "{" + ", ".join(
455 f"{public_value(key)}: {public_value(item)}" for key, item in value.items()
456 ) + "}"
457 return "redacted"
458
459
460 def flatten(value: Any, prefix: str = "") -> dict[str, Any]:
461 if isinstance(value, dict):
462 out: dict[str, Any] = {}
463 for key, item in value.items():
464 out.update(flatten(item, f"{prefix}.{key}" if prefix else str(key)))
465 return out
466 return {prefix: value}
467
468
469 def load_seed_spec(path: Path) -> dict[str, Any]:
470 import tomllib
471
472 if not path.is_file():
473 die(f"seed spec missing: {path}")
474 try:
475 spec = tomllib.loads(path.read_text(encoding="utf-8"))
476 except tomllib.TOMLDecodeError as exc:
477 die(f"{path}: invalid TOML ({exc})")
478 return normalize_seed_spec(spec, str(path))
479
480
481 def normalize_seed_spec(spec: dict[str, Any], source: str) -> dict[str, Any]:
482 """Validate the spec and expand shorthand model entries."""
483 if not isinstance(spec.get("source"), dict) or not spec["source"].get("url"):
484 die(f"{source}: [source].url is required")
485 meta = spec.get("meta", {})
486 if not isinstance(meta, dict) or not all(isinstance(v, str) for v in meta.values()):
487 die(f"{source}: [meta] values must be strings")
488 curated: dict[tuple[str, str], dict[str, Any]] = {}
489 for entry in spec.get("curated", []):
490 key = (entry.get("provider", ""), entry.get("id", ""))
491 if not all(key) or not str(entry.get("reason", "")).strip():
492 die(f"{source}: curated rows need provider, id and a reason")
493 if not isinstance(entry.get("row"), dict):
494 die(f"{source}: curated {key[0]}/{key[1]} needs a [curated.row] table")
495 if key in curated:
496 die(f"{source}: curated {key[0]}/{key[1]} is listed twice")
497 curated[key] = entry
498 providers = []
499 seen_providers: set[str] = set()
500 for provider in spec.get("providers", []):
501 unknown = set(provider) - SPEC_PROVIDER_KEYS
502 pid = provider.get("id", "")
503 if unknown:
504 die(f"{source}: provider {pid} has unknown keys {sorted(unknown)}")
505 if not pid or pid in seen_providers:
506 die(f"{source}: provider ids must be present and unique ({pid!r})")
507 seen_providers.add(pid)
508 models = []
509 seen_models: set[str] = set()
510 for raw in provider.get("models", []):
511 entry = {"id": raw} if isinstance(raw, str) else dict(raw)
512 unknown = set(entry) - SPEC_MODEL_KEYS
513 if unknown:
514 die(
515 f"{source}: {pid}/{entry.get('id')} has unknown keys {sorted(unknown)}; "
516 "the spec maps rows, it never restates upstream values "
517 "(policy goes in crates/config/assets/catalog_corrections.json)"
518 )
519 mid = entry.get("id", "")
520 if not mid or mid in seen_models:
521 die(f"{source}: {pid} model ids must be present and unique ({mid!r})")
522 seen_models.add(mid)
523 if entry.get("curated") and (pid, mid) not in curated:
524 die(f"{source}: {pid}/{mid} is marked curated but has no [[curated]] row")
525 if not entry.get("curated") and (pid, mid) in curated:
526 die(f"{source}: {pid}/{mid} has a [[curated]] row but is not marked curated")
527 models.append(entry)
528 defaults = [m["id"] for m in models if m["id"] == provider.get("default")]
529 if len(defaults) != 1:
530 die(f"{source}: provider {pid} must name exactly one default among its models")
531 providers.append({**provider, "upstream": provider.get("upstream", pid), "models": models})
532 used_curated = {
533 (p["id"], m["id"]) for p in providers for m in p["models"] if m.get("curated")
534 }
535 for key in curated:
536 if key not in used_curated:
537 die(f"{source}: curated {key[0]}/{key[1]} is not listed under its provider")
538 canonical = []
539 for entry in spec.get("canonical", []):
540 entry = {"key": entry} if isinstance(entry, str) else dict(entry)
541 if not entry.get("key"):
542 die(f"{source}: canonical entries need a key")
543 canonical.append({"key": entry["key"], "upstream": entry.get("upstream", entry["key"])})
544 return {
545 "source": spec["source"],
546 "meta": meta,
547 "providers": providers,
548 "curated": curated,
549 "canonical": canonical,
550 }
551
552
553 def upstream_ref(provider: dict[str, Any], model: dict[str, Any]) -> tuple[str, str]:
554 """The (upstream provider, upstream model id) a spec row projects."""
555 return (
556 model.get("from", provider["upstream"]),
557 model.get("upstream_id", model["id"]),
558 )
559
560
561 def find_row(rows: dict[str, Any], model_id: str) -> tuple[str, Any] | None:
562 """Exact id first, then a unique case-insensitive match (`GLM-5.2` ~ `glm-5.2`)."""
563 if model_id in rows:
564 return model_id, rows[model_id]
565 matches = [key for key in rows if key.lower() == model_id.lower()]
566 if len(matches) == 1:
567 return matches[0], rows[matches[0]]
568 return None
569
570
571 def build_seed_lock(
572 spec: dict[str, Any], upstream: dict[str, Any], raw: bytes, fetched_at: str
573 ) -> tuple[dict[str, Any], list[str]]:
574 """Project the spec's referenced upstream rows. Returns (lock, errors)."""
575 errors: list[str] = []
576 up_providers = upstream.get("providers") or {}
577 up_models = upstream.get("models") or {}
578 lock_providers: dict[str, dict[str, Any]] = {}
579 for provider in spec["providers"]:
580 for model in provider["models"]:
581 up_provider, up_id = upstream_ref(provider, model)
582 rows = (up_providers.get(up_provider) or {}).get("models") or {}
583 found = find_row(rows, up_id)
584 label = f"{provider['id']}/{model['id']}"
585 if model.get("curated"):
586 if found is not None:
587 errors.append(
588 f"{label}: curated, but upstream {up_provider} now lists "
589 f"{found[0]}; drop the [[curated]] row and derive it"
590 )
591 continue
592 if found is None:
593 errors.append(f"{label}: upstream {up_provider} does not list {up_id}")
594 continue
595 lock_providers.setdefault(up_provider, {})[found[0]] = project_fields(
596 found[1], PROVIDER_MODEL_FIELDS
597 )
598 lock_models: dict[str, Any] = {}
599 for entry in spec["canonical"]:
600 row = up_models.get(entry["upstream"])
601 if row is None:
602 errors.append(f"canonical {entry['key']}: upstream does not list {entry['upstream']}")
603 continue
604 lock_models[entry["upstream"]] = project_fields(row, CANONICAL_MODEL_FIELDS)
605 lock = {
606 "source": {
607 "url": spec["source"]["url"],
608 "fetched_at": fetched_at,
609 "sha256": hashlib.sha256(raw).hexdigest(),
610 },
611 "models": dict(sorted(lock_models.items())),
612 "providers": {
613 key: dict(sorted(rows.items())) for key, rows in sorted(lock_providers.items())
614 },
615 }
616 return lock, errors
617
618
619 def seed_lock_report(
620 spec: dict[str, Any],
621 old_lock: dict[str, Any] | None,
622 new_lock: dict[str, Any],
623 upstream: dict[str, Any],
624 corrections: dict[str, Any] | None,
625 ) -> list[str]:
626 """Human review lines for the PR body: what a re-lock changes."""
627 lines: list[str] = []
628 old_rows = flatten((old_lock or {}).get("providers", {}))
629 old_rows.update(flatten({"models": (old_lock or {}).get("models", {})}))
630 new_rows = flatten(new_lock.get("providers", {}))
631 new_rows.update(flatten({"models": new_lock.get("models", {})}))
632 changed = [
633 f" {path}: {public_value(old_rows.get(path))} -> {public_value(new_rows.get(path))}"
634 for path in sorted(set(old_rows) | set(new_rows))
635 if old_rows.get(path) != new_rows.get(path)
636 ]
637 lines.append(f"field changes: {len(changed)}")
638 lines.extend(changed)
639
640 if corrections:
641 stale = stale_corrections(spec, new_lock, corrections)
642 lines.append(f"stale corrections (upstream now agrees; delete them): {len(stale)}")
643 lines.extend(f" {line}" for line in stale)
644
645 referenced = {
646 upstream_ref(p, m)[0]: set() for p in spec["providers"] for m in p["models"]
647 }
648 for provider in spec["providers"]:
649 for model in provider["models"]:
650 up_provider, _ = upstream_ref(provider, model)
651 referenced[up_provider].update(new_lock["providers"].get(up_provider, {}))
652 new_upstream: list[str] = []
653 for up_provider, carried in sorted(referenced.items()):
654 rows = ((upstream.get("providers") or {}).get(up_provider) or {}).get("models") or {}
655 extra = sorted(set(rows) - carried)
656 if extra:
657 shown = ", ".join(public_value(model) for model in extra[:8])
658 more = f" (+{len(extra) - 8} more)" if len(extra) > 8 else ""
659 new_upstream.append(f" {up_provider}: {len(extra)} not carried: {shown}{more}")
660 lines.append("upstream models not carried (information only):")
661 lines.extend(new_upstream or [" none"])
662 return lines
663
664
665 def stale_corrections(
666 spec: dict[str, Any], lock: dict[str, Any], corrections: dict[str, Any]
667 ) -> list[str]:
668 """Corrections whose patched value upstream now states itself."""
669 rows_by_provider: dict[str, dict[str, Any]] = {}
670 for provider in spec["providers"]:
671 rows = rows_by_provider.setdefault(provider["id"], {})
672 for model in provider["models"]:
673 up_provider, up_id = upstream_ref(provider, model)
674 found = find_row(lock["providers"].get(up_provider, {}), up_id)
675 if found is not None:
676 rows[model["id"]] = found[1]
677 stale: list[str] = []
678 for rule in corrections.get("providers", []):
679 rows = rows_by_provider.get(rule.get("provider"), {})
680 if rows and not any(row.get("cost") for row in rows.values()):
681 stale.append(f"{rule['provider']}: pricing_withheld, but no carried row has a price")
682 for fix in corrections.get("models", []):
683 row = rows_by_provider.get(fix.get("provider"), {}).get(fix.get("id"))
684 if row is None:
685 continue
686 limit = row.get("limit") or {}
687 label = f"{fix['provider']}/{fix['id']}"
688 if "max_output" in fix and limit.get("output") == fix["max_output"]:
689 stale.append(f"{label}: max_output {fix['max_output']} equals upstream")
690 if "context_window" in fix and limit.get("context") == fix["context_window"]:
691 stale.append(f"{label}: context_window {fix['context_window']} equals upstream")
692 if "pricing_withheld" in fix and not row.get("cost"):
693 stale.append(f"{label}: pricing_withheld, but upstream lists no price")
694 if "reasoning_options" in fix and row.get("reasoning_options") == fix["reasoning_options"]:
695 stale.append(f"{label}: reasoning_options equal upstream")
696 return stale
697
698
699 def validate_reviewed(data: Any) -> dict[str, Any]:
700 """Refuse malformed authored supplements before the deterministic render."""
701 def identifier(value: Any) -> bool:
702 return isinstance(value, str) and bool(value) and len(value.encode("utf-8")) <= 512 and value == value.strip() and not any(ord(c) < 32 or 127 <= ord(c) <= 159 for c in value)
703
704 def positive(value: Any) -> bool:
705 return value is None or isinstance(value, int) and not isinstance(value, bool) and 0 < value <= 0xffffffff
706
707 if not isinstance(data, dict) or not identifier(data.get("revision")):
708 die("reviewed model catalog missing or malformed")
709 intrinsic = data.get("intrinsic", {})
710 if not isinstance(intrinsic, dict):
711 die("reviewed intrinsic facts must be an object")
712 for key, row in intrinsic.items():
713 if not identifier(key) or not isinstance(row, dict) or not identifier(row.get("source")) or any(not positive(row.get(field)) for field in ("context_window", "max_output", "generation_default")) or not (row.get("reasoning") is None or isinstance(row.get("reasoning"), bool)):
714 die("malformed reviewed intrinsic fact")
715 public = data.get("public_models", [])
716 if not isinstance(public, list):
717 die("reviewed public models must be an array")
718 seen: set[str] = set()
719 for row in public:
720 if not isinstance(row, dict) or not identifier(row.get("id")) or row["id"] in seen or row["id"].lower() not in intrinsic:
721 die("missing or duplicate reviewed public model")
722 seen.add(row["id"])
723 for key in ("compatibility_aliases", "completion_rosters", "constants", "groups"):
724 if not isinstance(data.get(key, {}), dict):
725 die(f"reviewed {key} must be an object")
726 for aliases in data.get("compatibility_aliases", {}).values():
727 if not isinstance(aliases, dict) or any(not identifier(key) or not identifier(value) for key, value in aliases.items()):
728 die("malformed reviewed alias")
729 for entries in data.get("completion_rosters", {}).values():
730 if not isinstance(entries, list) or any(not identifier(value) for value in entries):
731 die("malformed reviewed completion roster")
732 for reference in data.get("numeric_refs", {}).values():
733 if not isinstance(reference, dict) or reference.get("field") not in ("context_window", "max_output", "generation_default") or reference.get("model") not in intrinsic or not positive(intrinsic[reference["model"]].get(reference["field"])) or intrinsic[reference["model"]].get(reference["field"]) is None:
734 die("missing or malformed numeric model contract")
735 return data
736
737
738 def render_seed(spec: dict[str, Any], lock: dict[str, Any], reviewed_source: dict[str, Any]) -> str:
739 """Render the offline seed from spec + lock. Pure and deterministic."""
740 errors: list[str] = []
741 providers_out: dict[str, Any] = {}
742 row_count = 0
743 for provider in spec["providers"]:
744 models_out: dict[str, Any] = {}
745 for model in provider["models"]:
746 key = (provider["id"], model["id"])
747 up_provider, up_id = upstream_ref(provider, model)
748 locked = find_row(lock["providers"].get(up_provider, {}), up_id)
749 if model.get("curated"):
750 if locked is not None:
751 errors.append(f"{key[0]}/{key[1]}: curated row also present in the lock")
752 continue
753 row = project_fields(spec["curated"][key]["row"], PROVIDER_MODEL_FIELDS)
754 elif locked is None:
755 errors.append(
756 f"{key[0]}/{key[1]}: not in the lock (run `python3 scripts/catalog_models_dev.py seed lock`)"
757 )
758 continue
759 else:
760 row = dict(locked[1])
761 row["id"] = model["id"]
762 if model.get("base_model"):
763 row["base_model"] = model["base_model"]
764 row.pop("default", None)
765 if model["id"] == provider["default"]:
766 row["default"] = True
767 models_out[model["id"]] = project_fields(row, PROVIDER_MODEL_FIELDS)
768 row_count += 1
769 header = {"id": provider["id"]}
770 for field in ("name", "api", "npm", "doc", "env"):
771 if field in provider:
772 header[field] = provider[field]
773 providers_out[provider["id"]] = {**header, "models": models_out}
774 models_out = {}
775 for entry in spec["canonical"]:
776 row = lock["models"].get(entry["upstream"])
777 if row is None:
778 errors.append(f"canonical {entry['key']}: not in the lock")
779 continue
780 models_out[entry["key"]] = project_fields({**row, "id": entry["key"]}, CANONICAL_MODEL_FIELDS)
781 if errors:
782 die("seed render refused:\n " + "\n ".join(errors))
783 meta = dict(spec["meta"])
784 meta["generated_by"] = (
785 f"{SEED_RENDER_COMMAND} from {SEED_SPEC} and {SEED_LOCK}. "
786 "Do not edit this file by hand; see docs/CATALOG_REFRESH.md."
787 )
788 source = lock["source"]
789 meta["upstream"] = f"{source['url']} fetched {source['fetched_at']} sha256 {source['sha256']}"
790 meta["coverage"] = (
791 f"{len(providers_out)} providers, {row_count} provider model rows, "
792 f"{len(models_out)} canonical model entries."
793 )
794 reviewed_source = validate_reviewed(reviewed_source)
795 document = {"_meta": meta, "models": models_out, "providers": providers_out, "_reviewed": reviewed_source}
796 ensure_models_dev_shape(document, "rendered seed")
797 return json.dumps(document, indent=2, ensure_ascii=False) + "\n"
798
799
800 def read_json_file(path: Path) -> Any:
801 if not path.is_file():
802 die(f"missing {path}")
803 return load_json_bytes(path.read_bytes(), str(path))
804
805
806 def cmd_seed_lock(args: argparse.Namespace) -> None:
807 spec = load_seed_spec(Path(args.spec))
808 path_override = os.environ.get("CODEWHALE_MODELS_DEV_PATH", "").strip()
809 if path_override:
810 raw = Path(path_override).read_bytes()
811 source = f"file:{path_override}"
812 else:
813 url = os.environ.get("CODEWHALE_MODELS_DEV_URL", "").strip() or spec["source"]["url"]
814 raw = fetch_url(url)
815 source = f"url:{url}"
816 upstream = ensure_models_dev_shape(load_json_bytes(raw, source), source)
817 fetched_at = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
818 new_lock, errors = build_seed_lock(spec, upstream, raw, fetched_at)
819 lock_path = Path(args.lock)
820 old_lock = read_json_file(lock_path) if lock_path.is_file() else None
821 corrections_path = Path(args.corrections)
822 corrections = read_json_file(corrections_path) if corrections_path.is_file() else None
823 print(f"upstream: {public_source_label(source)} sha256 {new_lock['source']['sha256']}")
824 for line in seed_lock_report(spec, old_lock, new_lock, upstream, corrections):
825 print(line)
826 if errors:
827 die("seed lock refused:\n " + "\n ".join(errors))
828 if args.dry_run:
829 print("dry-run: lock not written")
830 return
831 lock_path.write_text(json.dumps(new_lock, indent=2, ensure_ascii=False) + "\n", encoding="utf-8")
832 print(f"wrote {lock_path}; next: {SEED_RENDER_COMMAND}")
833
834
835 def cmd_seed_render(args: argparse.Namespace) -> None:
836 spec = load_seed_spec(Path(args.spec))
837 lock = read_json_file(Path(args.lock))
838 reviewed = read_json_file(Path(args.corrections)).get("reviewed")
839 rendered = render_seed(spec, lock, reviewed)
840 target = Path(args.out)
841 if args.check:
842 current = target.read_text(encoding="utf-8") if target.is_file() else ""
843 if current == rendered:
844 print(f"ok: {target} matches {SEED_SPEC} + {SEED_LOCK}")
845 return
846 diff = list(
847 difflib.unified_diff(
848 current.splitlines(),
849 rendered.splitlines(),
850 fromfile=f"{target} (committed)",
851 tofile=f"{target} (rendered)",
852 lineterm="",
853 )
854 )
855 for line in diff[:200]:
856 print(line)
857 if len(diff) > 200:
858 print(f"... {len(diff) - 200} more diff lines")
859 die(
860 f"{target} is not the rendered seed. It is generated: put the change in "
861 f"{SEED_SPEC} (selection/mapping) or {CORRECTIONS_ASSET} (policy), then run "
862 f"`{SEED_RENDER_COMMAND}`"
863 )
864 target.write_text(rendered, encoding="utf-8")
865 print(f"wrote {target}")
866
867
868 def build_parser() -> argparse.ArgumentParser:
869 p = argparse.ArgumentParser(
870 description="Secret-free Models.dev / OpenRouter catalog automation (#4117)"
871 )
872 sub = p.add_subparsers(dest="cmd", required=True)
873
874 refresh = sub.add_parser("refresh", help="Fetch live catalog / provider models")
875 refresh.add_argument(
876 "--provider",
877 default=None,
878 help="Optional provider id (currently: openrouter). Omit for Models.dev.",
879 )
880 refresh.add_argument(
881 "--sort",
882 default="newest",
883 choices=["newest", "none"],
884 help="OpenRouter sort order (default: newest)",
885 )
886 refresh.add_argument(
887 "--limit",
888 type=int,
889 default=100,
890 help="OpenRouter row cap (default: 100; 0 = no cap)",
891 )
892 refresh.add_argument(
893 "--write-cache",
894 metavar="PATH",
895 help="Deprecated/unsupported: validate-only automation never writes fetched JSON",
896 )
897 refresh.add_argument(
898 "--write",
899 metavar="PATH",
900 help="Deprecated/unsupported alias of --write-cache",
901 )
902 refresh.set_defaults(func=cmd_refresh)
903
904 snapshot = sub.add_parser(
905 "snapshot",
906 help="Validate or write a Models.dev-shaped snapshot document",
907 )
908 snapshot.add_argument(
909 "path",
910 nargs="?",
911 default="crates/config/assets/models_dev.bundled.json",
912 help="Snapshot path (default: offline seed asset)",
913 )
914 snapshot.add_argument(
915 "--check",
916 action="store_true",
917 help="Validate existing file only (no network)",
918 )
919 snapshot.add_argument(
920 "--write",
921 action="store_true",
922 help="Deprecated/unsupported: validate-only automation never writes snapshots",
923 )
924 snapshot.add_argument(
925 "--force-full",
926 action="store_true",
927 help="Deprecated/unsupported with --write; retained for clear failure messages",
928 )
929 snapshot.set_defaults(func=cmd_snapshot)
930
931 seed = sub.add_parser(
932 "seed",
933 help="Generate the offline seed from the reviewed spec and a pinned lock",
934 )
935 seed_sub = seed.add_subparsers(dest="seed_cmd", required=True)
936 lock = seed_sub.add_parser(
937 "lock",
938 help="Fetch upstream, pin the rows the spec references, print the review report",
939 )
940 lock.add_argument("--dry-run", action="store_true", help="Print the report; write nothing")
941 render = seed_sub.add_parser(
942 "render", help="Render the seed from spec + lock (offline, deterministic)"
943 )
944 render.add_argument(
945 "--check",
946 action="store_true",
947 help="Fail with a diff when the committed seed differs from the rendered one",
948 )
949 render.add_argument("--out", default=str(SEED_ASSET), help="Seed path to write or check")
950 for command in (lock, render):
951 command.add_argument("--spec", default=str(SEED_SPEC))
952 command.add_argument("--lock", default=str(SEED_LOCK))
953 command.add_argument("--corrections", default=str(CORRECTIONS_ASSET))
954 lock.set_defaults(func=cmd_seed_lock)
955 render.set_defaults(func=cmd_seed_render)
956 return p
957
958
959 def main(argv: list[str] | None = None) -> None:
960 parser = build_parser()
961 args = parser.parse_args(argv)
962 args.func(args)
963
964
965 if __name__ == "__main__":
966 main()
967
967 lines PYTHON