Files
hack/tools/scripts/llm-key-hunter/hot_platforms.py
T
OpenCode 68a9071cf1 feat: reconcile local master with origin (exa/hot-platform hunters, leaks ledger, latest results)
Local branch had diverged from origin/master (sibling commits on the same base). Rewrote local history linearly on top of origin/master, folding in all local content: exa/hot-platform discovery hunters, leaks ledger, nightly sweep orchestrator, updated .gitignore and skill docs, plus the latest hunt output and vault state. Remote-only files (vault keys, channel scripts) were restored rather than dropped, so the resulting tree is a full union of both sides.
2026-08-07 12:53:47 +08:00

431 lines
16 KiB
Python

#!/usr/bin/env python3
"""Hot-platform discovery — find NEW AI platforms worth key-hunting.
Sources:
- GitHub Trending (weekly, all languages)
- HuggingFace trending API
Pipeline: scrape repo names -> fetch each repo's README (direct, GitHub is
reachable from this host) -> rule-based signal scoring:
*3 README documents a `*_API_KEY` / `*_KEY` / `*_TOKEN` env var
*3 README mentions "api key" in a developer context
*2 README contains an `api.<x>.ai|com|io` hostname
*2 README claims OpenAI-compatible / drop-in OpenAI replacement
*1 README mentions sign-up/console/developer platform/credits
*1 repo description mentions api/platform/llm
Entries scoring >= 3 become candidates, saved to hot_platforms.json as
disabled (human confirms -> enabled). Env-var names found in READMEs are
stored so hunt_hot.py can search them directly.
Usage:
python3 hot_platforms.py # discover + update json
python3 hot_platforms.py --list # print current platform table
python3 hot_platforms.py --score 4 # raise the bar (default 3)
"""
import argparse
import json
import os
import re
import sys
import time
import urllib.request
import urllib.error
from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timezone
from pathlib import Path
HERE = Path(__file__).resolve().parent
CONF = HERE / "hot_platforms.json"
UA = "Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 Chrome/126 Safari/537.36"
ENV_VAR_RE = re.compile(r"\b([A-Z][A-Z0-9_]{2,}_(?:API[_-]?KEY|APIKEY|KEY|TOKEN))\b")
API_KEY_MENTION = re.compile(r"\bapi[-_ ]?key\b", re.I)
API_HOST_RE = re.compile(r"\bapi\.([a-z0-9][a-z0-9\-]{1,30})\.(?:ai|com|io|dev|app)\b", re.I)
OPENAI_COMPAT = re.compile(r"openai[- ]compatible|drop[- ]in (?:replacement|drop) for openai|openai compat", re.I)
SIGNUP_RE = re.compile(r"\b(sign up|create (?:an? )?account|developer (?:portal|platform|console)|get (?:an? )?api key|console\.)", re.I)
CREDITS_RE = re.compile(r"\b(usage[- ]?based|credits|pay[- ]?as[- ]?you[- ]?go|free tier|pricing)\b", re.I)
EXCLUDE_OWNERS = {
"apps", "features", "collections", "topics", "marketplace", "about",
"readme", "contact", "orgs", "explore", "events", "login", "pricing",
"settings", "sponsors", "site", "support", "enterprise", "customers",
"security", "search", "who", "graphql", "community", "customer-stories",
"solutions", "new", "trending",
}
# Providers that already have dedicated hunts / vault dirs — never re-add
KNOWN_PROVIDERS = {
"openai", "anthropic", "google", "gemini", "mistral", "groq", "together",
"together-ai", "perplexity", "cohere", "azure", "aws", "huggingface",
"deepseek", "moonshot", "kimi", "minimax", "minimaxi", "zhipu", "bigmodel",
"baichuan", "baichuan-ai", "lingyiwanwu", "stepfun", "siliconflow",
"volcanoark", "volcengine", "dashscope", "qwen", "xiaomi", "mimo",
"qianfan", "baidu", "hunyuan", "tencent", "spark", "iflytek", "xunfei",
"longcat", "kiro", "freemodel", "ollama", "xkiro", "zyloo", "scnet",
"exa", "openrouter", "replicate", "fireworks", "codestral", "novita",
"arcee", "grok", "xai", "claude", "cursor", "windsurf", "openinterpreter",
"codeium", "copilot", "cline", "continue", "aws-sso",
}
def load_conf():
if CONF.exists():
try:
return json.loads(CONF.read_text())
except Exception:
pass
return {"updated": None, "platforms": []}
def save_conf(cfg):
CONF.write_text(json.dumps(cfg, indent=2, ensure_ascii=False) + "\n")
def scrape_gh_trending():
"""Return [owner/name] from GitHub trending weekly/daily/monthly."""
repos = []
for since in ("weekly", "daily", "monthly"):
req = urllib.request.Request(
f"https://github.com/trending?since={since}",
headers={"User-Agent": UA})
try:
with urllib.request.urlopen(req, timeout=30) as r:
html = r.read().decode("utf-8", "replace")
except Exception as e:
print(f" gh trending({since}) failed: {e}", file=sys.stderr)
continue
for m in re.finditer(r'href="/([A-Za-z0-9_.\-]+/[A-Za-z0-9_.\-]+)"', html):
r = m.group(1)
owner, _, repo = r.partition("/")
if owner.lower() in EXCLUDE_OWNERS:
continue
if owner.startswith(".") or repo.startswith("."):
continue
repos.append(r)
return repos
def scrape_hackernews():
"""Fresh HN stories mentioning AI platforms/API keys (7 days)."""
cutoff = int(time.time() - 7 * 86400)
url = (f"https://hn.algolia.com/api/v1/search_by_date?"
f"query=AI%20platform&tags=story&numericFilters=created_at_i%3E{cutoff}")
req = urllib.request.Request(url, headers={"User-Agent": UA})
try:
with urllib.request.urlopen(req, timeout=30) as r:
data = json.loads(r.read().decode("utf-8", "replace"))
except Exception as e:
print(f" hn failed: {e}", file=sys.stderr)
return []
repos, seen = [], set()
for hit in data.get("hits", []):
for url in (hit.get("url") or "").split():
m = re.match(r"https?://github\.com/([A-Za-z0-9_.\-]+/[A-Za-z0-9_.\-]+)/?",
url)
if m and m.group(1).lower() not in seen:
seen.add(m.group(1).lower())
repos.append(m.group(1))
return repos
def scan_leak_domains():
"""Mine hunt results for API hostnames (api.<x>.com/.ai) we don't know yet.
Returns {hostname: evidence_url}."""
hosts = {}
for txt in (HERE / "results").glob("*/*.txt"):
try:
content = txt.read_text(errors="replace")
except Exception:
continue
for u in re.findall(r"https?://[^\s|]+", content):
m = re.search(r"\bapi\.([a-z0-9][a-z0-9\-]{1,30})\.(?:ai|com|io|dev|app)\b",
u, re.I)
if m:
hosts.setdefault(m.group(1).lower(), u)
ledger = (HERE / "results" / "leaks_ledger.json")
if ledger.exists():
try:
data = json.loads(ledger.read_text())
for u in data.get("files", {}):
m = re.search(r"\bapi\.([a-z0-9][a-z0-9\-]{1,30})\.(?:ai|com|io|dev|app)\b",
u, re.I)
if m:
hosts.setdefault(m.group(1).lower(), u)
except Exception:
pass
return hosts
def scrape_hf_trending():
"""Return [owner/name] from HF trending API."""
req = urllib.request.Request("https://huggingface.co/api/trending",
headers={"User-Agent": UA})
try:
with urllib.request.urlopen(req, timeout=30) as r:
data = json.loads(r.read().decode("utf-8", "replace"))
except Exception as e:
print(f" hf trending failed: {e}", file=sys.stderr)
return []
out = []
for bucket in ("recentlyTrending", "dailyTrending"):
for item in data.get(bucket, []):
rd = item.get("repoData", {})
author, name = rd.get("author"), rd.get("name")
if author and name:
out.append(f"{author}/{name}")
return out
def fetch_readme(full_name, timeout=20):
"""Try common README filenames, return (text, url) or (None, None)."""
for fname in ("README.md", "README.rst", "README.txt", "README",
"readme.md", "README.MD"):
url = (f"https://raw.githubusercontent.com/{full_name}/HEAD/{fname}")
req = urllib.request.Request(url, headers={"User-Agent": UA})
try:
with urllib.request.urlopen(req, timeout=timeout) as r:
if r.getcode() == 200:
return r.read().decode("utf-8", "replace"), url
except urllib.error.HTTPError:
continue
except Exception:
return None, None
return None, None
def probe_disabled_platforms(cfg):
"""Liveness-probe disabled platforms that have an endpoint.
GET https://<endpoint> with no key — a 401/402/403/404 means the
endpoint is alive (worth hunting); net error means dead/host gone.
Writes the probe result into entry['probe']; returns count probed.
"""
probed = 0
for p in cfg["platforms"]:
if p.get("enabled") or not p.get("endpoint") or p.get("probe"):
continue
url = f"https://{p['endpoint']}"
req = urllib.request.Request(url, headers={"User-Agent": UA})
try:
with urllib.request.urlopen(req, timeout=10) as r:
p["probe"] = f"HTTP {r.getcode()}"
except urllib.error.HTTPError as e:
p["probe"] = f"HTTP {e.code}"
except Exception as e:
p["probe"] = f"net:{type(e).__name__}"
probed += 1
print(f" probe {p['name']:20s} -> {p['probe']}", flush=True)
return probed
def score_readme(text, full_name):
"""Return (score, signals, env_vars, hostnames)."""
score, signals, env_vars, hosts = 0, [], set(), set()
head = text[:6000]
low = head.lower()
for m in ENV_VAR_RE.finditer(head):
env = m.group(1)
base = env.rsplit("_", 2)[0].lower().replace("_", "-")
if base in KNOWN_PROVIDERS:
continue # example of a known provider's key — not a new platform
env_vars.add(env)
score += 3
signals.append(f"env:{env}")
for m in API_HOST_RE.finditer(head):
host = m.group(1).lower()
if host in ("example", "test", "demo", "localhost", "staging",
"your", "yourdomain", "sample", "placeholder", "api",
"docs", "app", "dev"):
continue
hosts.add(host)
score += 2
sig = f"host:{host}"
if sig not in signals:
signals.append(sig)
if OPENAI_COMPAT.search(low):
score += 2
signals.append("openai-compatible")
if API_KEY_MENTION.search(head) and (env_vars or hosts):
score += 1
signals.append("api-key-mention")
if SIGNUP_RE.search(low):
score += 1
signals.append("signup/console")
if CREDITS_RE.search(low):
score += 1
signals.append("usage-pricing")
return score, signals[:6], sorted(env_vars)[:6], sorted(hosts)[:6]
def derive_names(full_name, text, env_vars, hosts):
"""Best-effort platform name candidates."""
names = []
for h in hosts:
names.append(h)
for env in env_vars:
base = env.rsplit("_", 2)[0].lower().replace("_", "-")
if len(base) >= 3:
names.append(base)
# brand in README title often appears as <brand> AI / <brand> API
m = re.search(r"#\s+([A-Za-z0-9][A-Za-z0-9 .\-]{1,30})", text)
if m:
brand = re.sub(r"\s+(ai|api|sdk|platform)$", "", m.group(1).strip(), flags=re.I)
names.append(brand.lower().replace(" ", "-"))
names.append(full_name.split("/")[-1].lower())
# de-dup, keep order, drop junk
seen, out = set(), []
for n in names:
n = re.sub(r"[^a-z0-9.\-]", "", n.lower())
if not n or len(n) < 3 or n in seen:
continue
if n in ("readme", "github", "master", "main"):
continue
if n in ("example", "demo", "test", "sample", "docs", "quickstart",
"template", "api", "sdk", "app", "client", "python",
"typescript", "javascript", "node", "get-started", "tutorial",
"hello-world", "boilerplate", "starter", "guide"):
continue
if n in KNOWN_PROVIDERS:
continue
seen.add(n)
out.append(n)
return out
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--list", action="store_true")
ap.add_argument("--no-save", action="store_true")
ap.add_argument("--score", type=int, default=4,
help="minimum signal score to accept (default 4; 3 = soft)")
ap.add_argument("--max-readmes", type=int, default=60)
args = ap.parse_args()
cfg = load_conf()
if args.list:
for p in cfg["platforms"]:
flag = "ON " if p.get("enabled") else "off"
print(f" [{flag}] {p['name']:20s} env={','.join(p.get('env_vars', []))[:60]} "
f"src={p.get('source')} added={p.get('added')} "
f"score={p.get('score', 'n/a')} probe={p.get('probe', '-')}")
print(f"\n {len(cfg['platforms'])} platforms "
f"({sum(1 for p in cfg['platforms'] if p.get('enabled'))} enabled)")
return
print("scraping trending sources...", flush=True)
repos = (scrape_gh_trending() + scrape_hf_trending()
+ scrape_hackernews())
seen, uniq = set(), []
for r in repos:
if r not in seen:
seen.add(r)
uniq.append(r)
print(f" {len(uniq)} unique repos", flush=True)
known_names = set()
for p in cfg["platforms"]:
known_names.add(p["name"].lower())
known_names.update(a.lower() for a in p.get("aliases", []))
# leak-domain mining: api.<x>.* hostnames seen in hunt output become
# direct candidates (no README needed — someone already leaked a key
# against that endpoint)
now = datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%MZ")
leak_hosts = scan_leak_domains()
leak_added = 0
for host, ev in sorted(leak_hosts.items()):
if host in known_names or host in KNOWN_PROVIDERS:
continue
entry = {
"name": host,
"aliases": [host, f"api.{host}.com", f"api.{host}.ai"],
"env_vars": [host.upper().replace("-", "_") + "_API_KEY"],
"key_format": "unknown",
"endpoint": f"api.{host}.com",
"verify": None,
"source": f"leak-domain:{ev[:80]}",
"added": now,
"enabled": False,
"score": 3,
"signals": ["leak-domain"],
"readme": ev,
}
cfg["platforms"].append(entry)
known_names.add(host)
leak_added += 1
print(f" + {host:20s} (from leak domain)", flush=True)
# skip repos we already have (by last path segment)
fresh = [r for r in uniq if r.split("/")[-1].lower() not in known_names][:args.max_readmes]
print(f"fetching READMEs for {len(fresh)} fresh repos...", flush=True)
readmes = {}
with ThreadPoolExecutor(max_workers=12) as pool:
futs = {pool.submit(fetch_readme, r): r for r in fresh}
for fut in as_completed(futs):
r = futs[fut]
try:
text, url = fut.result()
except Exception:
text, url = None, None
if text:
readmes[r] = (text, url)
print(f" {len(readmes)} READMEs fetched", flush=True)
added = []
for full_name, (text, url) in sorted(readmes.items()):
score, signals, env_vars, hosts = score_readme(text, full_name)
if score < args.score:
continue
names = derive_names(full_name, text, env_vars, hosts)
if not names:
continue
name = names[0]
if name in known_names:
continue
entry = {
"name": name,
"aliases": names[:4],
"env_vars": env_vars or [n.upper().replace("-", "_") + "_API_KEY"
for n in names[:2]],
"key_format": "unknown",
"endpoint": f"api.{name}.ai" if hosts else None,
"verify": None,
"source": f"trending:{full_name}",
"added": now,
"enabled": False,
"score": score,
"signals": signals,
"readme": url,
}
cfg["platforms"].append(entry)
known_names.add(name)
added.append(entry)
print(f" + {name:20s} score={score} {','.join(signals)}", flush=True)
if added or leak_added:
cfg["updated"] = now
if not args.no_save:
save_conf(cfg)
print(f"\nsaved {len(added) + leak_added} new platforms to {CONF}")
else:
print("\nno new platforms.")
# endpoint liveness for disabled platforms (one-shot; results cached in json)
if not args.no_save:
probed = probe_disabled_platforms(cfg)
if probed:
save_conf(cfg)
print(f"\nprobed {probed} disabled platforms")
print(f"\n{len(cfg['platforms'])} total in registry "
f"({sum(1 for p in cfg['platforms'] if p.get('enabled'))} enabled)")
if __name__ == "__main__":
main()