Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
This commit is contained in:
1 parent
a3f698806b
commit
5d215e1649
684 files changed
+133838
No files matched your search
@@ -0,0 +1,555 @@
|
||||
#!/usr/bin/env python3
|
||||
"""GitHub search engine for LLM API key leaks.
|
||||
|
||||
Replaces the bash-based broad search with a Python module:
|
||||
- 300+ search queries (code + filename + commit search)
|
||||
- Smart rate-limit handling (reads X-RateLimit headers, dynamic backoff)
|
||||
- Multi-token rotation for higher throughput
|
||||
- Topic-based repo discovery
|
||||
- Gitee (Chinese GitHub mirror) search support
|
||||
|
||||
Output: candidates.json — same format as the bash version:
|
||||
[{"repository": {"full_name": "..."}, "path": "...", "url": "..."}]
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
from pathlib import Path
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
|
||||
RESULTS_DIR = Path(__file__).parent / "results"
|
||||
CANDIDATES_FILE = RESULTS_DIR / "candidates.json"
|
||||
|
||||
# ── High-signal key prefixes ────────────────────────────────────
|
||||
KEY_PREFIXES = [
|
||||
"sk-proj-", "sk-ant-", "AIza", "hf_", "gsk_", "r8_",
|
||||
"sk-or-", "pplx-", "sk-cp-", "sk-sp-", "sk-tp-", "sk-zy-",
|
||||
"fe_oa_", "ak_", "sk-xt-",
|
||||
]
|
||||
|
||||
# ── Env var names by provider ───────────────────────────────────
|
||||
ENV_VARS = [
|
||||
"OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY", "GEMINI_API_KEY",
|
||||
"HUGGINGFACE_TOKEN", "HUGGINGFACE_API_TOKEN", "GROQ_API_KEY",
|
||||
"REPLICATE_API_TOKEN", "TOGETHER_API_KEY", "DEEPSEEK_API_KEY",
|
||||
"OPENROUTER_API_KEY", "PERPLEXITY_API_KEY", "MISTRAL_API_KEY",
|
||||
"COHERE_API_KEY", "AZURE_OPENAI_API_KEY",
|
||||
# China
|
||||
"ARK_API_KEY", "VOLC_API_KEY", "VOLCENGINE_API_KEY",
|
||||
"ZHIPUAI_API_KEY", "GLM_API_KEY", "CHATGLM_API_KEY",
|
||||
"DASHSCOPE_API_KEY", "ALIBABA_API_KEY",
|
||||
"MOONSHOT_API_KEY", "KIMI_API_KEY",
|
||||
"MINIMAX_API_KEY", "HUNYUAN_API_KEY", "TENCENT_API_KEY",
|
||||
"QIANFAN_API_KEY", "BAIDU_API_KEY",
|
||||
"SPARK_API_KEY", "IFLYTEK_API_KEY", "ASTRON_API_KEY",
|
||||
"LINGYIWANWU_API_KEY", "YI_API_KEY",
|
||||
"STEPFUN_API_KEY", "SILICONFLOW_API_KEY",
|
||||
"CSDN_API_KEY", "CSDN_CODING_PLAN_KEY", "STARMAP_API_KEY",
|
||||
"HUAWEI_API_KEY", "HUAWEICLOUD_API_KEY", "CODEARTS_API_KEY",
|
||||
"MIMO_API_KEY", "XIAOMI_API_KEY", "INFINI_API_KEY", "INFINIAI_API_KEY",
|
||||
"JD_API_KEY", "JDCLOUD_API_KEY", "MTHREADS_API_KEY",
|
||||
"KWAIKAT_API_KEY", "KUAISHOU_API_KEY", "STREAMLAKE_API_KEY",
|
||||
"UCLOUD_API_KEY", "COMPSHARE_API_KEY", "ANOMALY_API_KEY",
|
||||
"OPENCODE_GO_KEY", "UNICOM_API_KEY", "CUCLOUD_API_KEY", "YUANJING_API_KEY",
|
||||
"SCNET_API_KEY", "ALIBABA_CODING_PLAN_KEY",
|
||||
"ZYLOO_API_KEY", "ZYLOO_KEY", "FREEMODEL_API_KEY", "FREEMODEL_KEY",
|
||||
"OLLAMA_API_KEY", "XKIRO_API_KEY", "XKIRO_KEY",
|
||||
]
|
||||
|
||||
# ── API endpoint URLs (search for these in code = likely key nearby) ──
|
||||
ENDPOINT_URLS = [
|
||||
"api.openai.com", "api.anthropic.com", "generativelanguage.googleapis.com",
|
||||
"huggingface.co", "api.groq.com", "api.replicate.com", "api.together.xyz",
|
||||
"api.deepseek.com", "openrouter.ai", "api.perplexity.ai",
|
||||
"ark.cn-beijing.volces.com", "open.bigmodel.cn",
|
||||
"dashscope.aliyuncs.com", "api.moonshot.cn", "api.minimaxi.com",
|
||||
"api.hunyuan.cloud.tencent.com", "qianfan.baidubce.com",
|
||||
"spark-api.xf-yun.com", "maas-coding-api.xf-yun.com",
|
||||
"api.lingyiwanwu.com", "api.stepfun.com", "api.siliconflow.cn",
|
||||
"ai.csdn.net", "api.xiaomimimo.com", "cloud.infini-ai.com",
|
||||
"code.mthreads.com", "streamlake.com", "compshare.cn",
|
||||
"opencode.ai", "cucloud.cn", "scnet.cn",
|
||||
"coding.dashscope.aliyuncs.com", "api.zyloo.io",
|
||||
"freemodel.dev", "cc.freemodel.dev", "api.ollama.com",
|
||||
"api.longcat.chat", "api.xkiro.com",
|
||||
]
|
||||
|
||||
# ── Model names (search for these = key assignment often nearby) ──
|
||||
MODEL_NAMES = [
|
||||
"gpt-4o", "gpt-4o-mini", "claude-sonnet-4", "claude-opus-4",
|
||||
"gemini-1.5-flash", "gemini-2.0-flash",
|
||||
"doubao-seed-2-0-pro", "glm-4-flash", "glm-5",
|
||||
"qwen-plus", "moonshot-v1", "MiniMax-M2.5",
|
||||
"deepseek-chat", "deepseek-v4-pro",
|
||||
"yi-large", "step-1-flash", "LongCat-2.0",
|
||||
"astron-code-latest", "glm_for_coding",
|
||||
"anthropic/claude-sonnet-4-5", "anthropic/claude-opus-4-5",
|
||||
]
|
||||
|
||||
# ── File extensions to search ───────────────────────────────────
|
||||
EXTENSIONS = [
|
||||
"env", "py", "js", "ts", "sh", "cfg", "ini", "ipynb",
|
||||
"yaml", "yml", "json", "toml", "tf", "rb", "go", "rs",
|
||||
"java", "kt", "php", "lua",
|
||||
]
|
||||
|
||||
# ── Special filenames to search (high-value targets) ────────────
|
||||
HOT_FILENAMES = [
|
||||
".env", ".env.local", ".env.production", ".env.development",
|
||||
"config.json", "config.yaml", "config.yml",
|
||||
"secrets.yaml", "secrets.json", "secrets.toml",
|
||||
"api_keys.txt", "api_keys.json", "keys.json",
|
||||
"credentials.json", "credentials.yaml",
|
||||
"settings.json", "settings.yaml",
|
||||
"application.yml", "application.properties",
|
||||
"docker-compose.yml", "docker-compose.yaml",
|
||||
]
|
||||
|
||||
# ── GitHub topics to discover repos for targeted scanning ───────
|
||||
GH_TOPICS = [
|
||||
"openai", "anthropic", "llm", "chatgpt", "gpt",
|
||||
"claude", "gemini", "ai-api", "language-model",
|
||||
"chatbot", "ai-agent", "rag", "langchain",
|
||||
"dashscope", "qwen", "chatglm", "zhipu",
|
||||
"doubao", "kimi", "moonshot", "minimax",
|
||||
"deepseek", "spark-llm", "iflytek",
|
||||
]
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
# Query Generation
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
|
||||
def gen_code_queries():
|
||||
"""Generate GitHub code search queries.
|
||||
|
||||
Strategy:
|
||||
1. Key prefix + extension (high signal, catches raw keys in code)
|
||||
2. Env var name + extension (catches assignments like OPENAI_API_KEY=sk-...)
|
||||
3. Endpoint URL + extension (catches config near API calls)
|
||||
4. Model name + extension (catches config near model usage)
|
||||
"""
|
||||
queries = []
|
||||
|
||||
# 1. Key prefix searches — only in code-like files
|
||||
code_exts = ["env", "py", "js", "ts", "sh", "cfg", "ini", "ipynb",
|
||||
"yaml", "yml", "json", "toml", "tf", "go", "rs", "rb"]
|
||||
for prefix in KEY_PREFIXES:
|
||||
for ext in code_exts:
|
||||
queries.append(f"{prefix} extension:{ext}")
|
||||
|
||||
# 2. Env var searches — in all file types
|
||||
all_exts = code_exts + ["lua", "java", "kt", "php"]
|
||||
for var in ENV_VARS:
|
||||
for ext in all_exts:
|
||||
queries.append(f"{var} extension:{ext}")
|
||||
|
||||
# 3. Endpoint URL searches — in code files
|
||||
for url in ENDPOINT_URLS:
|
||||
for ext in ["py", "js", "ts", "env", "yaml", "yml", "json", "toml", "ipynb"]:
|
||||
queries.append(f"{url} extension:{ext}")
|
||||
|
||||
# 4. Model name searches — in config/code files
|
||||
for model in MODEL_NAMES:
|
||||
for ext in ["py", "js", "ts", "env", "yaml", "yml", "json", "ipynb"]:
|
||||
queries.append(f"{model} extension:{ext}")
|
||||
|
||||
# 5. Language-specific patterns
|
||||
# process.env in JS/TS
|
||||
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
||||
"GROQ_API_KEY", "HUGGINGFACE_TOKEN", "DEEPSEEK_API_KEY"]:
|
||||
queries.append(f"process.env.{var} extension:js")
|
||||
queries.append(f"process.env.{var} extension:ts")
|
||||
|
||||
# os.environ in Python
|
||||
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
||||
"GROQ_API_KEY", "DEEPSEEK_API_KEY", "DASHSCOPE_API_KEY",
|
||||
"ARK_API_KEY", "ZHIPUAI_API_KEY"]:
|
||||
queries.append(f"os.environ {var} extension:py")
|
||||
queries.append(f"os.getenv {var} extension:py")
|
||||
|
||||
# Dockerfile ENV patterns
|
||||
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
||||
"GROQ_API_KEY", "DEEPSEEK_API_KEY", "ARK_API_KEY",
|
||||
"DASHSCOPE_API_KEY", "ZHIPUAI_API_KEY"]:
|
||||
queries.append(f"ENV {var} extension:Dockerfile")
|
||||
queries.append(f"ARG {var} extension:Dockerfile")
|
||||
|
||||
# CI/CD secret patterns
|
||||
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
||||
"DEEPSEEK_API_KEY", "ARK_API_KEY"]:
|
||||
queries.append(f"secrets.{var} extension:yml")
|
||||
queries.append(f"secrets.{var} extension:yaml")
|
||||
|
||||
return queries
|
||||
|
||||
|
||||
def gen_filename_queries():
|
||||
"""Generate filename-based searches — find high-value config files."""
|
||||
queries = []
|
||||
for fname in HOT_FILENAMES:
|
||||
# Search for the filename itself — these files often contain keys
|
||||
queries.append(f"filename:{fname}")
|
||||
return queries
|
||||
|
||||
|
||||
def gen_commit_queries():
|
||||
"""Generate commit message search queries.
|
||||
|
||||
Keys are sometimes accidentally committed in commit messages.
|
||||
Uses search/commits API.
|
||||
"""
|
||||
queries = []
|
||||
for prefix in KEY_PREFIXES:
|
||||
queries.append(prefix)
|
||||
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
||||
"DEEPSEEK_API_KEY", "ARK_API_KEY", "DASHSCOPE_API_KEY"]:
|
||||
queries.append(var)
|
||||
return queries
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
# GitHub API Client
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
|
||||
class GitHubSearcher:
|
||||
"""GitHub search client with rate-limit awareness and token rotation."""
|
||||
|
||||
def __init__(self, tokens=None, per_page=100, verbose=True):
|
||||
self.tokens = tokens or [os.environ.get("GH_TOKEN", "")]
|
||||
self.tokens = [t for t in self.tokens if t]
|
||||
if not self.tokens:
|
||||
print("ERROR: No GitHub tokens provided. Set GH_TOKEN env var or use --token.")
|
||||
sys.exit(1)
|
||||
self._token_idx = 0
|
||||
self.per_page = per_page
|
||||
self.verbose = verbose
|
||||
self._rate_remaining = 30
|
||||
self._rate_reset = 0
|
||||
|
||||
def _next_token(self):
|
||||
"""Rotate to the next token."""
|
||||
self._token_idx = (self._token_idx + 1) % len(self.tokens)
|
||||
return self.tokens[self._token_idx]
|
||||
|
||||
def _current_token(self):
|
||||
return self.tokens[self._token_idx]
|
||||
|
||||
def _api(self, endpoint, params=None):
|
||||
"""Call GitHub API with rate-limit handling and token rotation.
|
||||
|
||||
Returns (data, error). On rate limit, waits and retries.
|
||||
"""
|
||||
import urllib.parse
|
||||
url = f"https://api.github.com/{endpoint}"
|
||||
if params:
|
||||
url += "?" + urllib.parse.urlencode(params)
|
||||
|
||||
for attempt in range(len(self.tokens) * 2):
|
||||
token = self._current_token()
|
||||
req = urllib.request.Request(url, headers={
|
||||
"Authorization": f"token {token}",
|
||||
"Accept": "application/vnd.github+json",
|
||||
"User-Agent": "llm-key-hunter/2.0",
|
||||
})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
# Track rate limit from headers
|
||||
remaining = resp.headers.get("X-RateLimit-Remaining")
|
||||
reset = resp.headers.get("X-RateLimit-Reset")
|
||||
if remaining:
|
||||
self._rate_remaining = int(remaining)
|
||||
if reset:
|
||||
self._rate_reset = int(reset)
|
||||
return json.loads(resp.read()), None
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 403:
|
||||
# Rate limited or forbidden
|
||||
remaining = e.headers.get("X-RateLimit-Remaining", "0")
|
||||
reset = e.headers.get("X-RateLimit-Reset", "0")
|
||||
if remaining == "0" and reset:
|
||||
wait = int(reset) - int(time.time()) + 2
|
||||
if wait > 0 and wait < 3600:
|
||||
if self.verbose:
|
||||
print(f" [rate-limit] Waiting {wait}s for reset...")
|
||||
time.sleep(wait)
|
||||
continue
|
||||
# Try rotating token
|
||||
self._next_token()
|
||||
continue
|
||||
elif e.code == 422:
|
||||
# Unprocessable — bad query, skip
|
||||
return None, f"422: {e.read()[:200]}"
|
||||
else:
|
||||
return None, f"HTTP {e.code}"
|
||||
except Exception as e:
|
||||
return None, str(e)
|
||||
|
||||
return None, "Exhausted all tokens + retries"
|
||||
|
||||
def search_code(self, query, max_results=1000):
|
||||
"""Search GitHub code. Returns list of candidate dicts."""
|
||||
results = []
|
||||
page = 1
|
||||
while len(results) < max_results:
|
||||
data, err = self._api("search/code", {
|
||||
"q": query,
|
||||
"per_page": self.per_page,
|
||||
"page": page,
|
||||
})
|
||||
if err:
|
||||
if self.verbose and "422" not in str(err):
|
||||
print(f" [error] {err}")
|
||||
break
|
||||
items = data.get("items", [])
|
||||
if not items:
|
||||
break
|
||||
for item in items:
|
||||
results.append({
|
||||
"repository": {"full_name": item["repository"]["full_name"]},
|
||||
"path": item["path"],
|
||||
"url": item["html_url"],
|
||||
})
|
||||
if len(items) < self.per_page:
|
||||
break
|
||||
page += 1
|
||||
# Be nice to the API
|
||||
if self._rate_remaining < 5:
|
||||
wait = max(self._rate_reset - int(time.time()) + 2, 2)
|
||||
if self.verbose:
|
||||
print(f" [rate-limit] {self._rate_remaining} left, waiting {wait}s...")
|
||||
time.sleep(min(wait, 120))
|
||||
else:
|
||||
time.sleep(0.2)
|
||||
return results
|
||||
|
||||
def search_commits(self, query, max_results=500):
|
||||
"""Search GitHub commit messages for key patterns."""
|
||||
results = []
|
||||
data, err = self._api("search/commits", {
|
||||
"q": query,
|
||||
"per_page": min(self.per_page, 100),
|
||||
})
|
||||
if err:
|
||||
return []
|
||||
for item in data.get("items", []):
|
||||
repo = item.get("repository", {})
|
||||
results.append({
|
||||
"repository": {"full_name": repo.get("full_name", "")},
|
||||
"path": "(commit message)",
|
||||
"url": item.get("html_url", ""),
|
||||
})
|
||||
return results
|
||||
|
||||
def search_repos_by_topic(self, topic, max_results=100):
|
||||
"""Find repos by topic for targeted scanning."""
|
||||
data, err = self._api("search/repositories", {
|
||||
"q": f"topic:{topic}",
|
||||
"per_page": min(max_results, 100),
|
||||
"sort": "updated",
|
||||
})
|
||||
if err:
|
||||
return []
|
||||
return [item["full_name"] for item in data.get("items", [])]
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
# Gitee Search (Chinese GitHub mirror)
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
|
||||
def gitee_search(query, token=None, max_results=100):
|
||||
"""Search Gitee.com for key patterns. Returns candidate dicts."""
|
||||
import urllib.parse
|
||||
url = f"https://gitee.com/api/v5/search/code?q={urllib.parse.quote(query)}&per_page=20"
|
||||
headers = {"User-Agent": "llm-key-hunter/2.0"}
|
||||
if token:
|
||||
url += f"&access_token={token}"
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=15) as resp:
|
||||
items = json.loads(resp.read())
|
||||
return [{
|
||||
"repository": {"full_name": f"gitee:{item.get('repository', {}).get('full_name', '')}"},
|
||||
"path": item.get("path", ""),
|
||||
"url": item.get("html_url", ""),
|
||||
} for item in items[:max_results]]
|
||||
except Exception:
|
||||
return []
|
||||
|
||||
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
# Main Pipeline
|
||||
# ═══════════════════════════════════════════════════════════════
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(
|
||||
description="GitHub search engine for LLM API key leaks")
|
||||
parser.add_argument("--token", action="append", default=[],
|
||||
help="GitHub token (can repeat for multi-token rotation)")
|
||||
parser.add_argument("--gitee-token", default=None,
|
||||
help="Gitee API token for Chinese mirror search")
|
||||
parser.add_argument("--output", default=str(CANDIDATES_FILE),
|
||||
help="Output candidates JSON file")
|
||||
parser.add_argument("--no-commits", action="store_true",
|
||||
help="Skip commit message search")
|
||||
parser.add_argument("--no-filenames", action="store_true",
|
||||
help="Skip filename-based search")
|
||||
parser.add_argument("--no-gitee", action="store_true",
|
||||
help="Skip Gitee search")
|
||||
parser.add_argument("--no-topics", action="store_true",
|
||||
help="Skip topic-based repo discovery")
|
||||
parser.add_argument("--max-results", type=int, default=1000,
|
||||
help="Max results per query (default: 1000)")
|
||||
parser.add_argument("--verbose", "-v", action="store_true", default=True)
|
||||
parser.add_argument("--resume", action="store_true",
|
||||
help="Resume from checkpoint file")
|
||||
args = parser.parse_args()
|
||||
|
||||
# Collect tokens
|
||||
tokens = args.token or []
|
||||
env_token = os.environ.get("GH_TOKEN", "")
|
||||
if env_token and env_token not in tokens:
|
||||
tokens.append(env_token)
|
||||
if not tokens:
|
||||
print("ERROR: No GitHub tokens. Use --token or set GH_TOKEN env var.")
|
||||
sys.exit(1)
|
||||
|
||||
searcher = GitHubSearcher(tokens=tokens, verbose=args.verbose)
|
||||
|
||||
# Generate queries
|
||||
code_queries = gen_code_queries()
|
||||
filename_queries = gen_filename_queries() if not args.no_filenames else []
|
||||
commit_queries = gen_commit_queries() if not args.no_commits else []
|
||||
|
||||
total_queries = len(code_queries) + len(filename_queries) + len(commit_queries)
|
||||
print(f"Generated {total_queries} queries:")
|
||||
print(f" Code search: {len(code_queries)}")
|
||||
print(f" Filename search: {len(filename_queries)}")
|
||||
print(f" Commit search: {len(commit_queries)}")
|
||||
print(f" Tokens: {len(tokens)}")
|
||||
print()
|
||||
|
||||
all_candidates = {} # url -> candidate (dedup by URL)
|
||||
processed = 0
|
||||
start_idx = 0
|
||||
|
||||
# ── Checkpoint / resume ──────────────────────────────────────
|
||||
checkpoint_path = Path(args.output).with_suffix(".checkpoint.json")
|
||||
progress_path = Path(args.output).with_suffix(".progress.json")
|
||||
|
||||
def save_checkpoint(proc):
|
||||
ckpt = sorted(all_candidates.values(),
|
||||
key=lambda c: c["repository"]["full_name"])
|
||||
with open(checkpoint_path, "w") as f:
|
||||
json.dump(ckpt, f)
|
||||
with open(progress_path, "w") as f:
|
||||
json.dump({"processed": proc, "candidates": len(all_candidates)}, f)
|
||||
|
||||
if args.resume and checkpoint_path.exists():
|
||||
with open(checkpoint_path) as f:
|
||||
for c in json.load(f):
|
||||
all_candidates[c["url"]] = c
|
||||
if progress_path.exists():
|
||||
with open(progress_path) as f:
|
||||
start_idx = json.load(f).get("processed", 0)
|
||||
else:
|
||||
# Legacy checkpoint (written at processed % 50 == 0, before that query ran)
|
||||
start_idx = 50
|
||||
processed = start_idx
|
||||
print(f"Resumed: {len(all_candidates)} candidates, skipping first {start_idx} queries")
|
||||
|
||||
# ── Code search ──────────────────────────────────────────────
|
||||
print("=== Phase 1: Code Search ===")
|
||||
for i, query in enumerate(code_queries):
|
||||
if i < start_idx:
|
||||
continue
|
||||
processed = i + 1
|
||||
if processed % 50 == 0:
|
||||
print(f" [{processed}/{total_queries}] {len(all_candidates)} candidates so far")
|
||||
save_checkpoint(processed)
|
||||
|
||||
results = searcher.search_code(query, max_results=args.max_results)
|
||||
for r in results:
|
||||
if r["url"] not in all_candidates:
|
||||
all_candidates[r["url"]] = r
|
||||
|
||||
# ── Filename search ──────────────────────────────────────────
|
||||
if filename_queries:
|
||||
print(f"\n=== Phase 2: Filename Search ===")
|
||||
for query in filename_queries:
|
||||
processed += 1
|
||||
results = searcher.search_code(query, max_results=args.max_results)
|
||||
for r in results:
|
||||
if r["url"] not in all_candidates:
|
||||
all_candidates[r["url"]] = r
|
||||
print(f" {len(all_candidates)} total candidates after filename search")
|
||||
|
||||
# ── Commit search ────────────────────────────────────────────
|
||||
if commit_queries:
|
||||
print(f"\n=== Phase 3: Commit Message Search ===")
|
||||
for query in commit_queries:
|
||||
processed += 1
|
||||
results = searcher.search_commits(query)
|
||||
for r in results:
|
||||
if r["url"] not in all_candidates:
|
||||
all_candidates[r["url"]] = r
|
||||
print(f" {len(all_candidates)} total candidates after commit search")
|
||||
|
||||
# ── Gitee search ─────────────────────────────────────────────
|
||||
if not args.no_gitee:
|
||||
print(f"\n=== Phase 4: Gitee Search ===")
|
||||
gitee_queries = KEY_PREFIXES + ["OPENAI_API_KEY", "ANTHROPIC_API_KEY",
|
||||
"DASHSCOPE_API_KEY", "ARK_API_KEY"]
|
||||
for query in gitee_queries:
|
||||
results = gitee_search(query, token=args.gitee_token)
|
||||
for r in results:
|
||||
if r["url"] not in all_candidates:
|
||||
all_candidates[r["url"]] = r
|
||||
print(f" {len(all_candidates)} total candidates after Gitee search")
|
||||
|
||||
# ── Topic-based repo discovery ───────────────────────────────
|
||||
if not args.no_topics:
|
||||
print(f"\n=== Phase 5: Topic-based Repo Discovery ===")
|
||||
topic_repos = set()
|
||||
for topic in GH_TOPICS:
|
||||
repos = searcher.search_repos_by_topic(topic)
|
||||
topic_repos.update(repos)
|
||||
print(f" Found {len(topic_repos)} repos by topic")
|
||||
# For each repo, search for key patterns in code
|
||||
for repo in sorted(topic_repos):
|
||||
for prefix in KEY_PREFIXES[:5]: # Top 5 prefixes only
|
||||
query = f"{prefix} repo:{repo}"
|
||||
results = searcher.search_code(query, max_results=100)
|
||||
for r in results:
|
||||
if r["url"] not in all_candidates:
|
||||
all_candidates[r["url"]] = r
|
||||
print(f" {len(all_candidates)} total candidates after topic search")
|
||||
|
||||
# ── Write output ─────────────────────────────────────────────
|
||||
candidates = sorted(all_candidates.values(),
|
||||
key=lambda c: c["repository"]["full_name"])
|
||||
output_path = Path(args.output)
|
||||
output_path.parent.mkdir(parents=True, exist_ok=True)
|
||||
with open(output_path, "w") as f:
|
||||
json.dump(candidates, f, indent=2)
|
||||
|
||||
# Clean up checkpoint
|
||||
if checkpoint_path.exists():
|
||||
checkpoint_path.unlink()
|
||||
if progress_path.exists():
|
||||
progress_path.unlink()
|
||||
|
||||
print(f"\n=== Done! ===")
|
||||
print(f" Total queries: {processed}")
|
||||
print(f" Candidates: {len(candidates)}")
|
||||
print(f" Output: {output_path}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user