- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
556 lines
24 KiB
Python
556 lines
24 KiB
Python
#!/usr/bin/env python3
|
|
"""GitHub search engine for LLM API key leaks.
|
|
|
|
Replaces the bash-based broad search with a Python module:
|
|
- 300+ search queries (code + filename + commit search)
|
|
- Smart rate-limit handling (reads X-RateLimit headers, dynamic backoff)
|
|
- Multi-token rotation for higher throughput
|
|
- Topic-based repo discovery
|
|
- Gitee (Chinese GitHub mirror) search support
|
|
|
|
Output: candidates.json — same format as the bash version:
|
|
[{"repository": {"full_name": "..."}, "path": "...", "url": "..."}]
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import subprocess
|
|
import sys
|
|
import time
|
|
import urllib.request
|
|
import urllib.error
|
|
from pathlib import Path
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
|
|
RESULTS_DIR = Path(__file__).parent / "results"
|
|
CANDIDATES_FILE = RESULTS_DIR / "candidates.json"
|
|
|
|
# ── High-signal key prefixes ────────────────────────────────────
|
|
KEY_PREFIXES = [
|
|
"sk-proj-", "sk-ant-", "AIza", "hf_", "gsk_", "r8_",
|
|
"sk-or-", "pplx-", "sk-cp-", "sk-sp-", "sk-tp-", "sk-zy-",
|
|
"fe_oa_", "ak_", "sk-xt-",
|
|
]
|
|
|
|
# ── Env var names by provider ───────────────────────────────────
|
|
ENV_VARS = [
|
|
"OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY", "GEMINI_API_KEY",
|
|
"HUGGINGFACE_TOKEN", "HUGGINGFACE_API_TOKEN", "GROQ_API_KEY",
|
|
"REPLICATE_API_TOKEN", "TOGETHER_API_KEY", "DEEPSEEK_API_KEY",
|
|
"OPENROUTER_API_KEY", "PERPLEXITY_API_KEY", "MISTRAL_API_KEY",
|
|
"COHERE_API_KEY", "AZURE_OPENAI_API_KEY",
|
|
# China
|
|
"ARK_API_KEY", "VOLC_API_KEY", "VOLCENGINE_API_KEY",
|
|
"ZHIPUAI_API_KEY", "GLM_API_KEY", "CHATGLM_API_KEY",
|
|
"DASHSCOPE_API_KEY", "ALIBABA_API_KEY",
|
|
"MOONSHOT_API_KEY", "KIMI_API_KEY",
|
|
"MINIMAX_API_KEY", "HUNYUAN_API_KEY", "TENCENT_API_KEY",
|
|
"QIANFAN_API_KEY", "BAIDU_API_KEY",
|
|
"SPARK_API_KEY", "IFLYTEK_API_KEY", "ASTRON_API_KEY",
|
|
"LINGYIWANWU_API_KEY", "YI_API_KEY",
|
|
"STEPFUN_API_KEY", "SILICONFLOW_API_KEY",
|
|
"CSDN_API_KEY", "CSDN_CODING_PLAN_KEY", "STARMAP_API_KEY",
|
|
"HUAWEI_API_KEY", "HUAWEICLOUD_API_KEY", "CODEARTS_API_KEY",
|
|
"MIMO_API_KEY", "XIAOMI_API_KEY", "INFINI_API_KEY", "INFINIAI_API_KEY",
|
|
"JD_API_KEY", "JDCLOUD_API_KEY", "MTHREADS_API_KEY",
|
|
"KWAIKAT_API_KEY", "KUAISHOU_API_KEY", "STREAMLAKE_API_KEY",
|
|
"UCLOUD_API_KEY", "COMPSHARE_API_KEY", "ANOMALY_API_KEY",
|
|
"OPENCODE_GO_KEY", "UNICOM_API_KEY", "CUCLOUD_API_KEY", "YUANJING_API_KEY",
|
|
"SCNET_API_KEY", "ALIBABA_CODING_PLAN_KEY",
|
|
"ZYLOO_API_KEY", "ZYLOO_KEY", "FREEMODEL_API_KEY", "FREEMODEL_KEY",
|
|
"OLLAMA_API_KEY", "XKIRO_API_KEY", "XKIRO_KEY",
|
|
]
|
|
|
|
# ── API endpoint URLs (search for these in code = likely key nearby) ──
|
|
ENDPOINT_URLS = [
|
|
"api.openai.com", "api.anthropic.com", "generativelanguage.googleapis.com",
|
|
"huggingface.co", "api.groq.com", "api.replicate.com", "api.together.xyz",
|
|
"api.deepseek.com", "openrouter.ai", "api.perplexity.ai",
|
|
"ark.cn-beijing.volces.com", "open.bigmodel.cn",
|
|
"dashscope.aliyuncs.com", "api.moonshot.cn", "api.minimaxi.com",
|
|
"api.hunyuan.cloud.tencent.com", "qianfan.baidubce.com",
|
|
"spark-api.xf-yun.com", "maas-coding-api.xf-yun.com",
|
|
"api.lingyiwanwu.com", "api.stepfun.com", "api.siliconflow.cn",
|
|
"ai.csdn.net", "api.xiaomimimo.com", "cloud.infini-ai.com",
|
|
"code.mthreads.com", "streamlake.com", "compshare.cn",
|
|
"opencode.ai", "cucloud.cn", "scnet.cn",
|
|
"coding.dashscope.aliyuncs.com", "api.zyloo.io",
|
|
"freemodel.dev", "cc.freemodel.dev", "api.ollama.com",
|
|
"api.longcat.chat", "api.xkiro.com",
|
|
]
|
|
|
|
# ── Model names (search for these = key assignment often nearby) ──
|
|
MODEL_NAMES = [
|
|
"gpt-4o", "gpt-4o-mini", "claude-sonnet-4", "claude-opus-4",
|
|
"gemini-1.5-flash", "gemini-2.0-flash",
|
|
"doubao-seed-2-0-pro", "glm-4-flash", "glm-5",
|
|
"qwen-plus", "moonshot-v1", "MiniMax-M2.5",
|
|
"deepseek-chat", "deepseek-v4-pro",
|
|
"yi-large", "step-1-flash", "LongCat-2.0",
|
|
"astron-code-latest", "glm_for_coding",
|
|
"anthropic/claude-sonnet-4-5", "anthropic/claude-opus-4-5",
|
|
]
|
|
|
|
# ── File extensions to search ───────────────────────────────────
|
|
EXTENSIONS = [
|
|
"env", "py", "js", "ts", "sh", "cfg", "ini", "ipynb",
|
|
"yaml", "yml", "json", "toml", "tf", "rb", "go", "rs",
|
|
"java", "kt", "php", "lua",
|
|
]
|
|
|
|
# ── Special filenames to search (high-value targets) ────────────
|
|
HOT_FILENAMES = [
|
|
".env", ".env.local", ".env.production", ".env.development",
|
|
"config.json", "config.yaml", "config.yml",
|
|
"secrets.yaml", "secrets.json", "secrets.toml",
|
|
"api_keys.txt", "api_keys.json", "keys.json",
|
|
"credentials.json", "credentials.yaml",
|
|
"settings.json", "settings.yaml",
|
|
"application.yml", "application.properties",
|
|
"docker-compose.yml", "docker-compose.yaml",
|
|
]
|
|
|
|
# ── GitHub topics to discover repos for targeted scanning ───────
|
|
GH_TOPICS = [
|
|
"openai", "anthropic", "llm", "chatgpt", "gpt",
|
|
"claude", "gemini", "ai-api", "language-model",
|
|
"chatbot", "ai-agent", "rag", "langchain",
|
|
"dashscope", "qwen", "chatglm", "zhipu",
|
|
"doubao", "kimi", "moonshot", "minimax",
|
|
"deepseek", "spark-llm", "iflytek",
|
|
]
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════
|
|
# Query Generation
|
|
# ═══════════════════════════════════════════════════════════════
|
|
|
|
def gen_code_queries():
|
|
"""Generate GitHub code search queries.
|
|
|
|
Strategy:
|
|
1. Key prefix + extension (high signal, catches raw keys in code)
|
|
2. Env var name + extension (catches assignments like OPENAI_API_KEY=sk-...)
|
|
3. Endpoint URL + extension (catches config near API calls)
|
|
4. Model name + extension (catches config near model usage)
|
|
"""
|
|
queries = []
|
|
|
|
# 1. Key prefix searches — only in code-like files
|
|
code_exts = ["env", "py", "js", "ts", "sh", "cfg", "ini", "ipynb",
|
|
"yaml", "yml", "json", "toml", "tf", "go", "rs", "rb"]
|
|
for prefix in KEY_PREFIXES:
|
|
for ext in code_exts:
|
|
queries.append(f"{prefix} extension:{ext}")
|
|
|
|
# 2. Env var searches — in all file types
|
|
all_exts = code_exts + ["lua", "java", "kt", "php"]
|
|
for var in ENV_VARS:
|
|
for ext in all_exts:
|
|
queries.append(f"{var} extension:{ext}")
|
|
|
|
# 3. Endpoint URL searches — in code files
|
|
for url in ENDPOINT_URLS:
|
|
for ext in ["py", "js", "ts", "env", "yaml", "yml", "json", "toml", "ipynb"]:
|
|
queries.append(f"{url} extension:{ext}")
|
|
|
|
# 4. Model name searches — in config/code files
|
|
for model in MODEL_NAMES:
|
|
for ext in ["py", "js", "ts", "env", "yaml", "yml", "json", "ipynb"]:
|
|
queries.append(f"{model} extension:{ext}")
|
|
|
|
# 5. Language-specific patterns
|
|
# process.env in JS/TS
|
|
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
|
"GROQ_API_KEY", "HUGGINGFACE_TOKEN", "DEEPSEEK_API_KEY"]:
|
|
queries.append(f"process.env.{var} extension:js")
|
|
queries.append(f"process.env.{var} extension:ts")
|
|
|
|
# os.environ in Python
|
|
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
|
"GROQ_API_KEY", "DEEPSEEK_API_KEY", "DASHSCOPE_API_KEY",
|
|
"ARK_API_KEY", "ZHIPUAI_API_KEY"]:
|
|
queries.append(f"os.environ {var} extension:py")
|
|
queries.append(f"os.getenv {var} extension:py")
|
|
|
|
# Dockerfile ENV patterns
|
|
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
|
"GROQ_API_KEY", "DEEPSEEK_API_KEY", "ARK_API_KEY",
|
|
"DASHSCOPE_API_KEY", "ZHIPUAI_API_KEY"]:
|
|
queries.append(f"ENV {var} extension:Dockerfile")
|
|
queries.append(f"ARG {var} extension:Dockerfile")
|
|
|
|
# CI/CD secret patterns
|
|
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
|
"DEEPSEEK_API_KEY", "ARK_API_KEY"]:
|
|
queries.append(f"secrets.{var} extension:yml")
|
|
queries.append(f"secrets.{var} extension:yaml")
|
|
|
|
return queries
|
|
|
|
|
|
def gen_filename_queries():
|
|
"""Generate filename-based searches — find high-value config files."""
|
|
queries = []
|
|
for fname in HOT_FILENAMES:
|
|
# Search for the filename itself — these files often contain keys
|
|
queries.append(f"filename:{fname}")
|
|
return queries
|
|
|
|
|
|
def gen_commit_queries():
|
|
"""Generate commit message search queries.
|
|
|
|
Keys are sometimes accidentally committed in commit messages.
|
|
Uses search/commits API.
|
|
"""
|
|
queries = []
|
|
for prefix in KEY_PREFIXES:
|
|
queries.append(prefix)
|
|
for var in ["OPENAI_API_KEY", "ANTHROPIC_API_KEY", "GOOGLE_API_KEY",
|
|
"DEEPSEEK_API_KEY", "ARK_API_KEY", "DASHSCOPE_API_KEY"]:
|
|
queries.append(var)
|
|
return queries
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════
|
|
# GitHub API Client
|
|
# ═══════════════════════════════════════════════════════════════
|
|
|
|
class GitHubSearcher:
|
|
"""GitHub search client with rate-limit awareness and token rotation."""
|
|
|
|
def __init__(self, tokens=None, per_page=100, verbose=True):
|
|
self.tokens = tokens or [os.environ.get("GH_TOKEN", "")]
|
|
self.tokens = [t for t in self.tokens if t]
|
|
if not self.tokens:
|
|
print("ERROR: No GitHub tokens provided. Set GH_TOKEN env var or use --token.")
|
|
sys.exit(1)
|
|
self._token_idx = 0
|
|
self.per_page = per_page
|
|
self.verbose = verbose
|
|
self._rate_remaining = 30
|
|
self._rate_reset = 0
|
|
|
|
def _next_token(self):
|
|
"""Rotate to the next token."""
|
|
self._token_idx = (self._token_idx + 1) % len(self.tokens)
|
|
return self.tokens[self._token_idx]
|
|
|
|
def _current_token(self):
|
|
return self.tokens[self._token_idx]
|
|
|
|
def _api(self, endpoint, params=None):
|
|
"""Call GitHub API with rate-limit handling and token rotation.
|
|
|
|
Returns (data, error). On rate limit, waits and retries.
|
|
"""
|
|
import urllib.parse
|
|
url = f"https://api.github.com/{endpoint}"
|
|
if params:
|
|
url += "?" + urllib.parse.urlencode(params)
|
|
|
|
for attempt in range(len(self.tokens) * 2):
|
|
token = self._current_token()
|
|
req = urllib.request.Request(url, headers={
|
|
"Authorization": f"token {token}",
|
|
"Accept": "application/vnd.github+json",
|
|
"User-Agent": "llm-key-hunter/2.0",
|
|
})
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=30) as resp:
|
|
# Track rate limit from headers
|
|
remaining = resp.headers.get("X-RateLimit-Remaining")
|
|
reset = resp.headers.get("X-RateLimit-Reset")
|
|
if remaining:
|
|
self._rate_remaining = int(remaining)
|
|
if reset:
|
|
self._rate_reset = int(reset)
|
|
return json.loads(resp.read()), None
|
|
except urllib.error.HTTPError as e:
|
|
if e.code == 403:
|
|
# Rate limited or forbidden
|
|
remaining = e.headers.get("X-RateLimit-Remaining", "0")
|
|
reset = e.headers.get("X-RateLimit-Reset", "0")
|
|
if remaining == "0" and reset:
|
|
wait = int(reset) - int(time.time()) + 2
|
|
if wait > 0 and wait < 3600:
|
|
if self.verbose:
|
|
print(f" [rate-limit] Waiting {wait}s for reset...")
|
|
time.sleep(wait)
|
|
continue
|
|
# Try rotating token
|
|
self._next_token()
|
|
continue
|
|
elif e.code == 422:
|
|
# Unprocessable — bad query, skip
|
|
return None, f"422: {e.read()[:200]}"
|
|
else:
|
|
return None, f"HTTP {e.code}"
|
|
except Exception as e:
|
|
return None, str(e)
|
|
|
|
return None, "Exhausted all tokens + retries"
|
|
|
|
def search_code(self, query, max_results=1000):
|
|
"""Search GitHub code. Returns list of candidate dicts."""
|
|
results = []
|
|
page = 1
|
|
while len(results) < max_results:
|
|
data, err = self._api("search/code", {
|
|
"q": query,
|
|
"per_page": self.per_page,
|
|
"page": page,
|
|
})
|
|
if err:
|
|
if self.verbose and "422" not in str(err):
|
|
print(f" [error] {err}")
|
|
break
|
|
items = data.get("items", [])
|
|
if not items:
|
|
break
|
|
for item in items:
|
|
results.append({
|
|
"repository": {"full_name": item["repository"]["full_name"]},
|
|
"path": item["path"],
|
|
"url": item["html_url"],
|
|
})
|
|
if len(items) < self.per_page:
|
|
break
|
|
page += 1
|
|
# Be nice to the API
|
|
if self._rate_remaining < 5:
|
|
wait = max(self._rate_reset - int(time.time()) + 2, 2)
|
|
if self.verbose:
|
|
print(f" [rate-limit] {self._rate_remaining} left, waiting {wait}s...")
|
|
time.sleep(min(wait, 120))
|
|
else:
|
|
time.sleep(0.2)
|
|
return results
|
|
|
|
def search_commits(self, query, max_results=500):
|
|
"""Search GitHub commit messages for key patterns."""
|
|
results = []
|
|
data, err = self._api("search/commits", {
|
|
"q": query,
|
|
"per_page": min(self.per_page, 100),
|
|
})
|
|
if err:
|
|
return []
|
|
for item in data.get("items", []):
|
|
repo = item.get("repository", {})
|
|
results.append({
|
|
"repository": {"full_name": repo.get("full_name", "")},
|
|
"path": "(commit message)",
|
|
"url": item.get("html_url", ""),
|
|
})
|
|
return results
|
|
|
|
def search_repos_by_topic(self, topic, max_results=100):
|
|
"""Find repos by topic for targeted scanning."""
|
|
data, err = self._api("search/repositories", {
|
|
"q": f"topic:{topic}",
|
|
"per_page": min(max_results, 100),
|
|
"sort": "updated",
|
|
})
|
|
if err:
|
|
return []
|
|
return [item["full_name"] for item in data.get("items", [])]
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════
|
|
# Gitee Search (Chinese GitHub mirror)
|
|
# ═══════════════════════════════════════════════════════════════
|
|
|
|
def gitee_search(query, token=None, max_results=100):
|
|
"""Search Gitee.com for key patterns. Returns candidate dicts."""
|
|
import urllib.parse
|
|
url = f"https://gitee.com/api/v5/search/code?q={urllib.parse.quote(query)}&per_page=20"
|
|
headers = {"User-Agent": "llm-key-hunter/2.0"}
|
|
if token:
|
|
url += f"&access_token={token}"
|
|
req = urllib.request.Request(url, headers=headers)
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
items = json.loads(resp.read())
|
|
return [{
|
|
"repository": {"full_name": f"gitee:{item.get('repository', {}).get('full_name', '')}"},
|
|
"path": item.get("path", ""),
|
|
"url": item.get("html_url", ""),
|
|
} for item in items[:max_results]]
|
|
except Exception:
|
|
return []
|
|
|
|
|
|
# ═══════════════════════════════════════════════════════════════
|
|
# Main Pipeline
|
|
# ═══════════════════════════════════════════════════════════════
|
|
|
|
def main():
|
|
parser = argparse.ArgumentParser(
|
|
description="GitHub search engine for LLM API key leaks")
|
|
parser.add_argument("--token", action="append", default=[],
|
|
help="GitHub token (can repeat for multi-token rotation)")
|
|
parser.add_argument("--gitee-token", default=None,
|
|
help="Gitee API token for Chinese mirror search")
|
|
parser.add_argument("--output", default=str(CANDIDATES_FILE),
|
|
help="Output candidates JSON file")
|
|
parser.add_argument("--no-commits", action="store_true",
|
|
help="Skip commit message search")
|
|
parser.add_argument("--no-filenames", action="store_true",
|
|
help="Skip filename-based search")
|
|
parser.add_argument("--no-gitee", action="store_true",
|
|
help="Skip Gitee search")
|
|
parser.add_argument("--no-topics", action="store_true",
|
|
help="Skip topic-based repo discovery")
|
|
parser.add_argument("--max-results", type=int, default=1000,
|
|
help="Max results per query (default: 1000)")
|
|
parser.add_argument("--verbose", "-v", action="store_true", default=True)
|
|
parser.add_argument("--resume", action="store_true",
|
|
help="Resume from checkpoint file")
|
|
args = parser.parse_args()
|
|
|
|
# Collect tokens
|
|
tokens = args.token or []
|
|
env_token = os.environ.get("GH_TOKEN", "")
|
|
if env_token and env_token not in tokens:
|
|
tokens.append(env_token)
|
|
if not tokens:
|
|
print("ERROR: No GitHub tokens. Use --token or set GH_TOKEN env var.")
|
|
sys.exit(1)
|
|
|
|
searcher = GitHubSearcher(tokens=tokens, verbose=args.verbose)
|
|
|
|
# Generate queries
|
|
code_queries = gen_code_queries()
|
|
filename_queries = gen_filename_queries() if not args.no_filenames else []
|
|
commit_queries = gen_commit_queries() if not args.no_commits else []
|
|
|
|
total_queries = len(code_queries) + len(filename_queries) + len(commit_queries)
|
|
print(f"Generated {total_queries} queries:")
|
|
print(f" Code search: {len(code_queries)}")
|
|
print(f" Filename search: {len(filename_queries)}")
|
|
print(f" Commit search: {len(commit_queries)}")
|
|
print(f" Tokens: {len(tokens)}")
|
|
print()
|
|
|
|
all_candidates = {} # url -> candidate (dedup by URL)
|
|
processed = 0
|
|
start_idx = 0
|
|
|
|
# ── Checkpoint / resume ──────────────────────────────────────
|
|
checkpoint_path = Path(args.output).with_suffix(".checkpoint.json")
|
|
progress_path = Path(args.output).with_suffix(".progress.json")
|
|
|
|
def save_checkpoint(proc):
|
|
ckpt = sorted(all_candidates.values(),
|
|
key=lambda c: c["repository"]["full_name"])
|
|
with open(checkpoint_path, "w") as f:
|
|
json.dump(ckpt, f)
|
|
with open(progress_path, "w") as f:
|
|
json.dump({"processed": proc, "candidates": len(all_candidates)}, f)
|
|
|
|
if args.resume and checkpoint_path.exists():
|
|
with open(checkpoint_path) as f:
|
|
for c in json.load(f):
|
|
all_candidates[c["url"]] = c
|
|
if progress_path.exists():
|
|
with open(progress_path) as f:
|
|
start_idx = json.load(f).get("processed", 0)
|
|
else:
|
|
# Legacy checkpoint (written at processed % 50 == 0, before that query ran)
|
|
start_idx = 50
|
|
processed = start_idx
|
|
print(f"Resumed: {len(all_candidates)} candidates, skipping first {start_idx} queries")
|
|
|
|
# ── Code search ──────────────────────────────────────────────
|
|
print("=== Phase 1: Code Search ===")
|
|
for i, query in enumerate(code_queries):
|
|
if i < start_idx:
|
|
continue
|
|
processed = i + 1
|
|
if processed % 50 == 0:
|
|
print(f" [{processed}/{total_queries}] {len(all_candidates)} candidates so far")
|
|
save_checkpoint(processed)
|
|
|
|
results = searcher.search_code(query, max_results=args.max_results)
|
|
for r in results:
|
|
if r["url"] not in all_candidates:
|
|
all_candidates[r["url"]] = r
|
|
|
|
# ── Filename search ──────────────────────────────────────────
|
|
if filename_queries:
|
|
print(f"\n=== Phase 2: Filename Search ===")
|
|
for query in filename_queries:
|
|
processed += 1
|
|
results = searcher.search_code(query, max_results=args.max_results)
|
|
for r in results:
|
|
if r["url"] not in all_candidates:
|
|
all_candidates[r["url"]] = r
|
|
print(f" {len(all_candidates)} total candidates after filename search")
|
|
|
|
# ── Commit search ────────────────────────────────────────────
|
|
if commit_queries:
|
|
print(f"\n=== Phase 3: Commit Message Search ===")
|
|
for query in commit_queries:
|
|
processed += 1
|
|
results = searcher.search_commits(query)
|
|
for r in results:
|
|
if r["url"] not in all_candidates:
|
|
all_candidates[r["url"]] = r
|
|
print(f" {len(all_candidates)} total candidates after commit search")
|
|
|
|
# ── Gitee search ─────────────────────────────────────────────
|
|
if not args.no_gitee:
|
|
print(f"\n=== Phase 4: Gitee Search ===")
|
|
gitee_queries = KEY_PREFIXES + ["OPENAI_API_KEY", "ANTHROPIC_API_KEY",
|
|
"DASHSCOPE_API_KEY", "ARK_API_KEY"]
|
|
for query in gitee_queries:
|
|
results = gitee_search(query, token=args.gitee_token)
|
|
for r in results:
|
|
if r["url"] not in all_candidates:
|
|
all_candidates[r["url"]] = r
|
|
print(f" {len(all_candidates)} total candidates after Gitee search")
|
|
|
|
# ── Topic-based repo discovery ───────────────────────────────
|
|
if not args.no_topics:
|
|
print(f"\n=== Phase 5: Topic-based Repo Discovery ===")
|
|
topic_repos = set()
|
|
for topic in GH_TOPICS:
|
|
repos = searcher.search_repos_by_topic(topic)
|
|
topic_repos.update(repos)
|
|
print(f" Found {len(topic_repos)} repos by topic")
|
|
# For each repo, search for key patterns in code
|
|
for repo in sorted(topic_repos):
|
|
for prefix in KEY_PREFIXES[:5]: # Top 5 prefixes only
|
|
query = f"{prefix} repo:{repo}"
|
|
results = searcher.search_code(query, max_results=100)
|
|
for r in results:
|
|
if r["url"] not in all_candidates:
|
|
all_candidates[r["url"]] = r
|
|
print(f" {len(all_candidates)} total candidates after topic search")
|
|
|
|
# ── Write output ─────────────────────────────────────────────
|
|
candidates = sorted(all_candidates.values(),
|
|
key=lambda c: c["repository"]["full_name"])
|
|
output_path = Path(args.output)
|
|
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
with open(output_path, "w") as f:
|
|
json.dump(candidates, f, indent=2)
|
|
|
|
# Clean up checkpoint
|
|
if checkpoint_path.exists():
|
|
checkpoint_path.unlink()
|
|
if progress_path.exists():
|
|
progress_path.unlink()
|
|
|
|
print(f"\n=== Done! ===")
|
|
print(f" Total queries: {processed}")
|
|
print(f" Candidates: {len(candidates)}")
|
|
print(f" Output: {output_path}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|