- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
116 lines
3.4 KiB
Python
116 lines
3.4 KiB
Python
#!/usr/bin/env python3
|
|
"""Extract LLM API keys from candidate files — Python version (reliable)."""
|
|
|
|
import json
|
|
import re
|
|
import urllib.request
|
|
from pathlib import Path
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
|
|
RESULTS_DIR = Path(__file__).parent / "results"
|
|
CANDIDATES_FILE = RESULTS_DIR / "candidates.json"
|
|
PATTERNS_FILE = Path(__file__).parent / "patterns.conf"
|
|
OUTPUT_FILE = RESULTS_DIR / "extracted_keys.txt"
|
|
UNIQUE_FILE = RESULTS_DIR / "unique_keys.txt"
|
|
|
|
# Load patterns
|
|
patterns = []
|
|
with open(PATTERNS_FILE) as f:
|
|
for line in f:
|
|
line = line.strip()
|
|
if not line or line.startswith("#"):
|
|
continue
|
|
parts = line.split("|", 2)
|
|
if len(parts) == 3:
|
|
provider, regex, desc = parts
|
|
try:
|
|
patterns.append((provider, re.compile(regex), desc))
|
|
except re.error:
|
|
pass
|
|
|
|
print(f"Loaded {len(patterns)} regex patterns")
|
|
|
|
# Load candidates
|
|
with open(CANDIDATES_FILE) as f:
|
|
candidates = json.load(f)
|
|
|
|
print(f"Loaded {len(candidates)} candidate files")
|
|
|
|
|
|
def to_raw_url(url):
|
|
"""Convert GitHub/Gitee URL to raw content URL."""
|
|
if "github.com" in url:
|
|
return url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/")
|
|
elif "gitee.com" in url:
|
|
# Gitee raw URL format: https://gitee.com/user/repo/raw/branch/path
|
|
return url.replace("/blob/", "/raw/")
|
|
return url
|
|
|
|
|
|
def fetch_url(url):
|
|
"""Fetch URL content using urllib (no subprocess overhead)."""
|
|
raw_url = to_raw_url(url)
|
|
req = urllib.request.Request(raw_url, headers={"User-Agent": "Mozilla/5.0"})
|
|
try:
|
|
with urllib.request.urlopen(req, timeout=10) as resp:
|
|
return url, resp.read().decode("utf-8", errors="replace")
|
|
except Exception:
|
|
return url, ""
|
|
|
|
|
|
def extract_from_content(url, content):
|
|
"""Extract API keys from content using loaded regex patterns."""
|
|
found = []
|
|
for provider, regex, desc in patterns:
|
|
for match in regex.finditer(content):
|
|
key = match.group()
|
|
if len(key) < 10:
|
|
continue
|
|
found.append(f"{provider}|{key}|{url}|{desc}")
|
|
return found
|
|
|
|
|
|
# Process with thread pool
|
|
all_keys = set()
|
|
processed = 0
|
|
|
|
with ThreadPoolExecutor(max_workers=20) as pool:
|
|
futures = {pool.submit(fetch_url, c["url"]): c["url"] for c in candidates}
|
|
|
|
for future in as_completed(futures):
|
|
url = futures[future]
|
|
processed += 1
|
|
|
|
try:
|
|
_, content = future.result()
|
|
except Exception:
|
|
content = ""
|
|
|
|
if content:
|
|
keys = extract_from_content(url, content)
|
|
all_keys.update(keys)
|
|
|
|
if processed % 500 == 0:
|
|
print(f" Processed {processed}/{len(candidates)} — {len(all_keys)} unique keys so far")
|
|
|
|
print(f"\nDone: {processed} files processed, {len(all_keys)} unique keys found")
|
|
|
|
# Write output
|
|
with open(OUTPUT_FILE, "w") as f:
|
|
for k in sorted(all_keys):
|
|
f.write(k + "\n")
|
|
|
|
# Also write unique keys (just provider|key)
|
|
unique_keys = set()
|
|
for k in all_keys:
|
|
parts = k.split("|")
|
|
if len(parts) >= 2:
|
|
unique_keys.add(f"{parts[0]}|{parts[1]}")
|
|
|
|
with open(UNIQUE_FILE, "w") as f:
|
|
for k in sorted(unique_keys):
|
|
f.write(k + "\n")
|
|
|
|
print(f"Written: {OUTPUT_FILE} ({len(all_keys)} lines)")
|
|
print(f"Written: {UNIQUE_FILE} ({len(unique_keys)} lines)")
|