Files
hack/tools/scripts/llm-key-hunter/extract.py
T
chaos 5d215e1649 Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*,
  pivot miner, two-layer verify/content caches, per-provider verification)
- usable_keys: verified key vault across 12 providers (deepseek, minimax,
  volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.)
- .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow
- NewAPI channel import scripts and CDP capture helpers
- Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
2026-08-02 06:02:58 +08:00

116 lines
3.4 KiB
Python

#!/usr/bin/env python3
"""Extract LLM API keys from candidate files — Python version (reliable)."""
import json
import re
import urllib.request
from pathlib import Path
from concurrent.futures import ThreadPoolExecutor, as_completed
RESULTS_DIR = Path(__file__).parent / "results"
CANDIDATES_FILE = RESULTS_DIR / "candidates.json"
PATTERNS_FILE = Path(__file__).parent / "patterns.conf"
OUTPUT_FILE = RESULTS_DIR / "extracted_keys.txt"
UNIQUE_FILE = RESULTS_DIR / "unique_keys.txt"
# Load patterns
patterns = []
with open(PATTERNS_FILE) as f:
for line in f:
line = line.strip()
if not line or line.startswith("#"):
continue
parts = line.split("|", 2)
if len(parts) == 3:
provider, regex, desc = parts
try:
patterns.append((provider, re.compile(regex), desc))
except re.error:
pass
print(f"Loaded {len(patterns)} regex patterns")
# Load candidates
with open(CANDIDATES_FILE) as f:
candidates = json.load(f)
print(f"Loaded {len(candidates)} candidate files")
def to_raw_url(url):
"""Convert GitHub/Gitee URL to raw content URL."""
if "github.com" in url:
return url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/")
elif "gitee.com" in url:
# Gitee raw URL format: https://gitee.com/user/repo/raw/branch/path
return url.replace("/blob/", "/raw/")
return url
def fetch_url(url):
"""Fetch URL content using urllib (no subprocess overhead)."""
raw_url = to_raw_url(url)
req = urllib.request.Request(raw_url, headers={"User-Agent": "Mozilla/5.0"})
try:
with urllib.request.urlopen(req, timeout=10) as resp:
return url, resp.read().decode("utf-8", errors="replace")
except Exception:
return url, ""
def extract_from_content(url, content):
"""Extract API keys from content using loaded regex patterns."""
found = []
for provider, regex, desc in patterns:
for match in regex.finditer(content):
key = match.group()
if len(key) < 10:
continue
found.append(f"{provider}|{key}|{url}|{desc}")
return found
# Process with thread pool
all_keys = set()
processed = 0
with ThreadPoolExecutor(max_workers=20) as pool:
futures = {pool.submit(fetch_url, c["url"]): c["url"] for c in candidates}
for future in as_completed(futures):
url = futures[future]
processed += 1
try:
_, content = future.result()
except Exception:
content = ""
if content:
keys = extract_from_content(url, content)
all_keys.update(keys)
if processed % 500 == 0:
print(f" Processed {processed}/{len(candidates)} — {len(all_keys)} unique keys so far")
print(f"\nDone: {processed} files processed, {len(all_keys)} unique keys found")
# Write output
with open(OUTPUT_FILE, "w") as f:
for k in sorted(all_keys):
f.write(k + "\n")
# Also write unique keys (just provider|key)
unique_keys = set()
for k in all_keys:
parts = k.split("|")
if len(parts) >= 2:
unique_keys.add(f"{parts[0]}|{parts[1]}")
with open(UNIQUE_FILE, "w") as f:
for k in sorted(unique_keys):
f.write(k + "\n")
print(f"Written: {OUTPUT_FILE} ({len(all_keys)} lines)")
print(f"Written: {UNIQUE_FILE} ({len(unique_keys)} lines)")