#!/usr/bin/env python3 """Extract LLM API keys from candidate files — Python version (reliable).""" import json import re import urllib.request from pathlib import Path from concurrent.futures import ThreadPoolExecutor, as_completed RESULTS_DIR = Path(__file__).parent / "results" CANDIDATES_FILE = RESULTS_DIR / "candidates.json" PATTERNS_FILE = Path(__file__).parent / "patterns.conf" OUTPUT_FILE = RESULTS_DIR / "extracted_keys.txt" UNIQUE_FILE = RESULTS_DIR / "unique_keys.txt" # Load patterns patterns = [] with open(PATTERNS_FILE) as f: for line in f: line = line.strip() if not line or line.startswith("#"): continue parts = line.split("|", 2) if len(parts) == 3: provider, regex, desc = parts try: patterns.append((provider, re.compile(regex), desc)) except re.error: pass print(f"Loaded {len(patterns)} regex patterns") # Load candidates with open(CANDIDATES_FILE) as f: candidates = json.load(f) print(f"Loaded {len(candidates)} candidate files") def to_raw_url(url): """Convert GitHub/Gitee URL to raw content URL.""" if "github.com" in url: return url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/") elif "gitee.com" in url: # Gitee raw URL format: https://gitee.com/user/repo/raw/branch/path return url.replace("/blob/", "/raw/") return url def fetch_url(url): """Fetch URL content using urllib (no subprocess overhead).""" raw_url = to_raw_url(url) req = urllib.request.Request(raw_url, headers={"User-Agent": "Mozilla/5.0"}) try: with urllib.request.urlopen(req, timeout=10) as resp: return url, resp.read().decode("utf-8", errors="replace") except Exception: return url, "" def extract_from_content(url, content): """Extract API keys from content using loaded regex patterns.""" found = [] for provider, regex, desc in patterns: for match in regex.finditer(content): key = match.group() if len(key) < 10: continue found.append(f"{provider}|{key}|{url}|{desc}") return found # Process with thread pool all_keys = set() processed = 0 with ThreadPoolExecutor(max_workers=20) as pool: futures = {pool.submit(fetch_url, c["url"]): c["url"] for c in candidates} for future in as_completed(futures): url = futures[future] processed += 1 try: _, content = future.result() except Exception: content = "" if content: keys = extract_from_content(url, content) all_keys.update(keys) if processed % 500 == 0: print(f" Processed {processed}/{len(candidates)} — {len(all_keys)} unique keys so far") print(f"\nDone: {processed} files processed, {len(all_keys)} unique keys found") # Write output with open(OUTPUT_FILE, "w") as f: for k in sorted(all_keys): f.write(k + "\n") # Also write unique keys (just provider|key) unique_keys = set() for k in all_keys: parts = k.split("|") if len(parts) >= 2: unique_keys.add(f"{parts[0]}|{parts[1]}") with open(UNIQUE_FILE, "w") as f: for k in sorted(unique_keys): f.write(k + "\n") print(f"Written: {OUTPUT_FILE} ({len(all_keys)} lines)") print(f"Written: {UNIQUE_FILE} ({len(unique_keys)} lines)")