#!/usr/bin/env python3 """Deep-verify the 'medium' providers in one pass. Providers: Groq, TogetherAI, OpenAI, Anthropic, SiliconFlow, LingyiWanwu (01), StepFun. For every key in results/extracted_keys.txt whose provider matches, we: 1. Validate the key against the provider's documented regex. 2. Send a cheap chat completion (or /models GET) request. 3. Classify by the now-familiar rule: only 401 = DEAD; anything else (200 / 400 / 402 / 403 / 404 / 429 / 5xx) is kept. Outputs per provider under results/medium//. """ import argparse import json import re import sys import time import urllib.request import urllib.error from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path HERE = Path(__file__).parent sys.path.insert(0, str(HERE)) from verify_cache import CachedVerifier SOURCE = HERE / "results" / "extracted_keys.txt" OUT_ROOT = HERE / "results" / "medium" OUT_ROOT.mkdir(parents=True, exist_ok=True) UA = "curl/8.5.0" # ── Provider configs ──────────────────────────────────────────── # Each entry: # key_re : regex the full key must match (after provider tag) # fakes : substrings that mark a placeholder # requests : list of (name, method, path, headers, body) to try in order. # {key} is substituted; stop early once a better verdict found. def openai_like(base, model, key_header="Authorization", scheme="Bearer"): return [ { "name": f"chat:{model}", "method": "POST", "url": f"{base}/v1/chat/completions", "headers": {key_header: f"{scheme} {{key}}", "Content-Type": "application/json"}, "body": {"model": model, "messages": [{"role": "user", "content": "hi"}], "max_tokens": 1}, }, { "name": "models", "method": "GET", "url": f"{base}/v1/models", "headers": {key_header: f"{scheme} {{key}}"}, "body": None, }, ] PROVIDERS = { "Groq": { "key_re": re.compile(r"^gsk_[A-Za-z0-9]{48,}$"), "fakes": ("your-", "example", "xxxx"), "requests": openai_like("https://api.groq.com", "llama-3.1-8b-instant"), }, "TogetherAI": { "key_re": re.compile(r"^[0-9a-f]{64}$"), "fakes": ("0" * 64,), "requests": openai_like("https://api.together.xyz", "meta-llama/Llama-3.2-3B-Instruct-Turbo"), }, "OpenAI": { # Real OpenAI keys: sk-... (48-ish chars) or sk-proj-... (longer). # We deliberately exclude keys that look like other providers' # (sk-ant handled separately, T3BlbkFJ is a classic OpenAI marker). "key_re": re.compile(r"^sk-(?:proj-[A-Za-z0-9_-]{20,}|[A-Za-z0-9]{48})$"), "fakes": ("your-", "example", "xxxx", "sk-000000"), "requests": openai_like("https://api.openai.com", "gpt-4o-mini"), }, "Anthropic": { "key_re": re.compile(r"^sk-ant-[A-Za-z0-9_\-]{40,}$"), "fakes": ("your-", "example", "xxxx"), "requests": [ { "name": "messages", "method": "POST", "url": "https://api.anthropic.com/v1/messages", "headers": { "x-api-key": "{key}", "anthropic-version": "2023-06-01", "Content-Type": "application/json", }, "body": {"model": "claude-3-5-haiku-20241022", "messages": [{"role": "user", "content": "hi"}], "max_tokens": 1}, }, { "name": "models", "method": "GET", "url": "https://api.anthropic.com/v1/models", "headers": {"x-api-key": "{key}", "anthropic-version": "2023-06-01"}, "body": None, }, ], }, "SiliconFlow": { "key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"), "fakes": ("your-", "example", "xxxx"), "requests": openai_like("https://api.siliconflow.cn", "Qwen/Qwen2.5-7B-Instruct"), }, "LingyiWanwu": { "key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"), "fakes": ("your-", "example", "xxxx"), "requests": openai_like("https://api.lingyiwanwu.com", "yi-large"), }, "StepFun": { "key_re": re.compile(r"^sk-[a-zA-Z0-9]{40,}$"), "fakes": ("your-", "example", "xxxx"), "requests": openai_like("https://api.stepfun.com", "step-1-flash"), }, } RANK = {"USABLE": 4, "NO_BALANCE": 3, "NO_ACCESS": 2, "UNKNOWN": 1, "DEAD": 0} def http_request(method, url, headers, body, timeout=20): h = {"User-Agent": UA, "Accept": "*/*", **headers} data = json.dumps(body).encode() if body is not None else None req = urllib.request.Request(url, data=data, headers=h, method=method) try: with urllib.request.urlopen(req, timeout=timeout) as resp: return resp.getcode(), resp.read().decode("utf-8", errors="replace") except urllib.error.HTTPError as e: try: return e.code, e.read().decode("utf-8", errors="replace") except Exception: return e.code, "" except Exception as e: return 0, f"network: {type(e).__name__}: {e}" def classify(code, body): bl = (body or "").lower() if code == 200: if '"choices"' in bl or '"data"' in bl or '"id"' in bl: return "USABLE", "200 OK" if any(x in bl for x in ("balance", "quota", "arrearage", "insufficient")): return "NO_BALANCE", body[:140] return "USABLE", body[:140] if code in (402, 429): return "NO_BALANCE", f"HTTP {code}: {body[:140]}" if code == 401: return "DEAD", "401 unauthorized" if code == 0: return "UNKNOWN", body[:160] # 400 / 403 / 404 / 5xx — key may be valid, model/endpoint issue return "NO_ACCESS", f"HTTP {code}: {body[:140]}" def verify(provider, key): cfg = PROVIDERS[provider] best, best_detail = "DEAD", "" for req in cfg["requests"]: headers = {h: v.replace("{key}", key) for h, v in req["headers"].items()} code, body = http_request(req["method"], req["url"], headers, req["body"]) v, d = classify(code, body) d = f"[{req['name']}] {d}" if RANK[v] > RANK[best]: best, best_detail = v, d if best == "USABLE": return best, best_detail return best, best_detail def load_keys(provider): cfg = PROVIDERS[provider] keys = {} if not SOURCE.exists(): return keys with open(SOURCE) as f: for line in f: line = line.strip() if not line.startswith(f"{provider}|"): continue parts = line.split("|", 3) if len(parts) < 3: continue key = parts[1] url = parts[2] if len(parts) > 2 else "" low = key.lower() if any(x in low for x in cfg["fakes"]): continue if not cfg["key_re"].match(key): continue if key not in keys: keys[key] = url return keys def run_provider(provider, workers, limit, force_cache=False): keys = load_keys(provider) out_dir = OUT_ROOT / provider out_dir.mkdir(parents=True, exist_ok=True) print(f"\n=== {provider}: {len(keys)} unique keys matching regex ===") if limit: keys = dict(list(keys.items())[:limit]) print(f" (limited to first {limit})") if not keys: return buckets = {k: [] for k in RANK} start = time.time() processed = 0 with CachedVerifier(f"medium_{provider.lower()}", lambda k, _p=provider: verify(_p, k), force=force_cache) as ver: with ThreadPoolExecutor(max_workers=workers) as pool: futs = {pool.submit(ver, k): (k, s) for k, s in keys.items()} for fut in as_completed(futs): k, s = futs[fut] processed += 1 try: verdict, detail = fut.result() except Exception as e: verdict, detail = "UNKNOWN", str(e) buckets[verdict].append((k, s, detail)) if processed % 50 == 0: el = time.time() - start print(f" [{processed}/{len(keys)}] " f"usable={len(buckets['USABLE'])} " f"nobal={len(buckets['NO_BALANCE'])} " f"noacc={len(buckets['NO_ACCESS'])} " f"unk={len(buckets['UNKNOWN'])} " f"dead={len(buckets['DEAD'])} " f"hit={ver.hits} live={ver.live} " f"({processed/el:.1f}/s)") print(f" cache: {ver.hits} hits, {ver.live} live, {len(ver._cache)} cached") el = time.time() - start print(f" done in {el:.1f}s") for n in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"): print(f" {n:11s}: {len(buckets[n])}") name_map = { "USABLE": "usable.txt", "NO_BALANCE": "no_balance.txt", "NO_ACCESS": "no_access.txt", "UNKNOWN": "unknown.txt", "DEAD": "dead.txt", } for n, fn in name_map.items(): with open(out_dir / fn, "w") as f: for k, s, d in sorted(buckets[n]): f.write(f"{k}|{s}|{d}\n") with open(out_dir / "all_non_401.txt", "w") as f: for n in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN"): for k, s, d in sorted(buckets[n]): f.write(f"{n}|{k}|{s}|{d}\n") if buckets["USABLE"]: print(f" --- USABLE {provider} keys ---") for k, s, d in buckets["USABLE"][:30]: print(f" {k} {d[:80]}") print(f" src: {s}") def main(): ap = argparse.ArgumentParser() ap.add_argument("--providers", default="all", help="Comma-separated list, or 'all'") ap.add_argument("--workers", type=int, default=15) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--no-cache", action="store_true") args = ap.parse_args() if args.providers == "all": providers = list(PROVIDERS.keys()) else: providers = [p.strip() for p in args.providers.split(",") if p.strip() in PROVIDERS] unknown = [p for p in args.providers.split(",") if p.strip() not in PROVIDERS] if unknown: print(f"Unknown providers: {unknown}", file=sys.stderr) print(f"Verifying: {', '.join(providers)}") for p in providers: run_provider(p, args.workers, args.limit, args.no_cache) if __name__ == "__main__": main()