#!/usr/bin/env python3 """Hunt & verify the remaining Chinese coding-plan / token-plan providers. Providers covered (all OpenAI-compatible unless noted): - SCNet (超算互联网) sk-sp- / sk-tp- api.scnet.cn - Zyloo sk-zy- api.zyloo.io - LongCat (美团龙猫) ak_ (29 chars) api.longcat.chat - OllamaCloud <32hex>.<24alnum> api.ollama.com (native /api/chat) - iFlytek Astron (讯飞星辰) 32-hex maas-coding-api.xf-yun.com iFlytek Astron keys in the pool are EXTREMELY noisy (the bare 32-hex pattern matches Azure/AWS keys, test vectors, etc.). We only test a 32-hex value when its source file / description references an xf-yun / astron / maas endpoint, and even then we filter out low-entropy placeholders. Only HTTP 401 => DEAD. Real chat completion 200 with choices => USABLE. Uses verify_cache so already-tested keys are not re-queried. """ import argparse, json, os, re, sys, time import urllib.request, urllib.error from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from verify_cache import CachedVerifier OUT = Path("results/cn_coding"); OUT.mkdir(parents=True, exist_ok=True) POOL = Path("results/extracted_keys.txt") UA = "curl/8.5.0" # ── provider config ──────────────────────────────────────────── # model is the cheapest chat model for the auth/balance probe. PROVIDERS = { "SCNet": { "chat": "https://api.scnet.cn/api/llm/v1/chat/completions", "models": "https://api.scnet.cn/api/llm/v1/models", "model": "deepseek-v3", "auth": "Bearer", }, "Zyloo": { "chat": "https://api.zyloo.io/v1/chat/completions", "models": "https://api.zyloo.io/v1/models", "model": "zyloo/claude-opus-4-7", "auth": "Bearer", }, "LongCat": { "chat": "https://api.longcat.chat/openai/chat/completions", "models": "https://api.longcat.chat/openai/models", "model": "LongCat-2.0", "auth": "Bearer", }, "OllamaCloud": { # native ollama api; model probed at runtime "chat": "https://api.ollama.com/api/chat", "models": "https://api.ollama.com/api/tags", "model": None, "auth": "Bearer", "native": True, }, "iFlytekAstron": { "chat": "https://maas-coding-api.cn-huabei-1.xf-yun.com/v2/chat/completions", "models": "https://maas-coding-api.cn-huabei-1.xf-yun.com/v2/models", "model": "astron-code-latest", "auth": "Bearer", # try several documented models "alt_models": ["xopkimik26", "xopglm5", "xsparkx2flash"], }, } FAKES = ("xxxx", "your-", "example", "placeholder", "changeme", "sk-sp-123", "sk-tp-btf", "super-secret", "shared-test", "00000000000000000000000000000000000000") IFLYTEK_CTX = re.compile(r'xf-?yun|astron|maas-coding|maas-token|iflytek|spark-api', re.I) def http(method, url, key, auth="Bearer", body=None, timeout=20): headers = {"User-Agent": UA, "Accept": "*/*"} if auth == "Bearer": headers["Authorization"] = f"Bearer {key}" elif auth == "x-api-key": headers["x-api-key"] = key data = None if body is not None: data = json.dumps(body).encode() headers["Content-Type"] = "application/json" req = urllib.request.Request(url, data=data, headers=headers, method=method) try: with urllib.request.urlopen(req, timeout=timeout) as r: return r.getcode(), r.read().decode("utf-8", "replace") except urllib.error.HTTPError as e: try: return e.code, e.read().decode("utf-8", "replace") except Exception: return e.code, "" except Exception as e: return 0, f"network: {type(e).__name__}: {e}" def classify_chat(code, body): bl = (body or "").lower() if code == 200 and ('"choices"' in bl or '"message"' in bl): return "USABLE", f"chat 200 {body[:100]}" if code == 401: return "DEAD", f"401: {body[:140]}" if code in (402, 429): return "NO_BALANCE", f"chat={code}: {body[:140]}" if code == 403: return "NO_ACCESS", f"403: {body[:140]}" if code == 400: if any(w in bl for w in ("balance", "quota", "insufficient", "arrear", "suspend")): return "NO_BALANCE", f"400: {body[:140]}" # 400 with "model" error often means key works but model not granted if "model" in bl: return "NO_ACCESS", f"400 model: {body[:140]}" return "NO_ACCESS", f"400: {body[:140]}" if code == 404: return "NO_ACCESS", f"404: {body[:120]}" if code == 0: return "UNKNOWN", body[:160] if 500 <= code < 600: return "NO_ACCESS", f"5xx {code}: {body[:120]}" return "NO_ACCESS", f"HTTP {code}: {body[:140]}" def verify_openai(prov, key): cfg = PROVIDERS[prov] models = [cfg["model"]] + list(cfg.get("alt_models", [])) last = ("UNKNOWN", "no attempt") for model in models: if not model: continue code, body = http("POST", cfg["chat"], key, cfg["auth"], { "model": model, "messages": [{"role": "user", "content": "hi"}], "max_tokens": 5, "temperature": 0, }) v, d = classify_chat(code, body) if v == "USABLE": return "USABLE", f"[{model}] {d}" # 401 / 422 model-not-exist: try next model; 422 with model means model # name wrong but key may be fine — try alternates. if code == 401: return "DEAD", d last = (v, f"[{model}] {d}") if code == 422 and "model" in body.lower(): continue if v in ("NO_BALANCE", "NO_ACCESS"): # already a meaningful verdict, but try an alternate cheap model once continue return last def verify_ollama(key): # native ollama /api/chat code, body = http("POST", "https://api.ollama.com/api/chat", key, "Bearer", { "model": "llama3.2", "messages": [{"role": "user", "content": "hi"}], "stream": False, }, timeout=25) if code == 404: # try /api/generate code, body = http("POST", "https://api.ollama.com/api/generate", key, "Bearer", { "model": "llama3.2", "prompt": "hi", "stream": False, }, timeout=25) if code == 200: return "USABLE", f"ollama 200 {body[:100]}" if code == 401: return "DEAD", f"401: {body[:140]}" if code in (402, 429): return "NO_BALANCE", f"{code}: {body[:140]}" if code == 0: return "UNKNOWN", body[:160] return "NO_ACCESS", f"HTTP {code}: {body[:140]}" def verify(prov_key): prov, key = prov_key if prov == "OllamaCloud": return verify_ollama(key) return verify_openai(prov, key) def verify_cache_key(ck): """CachedVerifier needs a string key; wrap verify() to accept 'prov|key'.""" prov, key = ck.split("|", 1) return verify((prov, key)) # ── candidate loading ───────────────────────────────────────── def entropy_ok_hex(k): """Reject obvious placeholder 32/64-hex: all-same, sequential, mostly zeros.""" if len(k) not in (32, 40, 64): return False if len(set(k)) <= 4: return False if k.count("0") > len(k) * 0.7: return False if re.match(r'^0123456789abcdef+$', k): return False # ascending/descending runs if re.search(r'0123456789|9876543210|abcdef|fedcba', k): return False return True def load_candidates(providers): cands = {p: {} for p in providers} with open(POOL, errors="replace") as f: for ln in f: p = ln.rstrip("\n").split("|", 3) if len(p) < 3: continue tag, key, src = p[0], p[1], p[2] desc = p[3] if len(p) > 3 else "" if tag not in cands: continue if any(b in key.lower() for b in FAKES): continue if tag == "iFlytekCodingPlan": # only test hex in genuine xf-yun context if not (IFLYTEK_CTX.search(src) or IFLYTEK_CTX.search(desc)): continue if not re.fullmatch(r'[0-9a-f]{32}', key): continue if not entropy_ok_hex(key): continue if tag == "OllamaCloud": if not re.fullmatch(r'[0-9a-f]{32}\.[A-Za-z0-9]{20,30}', key): continue if tag == "SCNet": # sk-sp- keys in pool are actually Alibaba; only sk-tp- and plain if not (key.startswith("sk-tp-") or (key.startswith("sk-") and not key.startswith("sk-sp-"))): continue if key not in cands[tag]: cands[tag][key] = src return cands def main(): ap = argparse.ArgumentParser() ap.add_argument("--providers", default="all", help="comma list or all: " + ",".join(PROVIDERS)) ap.add_argument("--workers", type=int, default=12) ap.add_argument("--limit", type=int, default=0) ap.add_argument("--no-cache", action="store_true") args = ap.parse_args() provs = list(PROVIDERS) if args.providers == "all" else [ p.strip() for p in args.providers.split(",") if p.strip() in PROVIDERS] cands = load_candidates(provs) grand = {} for p in provs: items = list(cands[p].items()) if args.limit: items = items[:args.limit] for k, s in items: grand[(p, k)] = s print(f"{p:16s}: {len(items)} candidates") print(f"\nverifying {len(grand)} keys across {len(provs)} providers...\n") buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [], "UNKNOWN": [], "DEAD": []} with CachedVerifier("cn_coding", verify_cache_key, force=args.no_cache) as ver: with ThreadPoolExecutor(max_workers=args.workers) as pool: futs = {pool.submit(ver, f"{pk[0]}|{pk[1]}"): (pk, s) for pk, s in grand.items()} done = 0 for fut in as_completed(futs): (prov, key), src = futs[fut] done += 1 try: v, d = fut.result() except Exception as e: v, d = "UNKNOWN", f"exc:{e}" buckets[v].append((prov, key, src, d)) if done % 25 == 0: print(f" {done}/{len(grand)} usable={len(buckets['USABLE'])} " f"nobal={len(buckets['NO_BALANCE'])} " f"noacc={len(buckets['NO_ACCESS'])} dead={len(buckets['DEAD'])} " f"hit={ver.hits} live={ver.live}", flush=True) print(f"cache: {ver.hits} hits / {ver.live} live\n") name_map = {"USABLE": "usable", "NO_BALANCE": "no_balance", "NO_ACCESS": "no_access", "UNKNOWN": "unknown", "DEAD": "dead"} for label, fn in name_map.items(): with (OUT / f"{fn}.txt").open("w") as f: for prov, key, src, d in buckets[label]: f.write(f"{prov}|{key}|{src}|{d}\n") for label in ("USABLE", "NO_BALANCE", "NO_ACCESS", "UNKNOWN", "DEAD"): print(f" {label:11s}: {len(buckets[label])}") if buckets["USABLE"]: print("\n=== USABLE ===") for prov, key, src, d in buckets["USABLE"]: print(f" [{prov}] {key}\n {src}\n {d[:120]}") if __name__ == "__main__": main()