119 lines
4.7 KiB
Python
119 lines
4.7 KiB
Python
#!/usr/bin/env python3
|
|
"""Re-verify iFlytek MaaS + SenseNova candidate keys with corrected model lists.
|
|
|
|
Pulls candidate keys (provider<TAB>key<TAB>url) from the all_deep dead/unknown
|
|
buckets AND directly from the harvested candidate pool, then re-verifies using
|
|
the updated model lists. Only 401 / invalid-key messages => DEAD; 403/404 on a
|
|
specific model => NO_ACCESS (key is real but model not granted); a 200 with
|
|
choices on any model => USABLE.
|
|
"""
|
|
import json, re, sys, time, urllib.request, urllib.error
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from verify_cache import CachedVerifier
|
|
import providers_all_deep as P
|
|
import hunt_all_deep as H
|
|
|
|
OUT = Path("results/cn_batch2_reverify")
|
|
OUT.mkdir(parents=True, exist_ok=True)
|
|
|
|
# Provider keys we care about re-verifying
|
|
WANT = {"iflytek_maas", "sensenova", "tencent_lkeap", "360zhinao",
|
|
"tencent_hunyuan", "dmxapi", "closeai", "api2d", "aihubmix"}
|
|
|
|
|
|
def load_candidate_keys():
|
|
"""Pull every key from dead/unknown buckets + candidates.tsv for WANT provers."""
|
|
keys = {} # (prov, key) -> source
|
|
root = Path("results/all_deep")
|
|
# from bucket files (pipe or tab separated)
|
|
for fn in ("dead.txt", "unknown.txt", "no_access.txt", "no_balance.txt"):
|
|
p = root / fn
|
|
if not p.exists():
|
|
continue
|
|
for ln in p.read_text(errors="replace").splitlines():
|
|
# format: prov<TAB>key<TAB>url<TAB>detail (no tab between prov and key
|
|
# because hunt_all_deep writes f"{prov}\t{k}\t{s}\t{d}")
|
|
parts = ln.split("\t")
|
|
if len(parts) < 3:
|
|
continue
|
|
prov, key, url = parts[0], parts[1], parts[2]
|
|
# bug: some earlier writes had prov+key concatenated (no tab). Detect
|
|
# by checking if prov doesn't contain 'sk-' and key starts with sk-
|
|
if not key.startswith("sk-") and prov.startswith(("iflytek", "sensenova",
|
|
"tencent", "360zhinao",
|
|
"dmxapi", "closeai",
|
|
"api2d", "aihubmix")):
|
|
# split prov name from key
|
|
m = re.match(r"^([a-z0-9_]+?)(sk-.+)$", prov)
|
|
if m:
|
|
prov, key = m.group(1), m.group(2)
|
|
if prov in WANT and key.startswith(("sk-", "ak_")):
|
|
keys.setdefault((prov, key), url)
|
|
return keys
|
|
|
|
|
|
def main():
|
|
specs = P.build(H.openai_verify)
|
|
for s in specs.values():
|
|
s["patterns"] = [re.compile(p) for p in s["patterns"]]
|
|
|
|
keys = load_candidate_keys()
|
|
by_prov = {}
|
|
for (prov, k), src in keys.items():
|
|
if prov not in specs:
|
|
continue
|
|
by_prov.setdefault(prov, {})[k] = src
|
|
|
|
print("candidate keys to re-verify:")
|
|
for p, d in by_prov.items():
|
|
print(f" {p:18s}: {len(d)}")
|
|
total = sum(len(d) for d in by_prov.values())
|
|
print(f" total: {total}\n", flush=True)
|
|
|
|
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [],
|
|
"UNKNOWN": [], "DEAD": []}
|
|
|
|
for prov, d in by_prov.items():
|
|
if not d:
|
|
continue
|
|
print(f"=== re-verifying {prov}: {len(d)} keys ===", flush=True)
|
|
verify = specs[prov]["verify"]
|
|
# force fresh verification (new model lists / classifier)
|
|
with CachedVerifier(f"reverify_{prov}", verify, force=True) as ver:
|
|
with ThreadPoolExecutor(max_workers=24) as pool:
|
|
futs = {pool.submit(ver, k): (k, s) for k, s in d.items()}
|
|
done = 0
|
|
for f in as_completed(futs):
|
|
k, s = futs[f]; done += 1
|
|
try:
|
|
v, detail = f.result()
|
|
except Exception as e:
|
|
v, detail = "UNKNOWN", f"exc:{e}"
|
|
buckets[v].append((prov, k, s, detail))
|
|
if done % 25 == 0:
|
|
print(f" {prov} {done}/{len(d)} "
|
|
f"usable={len(buckets['USABLE'])} "
|
|
f"dead={len(buckets['DEAD'])}", flush=True)
|
|
|
|
for label, rows in buckets.items():
|
|
with (OUT / f"{label.lower()}.txt").open("w") as f:
|
|
for prov, k, s, d in rows:
|
|
f.write(f"{prov}\t{k}\t{s}\t{d}\n")
|
|
|
|
print("\n=== RE-VERIFY RESULTS ===")
|
|
for label in buckets:
|
|
print(f" {label:11s}: {len(buckets[label])}")
|
|
if buckets["USABLE"]:
|
|
print("\n=== USABLE KEYS ===")
|
|
for prov, k, s, d in buckets["USABLE"]:
|
|
print(f" [{prov}] {k}")
|
|
print(f" src: {s}")
|
|
print(f" {d[:140]}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|