Files
hack/tools/scripts/vault_alldeep_usable.py
T

256 lines
8.3 KiB
Python

#!/usr/bin/env python3
"""Vault the 26 new USABLE keys from results/all_deep/usable.txt and create
NewAPI channels for each. Dedups against existing vault keys.json and against
keys already configured in NewAPI.
Each usable.txt line (TAB-separated, but stored without visible tabs because
the harvest writer concatenates fields): we parse by walking known key
prefixes. Fields are: provider, key, source, model-detail.
"""
import json, re, sys, time
from pathlib import Path
sys.path.insert(0, str(Path(__file__).resolve().parent))
from newapi_client import (
add_channel, get_channel, list_channels, update_channel, test_channel,
TYPE_OPENAI, TYPE_ZHIPU_V4, TYPE_VOLC_ENGINE,
)
HERE = Path(__file__).resolve().parent
H = HERE / "llm-key-hunter"
USABLE = H / "results" / "all_deep" / "usable.txt"
VAULT = H / "usable_keys"
# provider label in usable.txt -> (vault dir, newapi channel spec)
# newapi spec: name_prefix, type, base_url, models(str)
PROVIDERS = {
"minimax": dict(
dir="minimax", name="MiniMax", type=TYPE_OPENAI,
base_url="https://api.minimaxi.com",
models=("MiniMax-M2.5,MiniMax-M2.7,MiniMax-M3,MiniMax-M2.5-highspeed,"
"abab6.5s-chat,abab6.5-chat"),
),
"siliconflow": dict(
dir="siliconflow", name="SiliconFlow", type=TYPE_OPENAI,
base_url="https://api.siliconflow.cn",
models="Qwen/Qwen2.5-7B-Instruct,deepseek-ai/DeepSeek-V3",
),
"zhipu_coding": dict(
dir="zhipu", name="ZhipuCoding", type=TYPE_ZHIPU_V4,
base_url="https://open.bigmodel.cn/api/coding/paas/v4",
models=("glm-4.5-flash,glm-4.6,glm-4.7,glm-4.7-flash,glm-5.2,"
"glm-5,coding-plan-chat"),
),
"zhipu_paas": dict(
dir="zhipu", name="ZhipuPaaS", type=TYPE_ZHIPU_V4,
base_url="https://open.bigmodel.cn/api/paas/v4",
models="glm-4-flash,glm-4.5-flash,glm-4-air,glm-4.6",
),
"volcano": dict(
dir="volcanoark", name="Volcano", type=TYPE_VOLC_ENGINE,
base_url="https://ark.cn-beijing.volces.com/api/v3",
models=("doubao-seed-1-6-flash-250615,doubao-seed-1-6-250615,"
"doubao-1-5-lite-32k-250115,doubao-1-5-pro-32k-250115"),
),
"longcat": dict(
dir="longcat", name="LongCat", type=TYPE_OPENAI,
base_url="https://api.longcat.chat/openai/v1",
models="LongCat-2.0,LongCat-2.0-Chat",
),
}
# regex to detect where the key starts in each line (provider then key)
KEY_PREFIXES = ["sk-cp-", "sk-", "ak_", "gsk_", "sk-or-v1-", "AIza",
"sk-ant", "ak-"]
UUID_RE = re.compile(r"[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}")
ZHIPU_RE = re.compile(r"[0-9a-f]{32}\.[A-Za-z0-9]{16}")
def parse_line(line):
"""Return (provider, key, source, detail) or None."""
line = line.strip()
if not line:
return None
# provider is one of the known labels, matched at the start (no separator
# between provider and key in the usable.txt format).
prov = None
for name in sorted(PROVIDERS, key=len, reverse=True):
if line.startswith(name):
prov = name
break
if not prov:
return None
rest = line[len(prov):].lstrip("\t ")
# find key by shape
key = None
if rest.startswith("sk-cp-"):
km = re.match(r"(sk-cp-[A-Za-z0-9_\-]{40,140})", rest)
key = km.group(1) if km else None
elif rest.startswith("ak_"):
km = re.match(r"(ak_[A-Za-z0-9]{16,40})", rest)
key = km.group(1) if km else None
elif rest.startswith("sk-"):
km = re.match(r"(sk-[A-Za-z0-9]{32,90})", rest)
key = km.group(1) if km else None
else:
# zhipu 32hex.16alnum
zm = ZHIPU_RE.match(rest)
if zm:
key = zm.group(0)
else:
um = UUID_RE.match(rest)
if um:
key = um.group(0)
if not key:
return None
after = rest[len(key):]
# source is a github URL
sm = re.search(r"(https://\S+?)(?:\s+(?:MiniMax|glm|doubao|LongCat|Qwen|deepseek)\b.*)?$", after)
source = sm.group(1) if sm else after.strip()
detail = after.replace(source, "").strip() if source else ""
return prov, key, source, detail
def load_vault_keys(vault_dir):
p = vault_dir / "keys.json"
if p.exists():
try:
return json.loads(p.read_text())
except Exception:
return []
return []
def vault_key(cfg, prov, key, source, detail):
vdir = VAULT / cfg["dir"]
vdir.mkdir(parents=True, exist_ok=True)
entries = load_vault_keys(vdir)
if any(e.get("key") == key for e in entries):
return False # already vaulted
entry = {
"key": key,
"source": source,
"detail": f"all_deep: {detail}" if detail else "all_deep usable",
"base_url": cfg["base_url"] + ("" if cfg["base_url"].endswith(("v3", "v4", "v1")) else "/v1"),
"models": cfg["models"].split(","),
"auth": "Bearer",
}
entries.append(entry)
(vdir / "keys.json").write_text(json.dumps(entries, indent=2, ensure_ascii=False))
# append key file
suffix = key[-6:].replace("/", "_")
kf = vdir / f"key_{len(entries):02d}_{suffix}.txt"
kf.write_text(key + "\n")
# append keys.env
envf = vdir / "keys.env"
with envf.open("a") as f:
f.write(f"{cfg['name'].upper()}_API_KEY={key}\n")
return True
def all_newapi_keys():
keys = set()
for ch in list_channels():
try:
d = get_channel(ch["id"])
if d and d.get("key"):
keys.add(d["key"])
except Exception:
pass
return keys
def make_channel(cfg, key, existing):
if key in existing:
return "dup", None
name = f"{cfg['name']}_{key[-6:]}"
channel = {
"name": name,
"type": cfg["type"],
"key": key,
"base_url": cfg["base_url"],
"models": cfg["models"],
"groups": ["default"],
"model_mapping": "",
"priority": 0,
"weight": 0,
"auto_ban": 1,
}
ok, msg = add_channel(channel)
if not ok:
return f"FAIL:{msg}", None
# find new id by scanning high ids (list_channels caps at 100, new chans land high)
new_id = None
for ch in list_channels():
if ch.get("name") == name:
new_id = ch["id"]
break
if new_id is None:
# scan high IDs
for cid in range(530, 700):
d = get_channel(cid)
if d and d.get("name") == name:
new_id = cid
break
if new_id is None:
return "added-no-id", None
t_ok, t_msg, _ = test_channel(new_id)
if t_ok:
existing.add(key)
return f"OK id={new_id} {t_msg[:60]}", new_id
update_channel(new_id, status=2)
return f"BAD id={new_id} disabled: {t_msg[:60]}", new_id
def main():
if not USABLE.exists():
print("no usable.txt"); return
rows = []
for line in USABLE.read_text().splitlines():
p = parse_line(line)
if p:
rows.append(p)
print(f"parsed {len(rows)} usable keys")
counts = {}
for prov, *_ in rows:
counts[prov] = counts.get(prov, 0) + 1
print("by provider:", counts)
# 1) vault
print("\n=== vaulting ===")
vaulted = 0
for prov, key, source, detail in rows:
cfg = PROVIDERS.get(prov)
if not cfg:
print(f" [{prov}] no vault cfg, skip {key[:12]}")
continue
if vault_key(cfg, prov, key, source, detail):
vaulted += 1
print(f" [{prov:14}] vaulted {key[:18]}...")
else:
print(f" [{prov:14}] already in vault {key[:18]}...")
print(f"vaulted {vaulted} new")
# 2) NewAPI channels
print("\n=== NewAPI channels ===")
existing = all_newapi_keys()
print(f" {len(existing)} keys already in NewAPI")
results = []
for prov, key, source, detail in rows:
cfg = PROVIDERS.get(prov)
if not cfg:
continue
status, cid = make_channel(cfg, key, existing)
results.append((prov, key, status, cid))
print(f" [{prov:14}] {key[:18]}... -> {status}")
time.sleep(0.4)
ok = [r for r in results if r[2].startswith("OK")]
print(f"\n=== DONE: {len(ok)}/{len(results)} channels live ===")
for prov, key, status, cid in ok:
print(f" {prov:14} {key[:20]}... {status}")
if __name__ == "__main__":
main()