#!/usr/bin/env python3 """Kiro client -- use a captured Kiro refresh token to chat with Claude 4.5. Kiro (AWS's AI IDE) uses OAuth refresh tokens (aorAAAAA...) instead of API keys. This client: 1. Exchanges the refresh token for a 1-hour access token (auto-caches/refreshes). 2. Sends chat messages to Claude Sonnet/Haiku 4.5 via the CodeWhisperer endpoint. 3. Parses the binary AWS event-stream response and prints the assistant reply. Usage: export KIRO_REFRESH="aorAAAAA....." # or pass --refresh python3 kiro_client.py # interactive REPL python3 kiro_client.py -m sonnet "write a snake game in python" python3 kiro_client.py --profile arn:aws:... # override/supply profileArn echo "explain quantum tunneling" | python3 kiro_client.py --stdin No dependencies beyond the Python stdlib. Endpoint is directly reachable (no proxy). """ import argparse, json, os, struct, sys, time, uuid, urllib.request, urllib.error REFRESH_URL = "https://prod.us-east-1.auth.desktop.kiro.dev/refreshToken" CHAT_URL = "https://q.us-east-1.amazonaws.com/generateAssistantResponse" MODELS = ("claude-sonnet-4.5", "claude-haiku-4.5", "claude-sonnet-4-20250514", "claude-haiku-4-5-20251001") # ---------- auth ---------------------------------------------------------- def refresh(refresh_token, machine_id=None): """Exchange a refresh token for an access token. Returns (access, profileArn, expiresIn).""" mid = machine_id or uuid.uuid4().hex[:16] body = json.dumps({"refreshToken": refresh_token}).encode() req = urllib.request.Request( REFRESH_URL, data=body, method="POST", headers={ "Content-Type": "application/json", "User-Agent": f"KiroIDE-1.6.0-{mid}", "Accept": "*/*", }) try: with urllib.request.urlopen(req, timeout=20) as r: data = json.loads(r.read().decode()) except urllib.error.HTTPError as e: msg = e.read().decode("utf-8", "replace")[:300] raise SystemExit(f"[!] refresh failed: HTTP {e.code}: {msg}") access = data.get("accessToken") or data.get("access_token") if not access: raise SystemExit(f"[!] no accessToken in response: {data}") return access, data.get("profileArn", ""), int(data.get("expiresIn", 3600)) class TokenManager: """Caches the access token and refreshes it ~5 min before expiry.""" def __init__(self, refresh_token, profile_arn=None, machine_id=None): self.rt = refresh_token self.profile = profile_arn self.mid = machine_id self.access = None self.expires_at = 0 def get(self): if time.time() >= self.expires_at - 300 or not self.access: self.access, p, exp = refresh(self.rt, self.mid) self.profile = self.profile or p self.expires_at = time.time() + exp sys.stderr.write(f"[+] refreshed access token " f"(expires in {exp}s, profile={self.profile or 'n/a'})\n") return self.access # ---------- event-stream parsing ------------------------------------------ # AWS event-stream framing: # total_len(4) | headers_len(4) | prelude_crc(4) | headers | payload | msg_crc(4) # Headers are: name_len(1) | name | value_type(1) | value... # We just care about the ":event-type" header and the JSON payload. def _u8(buf, o): return buf[o], o+1 def _u16(buf,o): return struct.unpack(">H", buf[o:o+2])[0], o+2 def _u32(buf,o): return struct.unpack(">I", buf[o:o+4])[0], o+4 def _u32b(b): return struct.unpack(">I", b)[0] def parse_events(raw): """Yield (event_type, payload_dict) for each event in the stream.""" o = 0 n = len(raw) while o + 12 <= n: total, o = _u32(raw, o) _hlen, o = _u32(raw, o) o += 4 # prelude crc if total < 16 or o + total - 12 > n: break end = o + total - 16 # subtract prelude(12) + trailing crc(4) # parse headers event_type = None ho = o header_end = None # we need header length; recompute: total - 12 - payload... but we # didn't store it. Headers run until the payload starts; just scan # for the known ":event-type" header by walking the header block. # Simpler: find header section boundary by re-reading prelude. # (We already advanced o past prelude; _hlen was at o-8.) hlen = _hlen header_end = o + hlen while ho < header_end: nlen, ho = _u8(raw, ho) name = raw[ho:ho+nlen].decode("ascii", "replace"); ho += nlen vtype, ho = _u8(raw, ho) if vtype == 0: # bool true val = True elif vtype == 1: # bool false val = False elif vtype in (2, 3): # byte / short sz = 1 if vtype == 2 else 2 val = raw[ho:ho+sz]; ho += sz elif vtype == 4: # int val, ho = _u32(raw, ho) elif vtype == 5: # long val = struct.unpack(">Q", raw[ho:ho+8])[0]; ho += 8 elif vtype == 6: # byte array / bytes blen, ho = _u16(raw, ho); val = raw[ho:ho+blen]; ho += blen elif vtype == 7: # string slen, ho = _u16(raw, ho); val = raw[ho:ho+slen].decode(); ho += slen elif vtype == 8: # timestamp val = struct.unpack(">q", raw[ho:ho+8])[0]; ho += 8 elif vtype == 9: # uuid val = raw[ho:ho+16].hex(); ho += 16 else: break if name == ":event-type" and isinstance(val, str): event_type = val payload = raw[header_end:end] if event_type and payload: try: yield event_type, json.loads(payload.decode("utf-8", "replace")) except Exception: yield event_type, {"_raw": payload[:200]} o = end + 4 # skip trailing crc # ---------- chat ---------------------------------------------------------- def chat(tm, message, model="claude-haiku-4.5", history=None, timeout=120): """Send one message; returns the assistant text. Updates history in place.""" history = history if history is not None else [] cid = str(uuid.uuid4()) body = { "profileArn": tm.profile, "conversationState": { "agentContinuationId": str(uuid.uuid4()), "agentTaskType": "vibe", "chatTriggerType": "MANUAL", "conversationId": cid, "currentMessage": { "userInputMessage": { "content": message, "modelId": model, "origin": "AI_EDITOR", } }, "history": history, }, } req = urllib.request.Request( CHAT_URL, data=json.dumps(body).encode(), method="POST", headers={ "Authorization": f"Bearer {tm.get()}", "Content-Type": "application/json", "Host": "q.us-east-1.amazonaws.com", "User-Agent": f"aws-sdk-js/1.0.27 KiroIDE-1.6.0-{uuid.uuid4().hex[:16]}", "x-amzn-codewhisperer-optout": "true", "x-amzn-kiro-agent-mode": "vibe", "amz-sdk-invocation-id": str(uuid.uuid4()), "amz-sdk-request": "attempt=1; max=1", }) with urllib.request.urlopen(req, timeout=timeout) as r: raw = r.read() chunks = [] for etype, payload in parse_events(raw): if etype == "assistantResponseEvent": c = payload.get("content") if c: chunks.append(c) elif etype == "exception" or "message" in payload and etype != "metadataEvent": sys.stderr.write(f"[event {etype}] {payload}\n") return "".join(chunks) # ---------- CLI ----------------------------------------------------------- def main(): ap = argparse.ArgumentParser(description="Use a captured Kiro account to chat with Claude 4.5") ap.add_argument("prompt", nargs="*", help="One-shot prompt. Omit for interactive REPL.") ap.add_argument("-m", "--model", default="claude-sonnet-4.5", choices=MODELS, help="Model to use (default: claude-sonnet-4.5)") ap.add_argument("--refresh", default=os.environ.get("KIRO_REFRESH"), help="Kiro refresh token (aorAAAAA...) or set KIRO_REFRESH") ap.add_argument("--profile", default=os.environ.get("KIRO_PROFILE"), help="profileArn (auto-detected from refresh if omitted)") ap.add_argument("--machine-id", default=None, help="Stable machine id segment (random by default)") ap.add_argument("--stdin", action="store_true", help="Read prompt from stdin") ap.add_argument("--list-models", action="store_true") args = ap.parse_args() if args.list_models: for m in MODELS: print(m) return if not args.refresh: sys.exit("[!] provide --refresh or set KIRO_REFRESH env var") tm = TokenManager(args.refresh, args.profile, args.machine_id) tm.get() # prime the token + profileArn prompt = " ".join(args.prompt) if args.stdin: prompt = (prompt + "\n" if prompt else "") + sys.stdin.read() if prompt.strip(): print(chat(tm, prompt, model=args.model)) return # interactive REPL sys.stderr.write(f"[+] model={args.model} (type /exit, /model , /clear)\n") history = [] while True: try: line = input("kiro> ").strip() except (EOFError, KeyboardInterrupt): print() break if not line: continue if line in ("/exit", "/quit"): break if line == "/clear": history.clear() sys.stderr.write("[+] history cleared\n") continue if line.startswith("/model"): parts = line.split(maxsplit=1) if len(parts) == 2 and parts[1] in MODELS: args.model = parts[1] sys.stderr.write(f"[+] model={args.model}\n") else: sys.stderr.write(f"[+] models: {', '.join(MODELS)}\n") continue reply = chat(tm, line, model=args.model, history=history) print(reply) history.append({"type": "user", "content": line}) history.append({"type": "assistant", "content": reply}) if __name__ == "__main__": main()