#!/usr/bin/env python3 """Chinese (mainland) HTTP/HTTPS proxy hunter. Pulls free proxy lists from GitHub, filters to China-IP ranges (APNIC delegation data), then validates each candidate by proxying a request to a domestic target. Classifies anonymity: elite - no Via/X-Forwarded-For leaked anonymous - real IP hidden but proxy headers present transparent - real IP forwarded Outputs in results/proxies/. """ import bisect import ipaddress import json import re import socket import sys import time import urllib.request import urllib.error from concurrent.futures import ThreadPoolExecutor, as_completed from pathlib import Path HERE = Path(__file__).parent OUT = HERE / "results" / "proxies" OUT.mkdir(parents=True, exist_ok=True) UA = "Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36" PROXY_SOURCES = [ "https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt", "https://raw.githubusercontent.com/TheSpeedX/SOCKS-List/master/socks5.txt", "https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt", "https://raw.githubusercontent.com/monosans/proxy-list/main/proxies_anonymous/http.txt", "https://raw.githubusercontent.com/hookzof/socks5_list/master/proxy.txt", "https://raw.githubusercontent.com/clarketm/proxy-list/master/proxy-list-raw.txt", "https://raw.githubusercontent.com/ShiftyTR/Proxy-List/master/http.txt", "https://raw.githubusercontent.com/ShiftyTR/Proxy-List/master/https.txt", "https://raw.githubusercontent.com/roosterkid/openproxylist/main/HTTPS_RAW.txt", "https://raw.githubusercontent.com/mmpx12/proxy-list/master/http.txt", "https://raw.githubusercontent.com/mmpx12/proxy-list/master/https.txt", "https://raw.githubusercontent.com/proxifly/free-proxy-list/main/proxies/protocols/http/data.txt", "https://raw.githubusercontent.com/proxifly/free-proxy-list/main/proxies/countries/CN/data.txt", "https://raw.githubusercontent.com/zloi-user/hideip.me/main/http.txt", "https://raw.githubusercontent.com/zloi-user/hideip.me/main/https.txt", "https://raw.githubusercontent.com/ErcinDedeoglu/proxies/main/proxies/http.txt", "https://raw.githubusercontent.com/ErcinDedeoglu/proxies/main/proxies/https.txt", "https://raw.githubusercontent.com/MuRongPIG/Proxy-Master/main/http.txt", "https://raw.githubusercontent.com/Zaeem20/FREE_PROXIES_LIST/master/http.txt", "https://raw.githubusercontent.com/Zaeem20/FREE_PROXIES_LIST/master/https.txt", "https://raw.githubusercontent.com/sunny9577/proxy-scraper/master/generated/http_proxies.txt", "https://raw.githubusercontent.com/officialputuid/KangProxy/KangProxy/http/http.txt", "https://raw.githubusercontent.com/officialputuid/KangProxy/KangProxy/https/https.txt", "https://raw.githubusercontent.com/yemixzy/proxy-list/main/proxies/http.txt", "https://raw.githubusercontent.com/vakhov/fresh-proxy-list/master/http.txt", "https://raw.githubusercontent.com/vakhov/fresh-proxy-list/master/https.txt", "https://raw.githubusercontent.com/proxy4parsing/proxy-list/main/http.txt", "https://raw.githubusercontent.com/Anonym0usWork1221/Free-Proxies/main/proxy_files/http_proxies.txt", ] # Validation targets — domestic Chinese endpoints (to confirm CN egress) TEST_TARGETS = [ "http://www.baidu.com/", "http://www.qq.com/", "http://httpbin.org/ip", # for anonymity header check (not CN, but reveals headers) ] IPPORT_RE = re.compile(rb"\b(\d{1,3}\.\d{1,3}\.\d{1,3}\.\d{1,3}):(\d{2,5})\b") def fetch(url, timeout=20): req = urllib.request.Request(url, headers={"User-Agent": UA}) try: with urllib.request.urlopen(req, timeout=timeout) as resp: return resp.read() except Exception as e: return b"" def load_china_cidrs(): """Return list of ipaddress networks for mainland China.""" cache = OUT / "cn_cidrs.json" if cache.exists() and (time.time() - cache.stat().st_mtime) < 86400: nets = [] for cidr in json.loads(cache.read_text()): try: nets.append(ipaddress.ip_network(cidr)) except ValueError: pass return nets nets = [] # APNIC delegated stats data = fetch("https://ftp.apnic.net/apnic/stats/apnic/delegated-apnic-latest") if not data: # GitHub mirror fallback data = fetch("https://raw.githubusercontent.com/itgoyo/China-IP-list/refs/heads/master/cn.txt") for line in data.decode(errors="replace").splitlines(): if not line.startswith("apnic|CN|ipv4|"): continue parts = line.split("|") if len(parts) >= 5: start, count = parts[3], int(parts[4]) try: net = ipaddress.ip_network((start, 32 - (count - 1).bit_length() + 1), strict=False) nets.append(net) except ValueError: pass # Also merge GitHub cn_ip lists extra_urls = [ "https://raw.githubusercontent.com/17mon/china_ip_list/master/china_ip_list.txt", "https://raw.githubusercontent.com/gaoyifan/china-operator-ip/ip-lists/china.txt", ] for u in extra_urls: d = fetch(u) for line in d.decode(errors="replace").splitlines(): line = line.strip() if "/" in line: try: nets.append(ipaddress.ip_network(line)) except ValueError: pass # dedup + merge collapsed = list(ipaddress.collapse_addresses(nets)) cache.write_text(json.dumps([str(n) for n in collapsed])) print(f"Loaded {len(collapsed)} China CIDRs (cached 24h)") return collapsed def is_china_ip(ip, boundaries): """Check membership using prebuilt (start_int, end_int) sorted boundaries.""" try: addr = int(ipaddress.ip_address(ip)) except ValueError: return False starts = boundaries[0] idx = bisect.bisect_right(starts, addr) - 1 if idx < 0: return False return addr <= boundaries[1][idx] def build_boundaries(cidrs): collapsed = sorted(ipaddress.collapse_addresses(cidrs)) starts = [int(n.network_address) for n in collapsed] ends = [int(n.broadcast_address) for n in collapsed] return (starts, ends) def scrape_proxies(): """Return set of ip:port candidates from all sources.""" all_proxies = set() for url in PROXY_SOURCES: data = fetch(url) if not data: continue for m in IPPORT_RE.finditer(data): ip, port = m.group(1).decode(), int(m.group(2)) if 1 <= port <= 65535: all_proxies.add(f"{ip}:{port}") print(f" {url.split('/')[-1]:30s} -> total {len(all_proxies)}") return all_proxies def test_proxy(proxy, timeout=8): """Try proxy against baidu. Returns (status, latency_ms, anon_level, detail).""" proxy_url = f"http://{proxy}" handler = urllib.request.ProxyHandler({"http": proxy_url, "https": proxy_url}) opener = urllib.request.build_opener(handler) # 1) reachability via baidu t0 = time.time() try: req = urllib.request.Request("http://www.baidu.com/", headers={"User-Agent": UA}) with opener.open(req, timeout=timeout) as resp: body = resp.read(2048) latency = int((time.time() - t0) * 1000) if resp.getcode() != 200 or b"baidu" not in body.lower(): return "fail", 0, "", f"baidu HTTP {resp.getcode()}" except urllib.error.HTTPError as e: return "fail", 0, "", f"HTTP {e.code}" except Exception as e: return "fail", 0, "", f"{type(e).__name__}: {e}" # 2) anonymity check via httpbin/ip or similar — use ip-api or a header echo anon = "unknown" try: req = urllib.request.Request( "http://httpbin.org/get", headers={"User-Agent": UA} ) with opener.open(req, timeout=timeout) as resp: data = json.loads(resp.read(4096)) hdrs = {k.lower(): v for k, v in data.get("headers", {}).items()} via = hdrs.get("via", "") xff = hdrs.get("x-forwarded-for", "") real_ip = hdrs.get("x-real-ip", "") if not via and not xff and not real_ip: anon = "elite" elif xff and proxy.split(":")[0] not in xff: anon = "anonymous" else: anon = "transparent" except Exception: anon = "elite?" # baidu worked, anonymity echo failed; assume likely elite return "ok", latency, anon, "" def main(): import argparse ap = argparse.ArgumentParser() ap.add_argument("--workers", type=int, default=80) ap.add_argument("--timeout", type=int, default=8) ap.add_argument("--limit", type=int, default=0) args = ap.parse_args() print("=== Stage 1: load China CIDRs ===") cidrs = load_china_cidrs() print("\n=== Stage 2: scrape proxy lists from GitHub ===") candidates = scrape_proxies() print(f" total scraped: {len(candidates)}") print("\n=== Stage 3: filter to China IPs ===") boundaries = build_boundaries(cidrs) cn = [p for p in candidates if is_china_ip(p.split(":")[0], boundaries)] print(f" China-IP candidates: {len(cn)}") (OUT / "all_cn.txt").write_text("\n".join(sorted(cn)) + "\n") if args.limit: cn = cn[:args.limit] print(f" (limited to first {args.limit})") print(f"\n=== Stage 4: validate {len(cn)} candidates (workers={args.workers}) ===") ok, fail = [], [] processed = 0 start = time.time() with ThreadPoolExecutor(max_workers=args.workers) as pool: futs = {pool.submit(test_proxy, p, args.timeout): p for p in cn} for fut in as_completed(futs): p = futs[fut] processed += 1 try: status, latency, anon, detail = fut.result() except Exception as e: status, latency, anon, detail = "fail", 0, "", str(e) if status == "ok": ok.append((p, latency, anon)) else: fail.append((p, detail)) if processed % 200 == 0: el = time.time() - start print(f" [{processed}/{len(cn)}] ok={len(ok)} fail={len(fail)} " f"({processed/el:.0f}/s)") elapsed = time.time() - start print(f"\nDone in {elapsed:.1f}s: {len(ok)} working, {len(fail)} failed") # Write outputs by_anon = {"elite": [], "anonymous": [], "transparent": [], "elite?": [], "unknown": []} for p, lat, anon in sorted(ok, key=lambda x: x[1]): by_anon.setdefault(anon, []).append((p, lat)) with open(OUT / "working_cn.txt", "w") as f: for p, lat, anon in sorted(ok, key=lambda x: x[1]): f.write(f"{p}\t{lat}ms\t{anon}\n") for anon, items in by_anon.items(): if items: safe = anon.replace("?", "_maybe") with open(OUT / f"{safe}.txt", "w") as f: for p, lat in items: f.write(f"{p}\t{lat}ms\n") with open(OUT / "failures.txt", "w") as f: for p, d in fail: f.write(f"{p}\t{d}\n") print("\n=== Working CN proxies ===") for p, lat, anon in sorted(ok, key=lambda x: x[1])[:50]: print(f" {p:24s} {lat:5d}ms {anon}") print(f"\nFiles written under {OUT}/") if __name__ == "__main__": main()