#!/usr/bin/env python # -*- coding: utf-8 -*- """公开免费代理源(国外可访问)+ 加权分流采集。 设计: - 每个源有 weight;采集时按权重随机/轮询分流 - 源连续拉不到可用代理 / 被 x.ai 相关链路废掉 → 降权或禁用 - 采集结果只返回候选 URL,是否入库由 proxy_manager 测 accounts.x.ai 决定 """ from __future__ import annotations import re import random import time from dataclasses import dataclass, field from typing import Callable from urllib.parse import urlparse try: from curl_cffi import requests as _req except Exception: # pragma: no cover import requests as _req # type: ignore LogFn = Callable[[str], None] def _noop(msg: str) -> None: return None @dataclass class ProxySource: id: str name: str url: str weight: int = 10 protocol: str = "http" # 结果默认 scheme kind: str = "txt" # txt | json_geonode | json_list enabled: bool = True # 运行时统计(内存;持久化在 ProxyStore.sources) fail_streak: int = 0 last_ok: float = 0.0 last_fail: float = 0.0 last_reason: str = "" harvested_total: int = 0 kept_total: int = 0 # 公开源(无需 key)。权重仅作初始值,运行中会按成功率动态调。 DEFAULT_SOURCES: list[ProxySource] = [ ProxySource( id="proxyscrape_http", name="ProxyScrape HTTP API", url="https://api.proxyscrape.com/v2/?request=displayproxies&protocol=http&timeout=8000&country=all&ssl=all&anonymity=all", weight=30, protocol="http", kind="txt", ), ProxySource( id="proxyscrape_socks5", name="ProxyScrape SOCKS5 API", url="https://api.proxyscrape.com/v2/?request=displayproxies&protocol=socks5&timeout=8000&country=all", weight=15, protocol="socks5", kind="txt", ), ProxySource( id="databay_http", name="Databay free-proxy-list HTTP", url="https://cdn.jsdelivr.net/gh/databay-labs/free-proxy-list@master/http.txt", weight=25, protocol="http", kind="txt", ), ProxySource( id="databay_socks5", name="Databay free-proxy-list SOCKS5", url="https://cdn.jsdelivr.net/gh/databay-labs/free-proxy-list@master/socks5.txt", weight=12, protocol="socks5", kind="txt", ), ProxySource( id="monosans_http", name="monosans/proxy-list HTTP", url="https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt", weight=18, protocol="http", kind="txt", ), ProxySource( id="monosans_socks5", name="monosans/proxy-list SOCKS5", url="https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/socks5.txt", weight=10, protocol="socks5", kind="txt", ), ProxySource( id="speedx_http", name="TheSpeedX PROXY-List HTTP", url="https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt", weight=12, protocol="http", kind="txt", ), ProxySource( id="openproxylist_http", name="openproxylist HTTP", url="https://api.openproxylist.xyz/http.txt", weight=10, protocol="http", kind="txt", ), ProxySource( id="geonode_http", name="Geonode free proxy API", url="https://proxylist.geonode.com/api/proxy-list?limit=200&page=1&sort_by=lastChecked&sort_type=desc&protocols=http%2Chttps", weight=8, protocol="http", kind="json_geonode", ), ] _HOSTPORT_RE = re.compile(r"^[\w\.\-]+:\d{2,5}$") def _normalize_line(line: str, default_scheme: str = "http") -> str: s = (line or "").strip() if not s or s.startswith("#") or s.startswith("<"): return "" s = s.split()[0].strip(",") if "://" in s: u = urlparse(s) if not u.hostname or not u.port: return "" scheme = (u.scheme or default_scheme).lower() if scheme == "https": scheme = "http" if scheme == "socks5h": scheme = "socks5" return f"{scheme}://{u.hostname}:{u.port}" if not _HOSTPORT_RE.match(s): return "" scheme = (default_scheme or "http").lower() if scheme == "https": scheme = "http" return f"{scheme}://{s}" def parse_txt(body: str, default_scheme: str = "http") -> list[str]: out: list[str] = [] seen: set[str] = set() for line in (body or "").splitlines(): p = _normalize_line(line, default_scheme) if p and p not in seen: seen.add(p) out.append(p) return out def parse_geonode_json(data, default_scheme: str = "http") -> list[str]: out: list[str] = [] seen: set[str] = set() rows = [] if isinstance(data, dict): rows = data.get("data") or data.get("proxies") or [] elif isinstance(data, list): rows = data for row in rows: if not isinstance(row, dict): continue ip = str(row.get("ip") or row.get("host") or "").strip() port = row.get("port") if not ip or not port: continue protocols = row.get("protocols") or row.get("protocol") or [default_scheme] if isinstance(protocols, str): protocols = [protocols] scheme = "http" for pr in protocols: pr = str(pr).lower() if pr in ("socks5", "socks4"): scheme = pr break if pr in ("http", "https"): scheme = "http" p = f"{scheme}://{ip}:{port}" if p not in seen: seen.add(p) out.append(p) return out def fetch_source( source: ProxySource, *, timeout: float = 20.0, log: LogFn | None = None, ) -> list[str]: log = log or _noop t0 = time.time() try: resp = _req.get( source.url, timeout=timeout, impersonate="chrome120", headers={"User-Agent": "Mozilla/5.0 (compatible; grok-register-proxy/1.0)"}, ) status = getattr(resp, "status_code", 0) if status != 200: raise RuntimeError(f"HTTP {status}") if source.kind == "json_geonode": try: data = resp.json() except Exception as exc: raise RuntimeError(f"json parse: {exc}") from exc items = parse_geonode_json(data, source.protocol) else: text = resp.text or "" if text.lstrip().startswith("<") or "error code" in text[:80].lower(): raise RuntimeError(f"bad body: {text[:80]!r}") items = parse_txt(text, source.protocol) elapsed = time.time() - t0 log(f"[proxy-src] {source.id} ok n={len(items)} {elapsed:.1f}s") return items except Exception as exc: log(f"[proxy-src] {source.id} fail: {exc}") raise def effective_weight(source: ProxySource, store_meta: dict | None = None) -> float: """运行权重 = 配置 weight × 健康系数。""" if not source.enabled: return 0.0 meta = store_meta or {} if meta.get("enabled") is False: return 0.0 base = max(0, int(source.weight)) fail = int(meta.get("fail_streak") or source.fail_streak or 0) ok_runs = int(meta.get("ok_runs") or 0) kept = int(meta.get("last_kept") or 0) # 连续失败指数衰减;有产出加成 health = 1.0 / (1.0 + fail * 1.5) if kept > 0: health *= 1.0 + min(1.0, kept / 20.0) if ok_runs > 0 and fail == 0: health *= 1.15 return max(0.0, base * health) def pick_sources( sources: list[ProxySource], *, k: int = 3, store=None, ) -> list[ProxySource]: """按权重不放回抽样最多 k 个源。""" candidates = [] weights = [] for s in sources: meta = store.source_meta(s.id) if store is not None else {} if not store.source_enabled(s.id, default=s.enabled) if store is not None else s.enabled: continue w = effective_weight(s, meta) if w <= 0: continue candidates.append(s) weights.append(w) if not candidates: return [] k = max(1, min(k, len(candidates))) picked: list[ProxySource] = [] pool = list(zip(candidates, weights)) for _ in range(k): if not pool: break cs, ws = zip(*pool) choice = random.choices(list(cs), weights=list(ws), k=1)[0] picked.append(choice) pool = [(c, w) for c, w in pool if c.id != choice.id] return picked def harvest( sources: list[ProxySource] | None = None, *, pick_k: int = 3, per_source_limit: int = 200, total_limit: int = 600, timeout: float = 20.0, store=None, log: LogFn | None = None, ) -> tuple[list[str], dict[str, list[str]]]: """从加权选中的源采集候选代理。 返回 (all_candidates, by_source)。 """ log = log or _noop sources = list(sources or DEFAULT_SOURCES) chosen = pick_sources(sources, k=pick_k, store=store) if not chosen: # 全部被禁用时尝试强制取 weight 最高的 2 个做恢复 ranked = sorted(sources, key=lambda s: s.weight, reverse=True)[:2] chosen = ranked log("[proxy-src] 所有源被禁用/权重为0,强制尝试恢复 top weight 源") by_source: dict[str, list[str]] = {} all_seen: set[str] = set() all_list: list[str] = [] for src in chosen: try: items = fetch_source(src, timeout=timeout, log=log) if per_source_limit > 0: random.shuffle(items) items = items[:per_source_limit] by_source[src.id] = items for p in items: if p not in all_seen: all_seen.add(p) all_list.append(p) if store is not None: # 先记 harvest;kept 在测完后由 manager 写 store.source_ok(src.id, harvested=len(items), kept=0) except Exception as exc: by_source[src.id] = [] if store is not None: disabled = store.source_fail(src.id, reason=str(exc), disable_after=3) if disabled: log(f"[proxy-src] 源已禁用: {src.id} ({exc})") continue if total_limit and len(all_list) >= total_limit: break if total_limit and len(all_list) > total_limit: random.shuffle(all_list) all_list = all_list[:total_limit] log(f"[proxy-src] harvest done sources={len(chosen)} candidates={len(all_list)}") return all_list, by_source