Keep proxy_pool_enabled/managed off for direct registration, while shipping SSO OAuth CPA export, visible Turnstile click handling, and proxy harvest tooling.
344 lines
11 KiB
Python
344 lines
11 KiB
Python
#!/usr/bin/env python
|
||
# -*- coding: utf-8 -*-
|
||
"""公开免费代理源(国外可访问)+ 加权分流采集。
|
||
|
||
设计:
|
||
- 每个源有 weight;采集时按权重随机/轮询分流
|
||
- 源连续拉不到可用代理 / 被 x.ai 相关链路废掉 → 降权或禁用
|
||
- 采集结果只返回候选 URL,是否入库由 proxy_manager 测 accounts.x.ai 决定
|
||
"""
|
||
|
||
from __future__ import annotations
|
||
|
||
import re
|
||
import random
|
||
import time
|
||
from dataclasses import dataclass, field
|
||
from typing import Callable
|
||
from urllib.parse import urlparse
|
||
|
||
try:
|
||
from curl_cffi import requests as _req
|
||
except Exception: # pragma: no cover
|
||
import requests as _req # type: ignore
|
||
|
||
|
||
LogFn = Callable[[str], None]
|
||
|
||
|
||
def _noop(msg: str) -> None:
|
||
return None
|
||
|
||
|
||
@dataclass
|
||
class ProxySource:
|
||
id: str
|
||
name: str
|
||
url: str
|
||
weight: int = 10
|
||
protocol: str = "http" # 结果默认 scheme
|
||
kind: str = "txt" # txt | json_geonode | json_list
|
||
enabled: bool = True
|
||
# 运行时统计(内存;持久化在 ProxyStore.sources)
|
||
fail_streak: int = 0
|
||
last_ok: float = 0.0
|
||
last_fail: float = 0.0
|
||
last_reason: str = ""
|
||
harvested_total: int = 0
|
||
kept_total: int = 0
|
||
|
||
|
||
# 公开源(无需 key)。权重仅作初始值,运行中会按成功率动态调。
|
||
DEFAULT_SOURCES: list[ProxySource] = [
|
||
ProxySource(
|
||
id="proxyscrape_http",
|
||
name="ProxyScrape HTTP API",
|
||
url="https://api.proxyscrape.com/v2/?request=displayproxies&protocol=http&timeout=8000&country=all&ssl=all&anonymity=all",
|
||
weight=30,
|
||
protocol="http",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="proxyscrape_socks5",
|
||
name="ProxyScrape SOCKS5 API",
|
||
url="https://api.proxyscrape.com/v2/?request=displayproxies&protocol=socks5&timeout=8000&country=all",
|
||
weight=15,
|
||
protocol="socks5",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="databay_http",
|
||
name="Databay free-proxy-list HTTP",
|
||
url="https://cdn.jsdelivr.net/gh/databay-labs/free-proxy-list@master/http.txt",
|
||
weight=25,
|
||
protocol="http",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="databay_socks5",
|
||
name="Databay free-proxy-list SOCKS5",
|
||
url="https://cdn.jsdelivr.net/gh/databay-labs/free-proxy-list@master/socks5.txt",
|
||
weight=12,
|
||
protocol="socks5",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="monosans_http",
|
||
name="monosans/proxy-list HTTP",
|
||
url="https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt",
|
||
weight=18,
|
||
protocol="http",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="monosans_socks5",
|
||
name="monosans/proxy-list SOCKS5",
|
||
url="https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/socks5.txt",
|
||
weight=10,
|
||
protocol="socks5",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="speedx_http",
|
||
name="TheSpeedX PROXY-List HTTP",
|
||
url="https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt",
|
||
weight=12,
|
||
protocol="http",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="openproxylist_http",
|
||
name="openproxylist HTTP",
|
||
url="https://api.openproxylist.xyz/http.txt",
|
||
weight=10,
|
||
protocol="http",
|
||
kind="txt",
|
||
),
|
||
ProxySource(
|
||
id="geonode_http",
|
||
name="Geonode free proxy API",
|
||
url="https://proxylist.geonode.com/api/proxy-list?limit=200&page=1&sort_by=lastChecked&sort_type=desc&protocols=http%2Chttps",
|
||
weight=8,
|
||
protocol="http",
|
||
kind="json_geonode",
|
||
),
|
||
]
|
||
|
||
|
||
_HOSTPORT_RE = re.compile(r"^[\w\.\-]+:\d{2,5}$")
|
||
|
||
|
||
def _normalize_line(line: str, default_scheme: str = "http") -> str:
|
||
s = (line or "").strip()
|
||
if not s or s.startswith("#") or s.startswith("<"):
|
||
return ""
|
||
s = s.split()[0].strip(",")
|
||
if "://" in s:
|
||
u = urlparse(s)
|
||
if not u.hostname or not u.port:
|
||
return ""
|
||
scheme = (u.scheme or default_scheme).lower()
|
||
if scheme == "https":
|
||
scheme = "http"
|
||
if scheme == "socks5h":
|
||
scheme = "socks5"
|
||
return f"{scheme}://{u.hostname}:{u.port}"
|
||
if not _HOSTPORT_RE.match(s):
|
||
return ""
|
||
scheme = (default_scheme or "http").lower()
|
||
if scheme == "https":
|
||
scheme = "http"
|
||
return f"{scheme}://{s}"
|
||
|
||
|
||
def parse_txt(body: str, default_scheme: str = "http") -> list[str]:
|
||
out: list[str] = []
|
||
seen: set[str] = set()
|
||
for line in (body or "").splitlines():
|
||
p = _normalize_line(line, default_scheme)
|
||
if p and p not in seen:
|
||
seen.add(p)
|
||
out.append(p)
|
||
return out
|
||
|
||
|
||
def parse_geonode_json(data, default_scheme: str = "http") -> list[str]:
|
||
out: list[str] = []
|
||
seen: set[str] = set()
|
||
rows = []
|
||
if isinstance(data, dict):
|
||
rows = data.get("data") or data.get("proxies") or []
|
||
elif isinstance(data, list):
|
||
rows = data
|
||
for row in rows:
|
||
if not isinstance(row, dict):
|
||
continue
|
||
ip = str(row.get("ip") or row.get("host") or "").strip()
|
||
port = row.get("port")
|
||
if not ip or not port:
|
||
continue
|
||
protocols = row.get("protocols") or row.get("protocol") or [default_scheme]
|
||
if isinstance(protocols, str):
|
||
protocols = [protocols]
|
||
scheme = "http"
|
||
for pr in protocols:
|
||
pr = str(pr).lower()
|
||
if pr in ("socks5", "socks4"):
|
||
scheme = pr
|
||
break
|
||
if pr in ("http", "https"):
|
||
scheme = "http"
|
||
p = f"{scheme}://{ip}:{port}"
|
||
if p not in seen:
|
||
seen.add(p)
|
||
out.append(p)
|
||
return out
|
||
|
||
|
||
def fetch_source(
|
||
source: ProxySource,
|
||
*,
|
||
timeout: float = 20.0,
|
||
log: LogFn | None = None,
|
||
) -> list[str]:
|
||
log = log or _noop
|
||
t0 = time.time()
|
||
try:
|
||
resp = _req.get(
|
||
source.url,
|
||
timeout=timeout,
|
||
impersonate="chrome120",
|
||
headers={"User-Agent": "Mozilla/5.0 (compatible; grok-register-proxy/1.0)"},
|
||
)
|
||
status = getattr(resp, "status_code", 0)
|
||
if status != 200:
|
||
raise RuntimeError(f"HTTP {status}")
|
||
if source.kind == "json_geonode":
|
||
try:
|
||
data = resp.json()
|
||
except Exception as exc:
|
||
raise RuntimeError(f"json parse: {exc}") from exc
|
||
items = parse_geonode_json(data, source.protocol)
|
||
else:
|
||
text = resp.text or ""
|
||
if text.lstrip().startswith("<") or "error code" in text[:80].lower():
|
||
raise RuntimeError(f"bad body: {text[:80]!r}")
|
||
items = parse_txt(text, source.protocol)
|
||
elapsed = time.time() - t0
|
||
log(f"[proxy-src] {source.id} ok n={len(items)} {elapsed:.1f}s")
|
||
return items
|
||
except Exception as exc:
|
||
log(f"[proxy-src] {source.id} fail: {exc}")
|
||
raise
|
||
|
||
|
||
def effective_weight(source: ProxySource, store_meta: dict | None = None) -> float:
|
||
"""运行权重 = 配置 weight × 健康系数。"""
|
||
if not source.enabled:
|
||
return 0.0
|
||
meta = store_meta or {}
|
||
if meta.get("enabled") is False:
|
||
return 0.0
|
||
base = max(0, int(source.weight))
|
||
fail = int(meta.get("fail_streak") or source.fail_streak or 0)
|
||
ok_runs = int(meta.get("ok_runs") or 0)
|
||
kept = int(meta.get("last_kept") or 0)
|
||
# 连续失败指数衰减;有产出加成
|
||
health = 1.0 / (1.0 + fail * 1.5)
|
||
if kept > 0:
|
||
health *= 1.0 + min(1.0, kept / 20.0)
|
||
if ok_runs > 0 and fail == 0:
|
||
health *= 1.15
|
||
return max(0.0, base * health)
|
||
|
||
|
||
def pick_sources(
|
||
sources: list[ProxySource],
|
||
*,
|
||
k: int = 3,
|
||
store=None,
|
||
) -> list[ProxySource]:
|
||
"""按权重不放回抽样最多 k 个源。"""
|
||
candidates = []
|
||
weights = []
|
||
for s in sources:
|
||
meta = store.source_meta(s.id) if store is not None else {}
|
||
if not store.source_enabled(s.id, default=s.enabled) if store is not None else s.enabled:
|
||
continue
|
||
w = effective_weight(s, meta)
|
||
if w <= 0:
|
||
continue
|
||
candidates.append(s)
|
||
weights.append(w)
|
||
if not candidates:
|
||
return []
|
||
k = max(1, min(k, len(candidates)))
|
||
picked: list[ProxySource] = []
|
||
pool = list(zip(candidates, weights))
|
||
for _ in range(k):
|
||
if not pool:
|
||
break
|
||
cs, ws = zip(*pool)
|
||
choice = random.choices(list(cs), weights=list(ws), k=1)[0]
|
||
picked.append(choice)
|
||
pool = [(c, w) for c, w in pool if c.id != choice.id]
|
||
return picked
|
||
|
||
|
||
def harvest(
|
||
sources: list[ProxySource] | None = None,
|
||
*,
|
||
pick_k: int = 3,
|
||
per_source_limit: int = 200,
|
||
total_limit: int = 600,
|
||
timeout: float = 20.0,
|
||
store=None,
|
||
log: LogFn | None = None,
|
||
) -> tuple[list[str], dict[str, list[str]]]:
|
||
"""从加权选中的源采集候选代理。
|
||
|
||
返回 (all_candidates, by_source)。
|
||
"""
|
||
log = log or _noop
|
||
sources = list(sources or DEFAULT_SOURCES)
|
||
chosen = pick_sources(sources, k=pick_k, store=store)
|
||
if not chosen:
|
||
# 全部被禁用时尝试强制取 weight 最高的 2 个做恢复
|
||
ranked = sorted(sources, key=lambda s: s.weight, reverse=True)[:2]
|
||
chosen = ranked
|
||
log("[proxy-src] 所有源被禁用/权重为0,强制尝试恢复 top weight 源")
|
||
|
||
by_source: dict[str, list[str]] = {}
|
||
all_seen: set[str] = set()
|
||
all_list: list[str] = []
|
||
|
||
for src in chosen:
|
||
try:
|
||
items = fetch_source(src, timeout=timeout, log=log)
|
||
if per_source_limit > 0:
|
||
random.shuffle(items)
|
||
items = items[:per_source_limit]
|
||
by_source[src.id] = items
|
||
for p in items:
|
||
if p not in all_seen:
|
||
all_seen.add(p)
|
||
all_list.append(p)
|
||
if store is not None:
|
||
# 先记 harvest;kept 在测完后由 manager 写
|
||
store.source_ok(src.id, harvested=len(items), kept=0)
|
||
except Exception as exc:
|
||
by_source[src.id] = []
|
||
if store is not None:
|
||
disabled = store.source_fail(src.id, reason=str(exc), disable_after=3)
|
||
if disabled:
|
||
log(f"[proxy-src] 源已禁用: {src.id} ({exc})")
|
||
continue
|
||
if total_limit and len(all_list) >= total_limit:
|
||
break
|
||
|
||
if total_limit and len(all_list) > total_limit:
|
||
random.shuffle(all_list)
|
||
all_list = all_list[:total_limit]
|
||
log(f"[proxy-src] harvest done sources={len(chosen)} candidates={len(all_list)}")
|
||
return all_list, by_source
|