Files
grok-keygen-new/proxy_sources.py
T
chaos a4dbe948b2 Disable proxy pool by default; add managed proxy lifecycle and Turnstile fixes.
Keep proxy_pool_enabled/managed off for direct registration, while shipping
SSO OAuth CPA export, visible Turnstile click handling, and proxy harvest tooling.
2026-07-13 07:16:11 +08:00

344 lines
11 KiB
Python
Raw Blame History

This file contains ambiguous Unicode characters
This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.
#!/usr/bin/env python
# -*- coding: utf-8 -*-
"""公开免费代理源(国外可访问)+ 加权分流采集。
设计:
- 每个源有 weight;采集时按权重随机/轮询分流
- 源连续拉不到可用代理 / 被 x.ai 相关链路废掉 → 降权或禁用
- 采集结果只返回候选 URL,是否入库由 proxy_manager 测 accounts.x.ai 决定
"""
from __future__ import annotations
import re
import random
import time
from dataclasses import dataclass, field
from typing import Callable
from urllib.parse import urlparse
try:
from curl_cffi import requests as _req
except Exception: # pragma: no cover
import requests as _req # type: ignore
LogFn = Callable[[str], None]
def _noop(msg: str) -> None:
return None
@dataclass
class ProxySource:
id: str
name: str
url: str
weight: int = 10
protocol: str = "http" # 结果默认 scheme
kind: str = "txt" # txt | json_geonode | json_list
enabled: bool = True
# 运行时统计(内存;持久化在 ProxyStore.sources)
fail_streak: int = 0
last_ok: float = 0.0
last_fail: float = 0.0
last_reason: str = ""
harvested_total: int = 0
kept_total: int = 0
# 公开源(无需 key)。权重仅作初始值,运行中会按成功率动态调。
DEFAULT_SOURCES: list[ProxySource] = [
ProxySource(
id="proxyscrape_http",
name="ProxyScrape HTTP API",
url="https://api.proxyscrape.com/v2/?request=displayproxies&protocol=http&timeout=8000&country=all&ssl=all&anonymity=all",
weight=30,
protocol="http",
kind="txt",
),
ProxySource(
id="proxyscrape_socks5",
name="ProxyScrape SOCKS5 API",
url="https://api.proxyscrape.com/v2/?request=displayproxies&protocol=socks5&timeout=8000&country=all",
weight=15,
protocol="socks5",
kind="txt",
),
ProxySource(
id="databay_http",
name="Databay free-proxy-list HTTP",
url="https://cdn.jsdelivr.net/gh/databay-labs/free-proxy-list@master/http.txt",
weight=25,
protocol="http",
kind="txt",
),
ProxySource(
id="databay_socks5",
name="Databay free-proxy-list SOCKS5",
url="https://cdn.jsdelivr.net/gh/databay-labs/free-proxy-list@master/socks5.txt",
weight=12,
protocol="socks5",
kind="txt",
),
ProxySource(
id="monosans_http",
name="monosans/proxy-list HTTP",
url="https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/http.txt",
weight=18,
protocol="http",
kind="txt",
),
ProxySource(
id="monosans_socks5",
name="monosans/proxy-list SOCKS5",
url="https://raw.githubusercontent.com/monosans/proxy-list/main/proxies/socks5.txt",
weight=10,
protocol="socks5",
kind="txt",
),
ProxySource(
id="speedx_http",
name="TheSpeedX PROXY-List HTTP",
url="https://raw.githubusercontent.com/TheSpeedX/PROXY-List/master/http.txt",
weight=12,
protocol="http",
kind="txt",
),
ProxySource(
id="openproxylist_http",
name="openproxylist HTTP",
url="https://api.openproxylist.xyz/http.txt",
weight=10,
protocol="http",
kind="txt",
),
ProxySource(
id="geonode_http",
name="Geonode free proxy API",
url="https://proxylist.geonode.com/api/proxy-list?limit=200&page=1&sort_by=lastChecked&sort_type=desc&protocols=http%2Chttps",
weight=8,
protocol="http",
kind="json_geonode",
),
]
_HOSTPORT_RE = re.compile(r"^[\w\.\-]+:\d{2,5}$")
def _normalize_line(line: str, default_scheme: str = "http") -> str:
s = (line or "").strip()
if not s or s.startswith("#") or s.startswith("<"):
return ""
s = s.split()[0].strip(",")
if "://" in s:
u = urlparse(s)
if not u.hostname or not u.port:
return ""
scheme = (u.scheme or default_scheme).lower()
if scheme == "https":
scheme = "http"
if scheme == "socks5h":
scheme = "socks5"
return f"{scheme}://{u.hostname}:{u.port}"
if not _HOSTPORT_RE.match(s):
return ""
scheme = (default_scheme or "http").lower()
if scheme == "https":
scheme = "http"
return f"{scheme}://{s}"
def parse_txt(body: str, default_scheme: str = "http") -> list[str]:
out: list[str] = []
seen: set[str] = set()
for line in (body or "").splitlines():
p = _normalize_line(line, default_scheme)
if p and p not in seen:
seen.add(p)
out.append(p)
return out
def parse_geonode_json(data, default_scheme: str = "http") -> list[str]:
out: list[str] = []
seen: set[str] = set()
rows = []
if isinstance(data, dict):
rows = data.get("data") or data.get("proxies") or []
elif isinstance(data, list):
rows = data
for row in rows:
if not isinstance(row, dict):
continue
ip = str(row.get("ip") or row.get("host") or "").strip()
port = row.get("port")
if not ip or not port:
continue
protocols = row.get("protocols") or row.get("protocol") or [default_scheme]
if isinstance(protocols, str):
protocols = [protocols]
scheme = "http"
for pr in protocols:
pr = str(pr).lower()
if pr in ("socks5", "socks4"):
scheme = pr
break
if pr in ("http", "https"):
scheme = "http"
p = f"{scheme}://{ip}:{port}"
if p not in seen:
seen.add(p)
out.append(p)
return out
def fetch_source(
source: ProxySource,
*,
timeout: float = 20.0,
log: LogFn | None = None,
) -> list[str]:
log = log or _noop
t0 = time.time()
try:
resp = _req.get(
source.url,
timeout=timeout,
impersonate="chrome120",
headers={"User-Agent": "Mozilla/5.0 (compatible; grok-register-proxy/1.0)"},
)
status = getattr(resp, "status_code", 0)
if status != 200:
raise RuntimeError(f"HTTP {status}")
if source.kind == "json_geonode":
try:
data = resp.json()
except Exception as exc:
raise RuntimeError(f"json parse: {exc}") from exc
items = parse_geonode_json(data, source.protocol)
else:
text = resp.text or ""
if text.lstrip().startswith("<") or "error code" in text[:80].lower():
raise RuntimeError(f"bad body: {text[:80]!r}")
items = parse_txt(text, source.protocol)
elapsed = time.time() - t0
log(f"[proxy-src] {source.id} ok n={len(items)} {elapsed:.1f}s")
return items
except Exception as exc:
log(f"[proxy-src] {source.id} fail: {exc}")
raise
def effective_weight(source: ProxySource, store_meta: dict | None = None) -> float:
"""运行权重 = 配置 weight × 健康系数。"""
if not source.enabled:
return 0.0
meta = store_meta or {}
if meta.get("enabled") is False:
return 0.0
base = max(0, int(source.weight))
fail = int(meta.get("fail_streak") or source.fail_streak or 0)
ok_runs = int(meta.get("ok_runs") or 0)
kept = int(meta.get("last_kept") or 0)
# 连续失败指数衰减;有产出加成
health = 1.0 / (1.0 + fail * 1.5)
if kept > 0:
health *= 1.0 + min(1.0, kept / 20.0)
if ok_runs > 0 and fail == 0:
health *= 1.15
return max(0.0, base * health)
def pick_sources(
sources: list[ProxySource],
*,
k: int = 3,
store=None,
) -> list[ProxySource]:
"""按权重不放回抽样最多 k 个源。"""
candidates = []
weights = []
for s in sources:
meta = store.source_meta(s.id) if store is not None else {}
if not store.source_enabled(s.id, default=s.enabled) if store is not None else s.enabled:
continue
w = effective_weight(s, meta)
if w <= 0:
continue
candidates.append(s)
weights.append(w)
if not candidates:
return []
k = max(1, min(k, len(candidates)))
picked: list[ProxySource] = []
pool = list(zip(candidates, weights))
for _ in range(k):
if not pool:
break
cs, ws = zip(*pool)
choice = random.choices(list(cs), weights=list(ws), k=1)[0]
picked.append(choice)
pool = [(c, w) for c, w in pool if c.id != choice.id]
return picked
def harvest(
sources: list[ProxySource] | None = None,
*,
pick_k: int = 3,
per_source_limit: int = 200,
total_limit: int = 600,
timeout: float = 20.0,
store=None,
log: LogFn | None = None,
) -> tuple[list[str], dict[str, list[str]]]:
"""从加权选中的源采集候选代理。
返回 (all_candidates, by_source)。
"""
log = log or _noop
sources = list(sources or DEFAULT_SOURCES)
chosen = pick_sources(sources, k=pick_k, store=store)
if not chosen:
# 全部被禁用时尝试强制取 weight 最高的 2 个做恢复
ranked = sorted(sources, key=lambda s: s.weight, reverse=True)[:2]
chosen = ranked
log("[proxy-src] 所有源被禁用/权重为0,强制尝试恢复 top weight 源")
by_source: dict[str, list[str]] = {}
all_seen: set[str] = set()
all_list: list[str] = []
for src in chosen:
try:
items = fetch_source(src, timeout=timeout, log=log)
if per_source_limit > 0:
random.shuffle(items)
items = items[:per_source_limit]
by_source[src.id] = items
for p in items:
if p not in all_seen:
all_seen.add(p)
all_list.append(p)
if store is not None:
# 先记 harvest;kept 在测完后由 manager 写
store.source_ok(src.id, harvested=len(items), kept=0)
except Exception as exc:
by_source[src.id] = []
if store is not None:
disabled = store.source_fail(src.id, reason=str(exc), disable_after=3)
if disabled:
log(f"[proxy-src] 源已禁用: {src.id} ({exc})")
continue
if total_limit and len(all_list) >= total_limit:
break
if total_limit and len(all_list) > total_limit:
random.shuffle(all_list)
all_list = all_list[:total_limit]
log(f"[proxy-src] harvest done sources={len(chosen)} candidates={len(all_list)}")
return all_list, by_source