feat: merge public subscription sources and classify TCP connectivity

This commit is contained in:
ermaozi 2026-09-08 07:37:26 +08:00
parent 243b33ae81
commit ac1ac622c8
11 changed files with 509 additions and 169 deletions

View File

@ -3,11 +3,16 @@ on:
workflow_dispatch:
schedule:
- cron: '0 2 * * 1'
permissions:
contents: write
concurrency:
group: repository-updates
cancel-in-progress: false
jobs:
build:
runs-on: ubuntu-latest
steps:
- uses: actions/checkout@v2
- uses: actions/checkout@v7
- name: 删除提交记录
run: |
git config --global user.name "ermaozi"

View File

@ -3,36 +3,40 @@ on:
workflow_dispatch:
schedule:
- cron: '0 */12 * * *'
permissions:
contents: write
concurrency:
group: repository-updates
cancel-in-progress: false
env:
TZ: Asia/Shanghai
jobs:
deploy:
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- name: 迁出代码
uses: actions/checkout@v2
uses: actions/checkout@v7
- name: 安装Python
uses: actions/setup-python@v2
uses: actions/setup-python@v7
with:
python-version: '3.11.6'
- name: 加载缓存
uses: actions/cache@v3
with:
path: ~/.cache/pip
key: ${{ runner.os }}-pip-${{ hashFiles('**/run_in_Actions/requirements.txt') }}
restore-keys: |
${{ runner.os }}-pip-
- name: 设置时区
run: sudo timedatectl set-timezone 'Asia/Shanghai'
python-version: '3.12'
cache: pip
cache-dependency-path: requirements.txt
- name: 安装依赖
run: |
pip install -r requirements.txt
python -m pip install -r requirements.txt
- name: 执行任务
env:
GH_TOKEN: ${{ github.token }}
run: |
python get_projaec_info.py --user ermaozi --project get_subscribe --save_path mail/project_info.svg --theme dark --token ${{ secrets.TOKEN }}
python get_projaec_info.py --user ermaozi --project get_subscribe --save_path mail/project_info.svg --theme dark
- name: 提交更改
run: |
git config core.ignorecase false
git config --local user.email "admin@ermao.net"
git config --local user.name "ermaozi"
git add .
git add mail/project_info.svg
git diff --cached --quiet && exit 0
git commit -m "更新项目信息"
git push

View File

@ -5,40 +5,46 @@ on:
workflow_dispatch:
# 定时触发
schedule:
# 每6小时获取一次
# 每12小时获取一次
- cron: '0 */12 * * *'
permissions:
contents: write
concurrency:
group: repository-updates
cancel-in-progress: false
env:
TZ: Asia/Shanghai
jobs:
deploy:
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- name: 迁出代码
uses: actions/checkout@v2
uses: actions/checkout@v7
- name: 安装Python
uses: actions/setup-python@v4
uses: actions/setup-python@v7
with:
python-version: '3.10'
- name: 加载缓存
uses: actions/cache@v3
with:
path: ~/.cache/pip
key: ${{ runner.os }}-pip-${{ hashFiles('**/run_in_Actions/requirements.txt') }}
restore-keys: |
${{ runner.os }}-pip-
- name: 设置时区
run: sudo timedatectl set-timezone 'Asia/Shanghai'
python-version: '3.12'
cache: pip
cache-dependency-path: requirements.txt
- name: 安装依赖
run: |
pip install -r requirements.txt
python -m pip install -r requirements.txt
- name: 回归检查
run: python -m unittest discover -s tests
- name: 执行任务
run: |
python main.py
- name: TCP 连通性分类
run: python check_nodes.py
- name: 提交更改
run: |
git config core.ignorecase false
git config --local user.email "admin@ermao.net"
git config --local user.name "ermaozi"
git add .
git add subscribe log
git diff --cached --quiet && exit 0
git commit -m "$(date '+%Y-%m-%d %H:%M:%S')更新订阅链接"
git push

5
.gitignore vendored
View File

@ -1 +1,4 @@
venv/
venv/
.venv/
__pycache__/
*.pyc

View File

@ -53,3 +53,45 @@ https://www.ermao.net/sub/v2ray/ermao.net
我搜罗的一些比较便宜好用的机场,觉得免费订阅不好使的朋友们可以在这里面找找。
[https://www.ermao.net/posts/vpn](https://www.ermao.net/posts/vpn)
## 采集与发布架构
本仓库通过 GitHub Actions 每 12 小时采集长风分享、
[NoMoreWalls](https://github.com/peasoft/NoMoreWalls) 和
[ProxyPool](https://github.com/snakem982/proxypool) 的公开订阅。
BestClash 不在采集来源中。
Clash 按节点配置去重(忽略名称),重名节点自动改名,并加入原有节点选择组;
保留第一个可用来源的分流规则。V2Ray 兼容明文和 Base64 来源,按完整 URI 去重,
统一保存为主项目现有的明文节点列表。一个来源失败会继续其他来源,
缺少任一种有效订阅时保留原有文件并令任务失败。格式检查不代表节点连通或速度保证。
当前公开 `/sub/` 地址由独立 Cloudflare Worker `sub` 提供,Worker 自行采集并写入 R2,
不是从本仓库读取文件。因此本仓库增加来源不会自动同步到线上 Worker;
GitHub 版本可通过仓库中的 `subscribe/clash.yml` 和 `subscribe/v2ray.txt` 获取。
使用 Python 3.12 执行本地检查:
```sh
python -m pip install -r requirements.txt
python -m unittest discover -s tests -v
python main.py
```
保留 `SUBSCRIBE_PROXY` 代理环境变量。GitHub 工作流使用内置 `GITHUB_TOKEN`,
无需额外的 `TOKEN` Secret。写入工作流共享并发组,避免同时更新分支;
原有每周清理提交历史行为保持不变。频繁手动触发时,GitHub 可能替换仍在排队的任务。
## 连通性分类
每次自动采集后运行 `python check_nodes.py`,保留 `subscribe/` 下的完整订阅,
另生成以下三个目录(各含 `clash.yml` 和 `v2ray.txt`):
- `subscribe/reachable/`:从本次运行机器可以建立 TCP 连接的节点。
- `subscribe/unreachable/`:TCP 连接或 DNS 解析失败的节点。
- `subscribe/untested/`:UDP/QUIC 协议、无法解析地址或非公网地址,不判定为不可用。
`subscribe/health.json` 保存检测时间和分类数量。同一主机端口只检查一次,
最多并发 24 个端点,每个端点的 TCP 连接总预算为 3 秒(不含系统 DNS 解析时间)。
这不是代理认证、出口访问或速度测试;可连接不等于代理可用,失败也可能是运行机器的网络限制。

123
check_nodes.py Normal file
View File

@ -0,0 +1,123 @@
"""按 TCP 握手结果分类订阅;UDP/QUIC 和无法解析的格式不作失效判断。"""
import base64
from concurrent.futures import ThreadPoolExecutor
from datetime import datetime, timezone
import ipaddress
import json
from pathlib import Path
import socket
import time
from urllib.parse import parse_qs, urlsplit
import yaml
TIMEOUT = 3
WORKERS = 24
STATUSES = ('reachable', 'unreachable', 'untested')
UDP_TYPES = {'hysteria', 'hysteria2', 'hy2', 'tuic', 'wireguard', 'mieru'}
def _decode(value):
return base64.urlsafe_b64decode(value + '=' * (-len(value) % 4)).decode('utf-8')
def endpoint(node):
"""返回 TCP 主机和端口;不支持或使用 UDP 的节点返回 None。"""
try:
if isinstance(node, dict):
if node.get('type') in UDP_TYPES or node.get('network') in {'quic', 'kcp'}:
return None
host, port = node.get('server'), node.get('port')
else:
scheme, body = node.split('://', 1)
if scheme in UDP_TYPES:
return None
if scheme == 'vmess':
data = json.loads(_decode(body.split('#', 1)[0]))
if not isinstance(data, dict) or data.get('net') in {'quic', 'kcp'}:
return None
host, port = data.get('add'), data.get('port')
elif scheme == 'ssr':
host, port, *_ = _decode(body.split('#', 1)[0]).split('/?', 1)[0].rsplit(':', 5)
else:
if scheme == 'ss' and '@' not in body:
node = 'ss://' + _decode(body.split('#', 1)[0])
parsed = urlsplit(node)
if any(v in {'quic', 'kcp'} for v in parse_qs(parsed.query).get('type', [])):
return None
host, port = parsed.hostname, parsed.port
port = int(port)
if not isinstance(host, str) or not host or not 1 <= port <= 65535:
return None
return host.strip('[]').lower().rstrip('.'), port
except (ValueError, TypeError, KeyError, UnicodeError):
return None
def tcp_status(address):
# ponytail: 只测 TCP 握手;需要验证认证和代理出口时再引入代理内核。
host, port = address
try:
resolved = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM)
except OSError:
return 'unreachable'
public = []
for family, socktype, proto, _, sockaddr in resolved:
# 来源不可信,不探测宿主机内网、回环或云实例元数据地址。
if ipaddress.ip_address(sockaddr[0]).is_global:
public.append((family, socktype, proto, sockaddr))
if not public:
return 'untested'
deadline = time.monotonic() + TIMEOUT
for family, socktype, proto, sockaddr in public:
remaining = deadline - time.monotonic()
if remaining <= 0:
break
try:
with socket.socket(family, socktype, proto) as connection:
connection.settimeout(remaining)
connection.connect(sockaddr)
return 'reachable'
except OSError:
continue
return 'unreachable'
def classify(directory='subscribe'):
directory = Path(directory)
config = yaml.safe_load((directory / 'clash.yml').read_text(encoding='utf-8'))
uris = (directory / 'v2ray.txt').read_text(encoding='utf-8').splitlines()
proxies = config['proxies']
proxy_addresses = [endpoint(proxy) for proxy in proxies]
uri_addresses = [endpoint(uri) for uri in uris]
addresses = list(dict.fromkeys(a for a in proxy_addresses + uri_addresses if a is not None))
with ThreadPoolExecutor(max_workers=WORKERS) as pool:
results = dict(zip(addresses, pool.map(tcp_status, addresses)))
summary = {'checked_at': datetime.now(timezone.utc).isoformat(), 'method': 'TCP connect',
'timeout_seconds': TIMEOUT, 'unique_endpoints': len(addresses), 'counts': {}}
all_names = {proxy['name'] for proxy in proxies}
for status in STATUSES:
selected = [p for p, address in zip(proxies, proxy_addresses) if results.get(address, 'untested') == status]
selected_uris = [uri for uri, address in zip(uris, uri_addresses) if results.get(address, 'untested') == status]
removed = all_names - {proxy['name'] for proxy in selected}
filtered = {**config, 'proxies': selected}
groups = []
for group in config.get('proxy-groups', []):
group = dict(group)
if 'proxies' in group:
group['proxies'] = [name for name in group['proxies'] if name not in removed] or ['REJECT']
groups.append(group)
if 'proxy-groups' in config:
filtered['proxy-groups'] = groups
target = directory / status
target.mkdir(exist_ok=True)
(target / 'clash.yml').write_text(yaml.safe_dump(filtered, allow_unicode=True, sort_keys=False), encoding='utf-8')
(target / 'v2ray.txt').write_text('\n'.join(selected_uris) + ('\n' if selected_uris else ''), encoding='utf-8')
summary['counts'][status] = {'clash': len(selected), 'v2ray': len(selected_uris)}
(directory / 'health.json').write_text(json.dumps(summary, ensure_ascii=False, indent=2) + '\n', encoding='utf-8')
print(json.dumps(summary, ensure_ascii=False))
return summary
if __name__ == '__main__':
classify()

View File

@ -1,4 +1,5 @@
import argparse
import os
import re
import matplotlib.pyplot as plt
@ -19,11 +20,12 @@ def get_project_info(user, project, name, item, date_key, token=""):
})
data_list = []
page = 0
date_pat = re.compile("\d{4}-\d{2}-\d{2}")
date_pat = re.compile(r"\d{4}-\d{2}-\d{2}")
while True:
page += 1
url = f"https://api.github.com/repos/{user}/{project}/{item}?page={page}"
req = requests.get(url, headers=header)
url = f"https://api.github.com/repos/{user}/{project}/{item}?page={page}&per_page=100"
req = requests.get(url, headers=header, timeout=30)
req.raise_for_status()
datas = req.json()
if not datas:
break
@ -31,7 +33,7 @@ def get_project_info(user, project, name, item, date_key, token=""):
date_dic = {}
start_date = min(data_list)
start_date = min(data_list) if data_list else date.today().isoformat()
end_date = date.today()
for date_str in data_list:
if not date_dic.get(date_str):
@ -92,6 +94,7 @@ def create_svg(project, datas, save_path, theme=""):
ax.grid(True, linestyle='-.')
plt.savefig(save_path)
plt.close(fig)
def main(user, project, save_path, theme="", token=""):
@ -108,6 +111,6 @@ if __name__ == "__main__":
parser.add_argument("--project", type=str)
parser.add_argument("--save_path", type=str)
parser.add_argument("--theme", type=str, default="")
parser.add_argument("--token", type=str, default="")
parser.add_argument("--token", type=str, default=os.environ.get("GH_TOKEN", ""))
args = parser.parse_args()
main(args.user, args.project, args.save_path, args.theme, args.token)

260
main.py
View File

@ -1,33 +1,39 @@
import base64
import os
import re
import smtplib
import sys
import time
import html
import json
from pathlib import Path
from urllib.parse import urlsplit
import xml.etree.ElementTree as ET
from email.mime.text import MIMEText
from email.utils import formataddr
import requests
import yaml
from requests.adapters import HTTPAdapter
from urllib3.util.retry import Retry
requests.packages.urllib3.disable_warnings()
ok_code = [200, 201, 202, 203, 204, 205, 206]
# 邮箱域名过滤列表
blackhole_list = ["cnr.cn", "cyberpolice.cn", "gov.cn", "samr.gov.cn", "12321.cn"
"miit.gov.cn", "chinatcc.gov.cn"]
ok_code = [200]
DIRECT_SOURCES = {
"NoMoreWalls": [
"https://raw.githubusercontent.com/peasoft/NoMoreWalls/master/list.meta.yml",
"https://raw.githubusercontent.com/peasoft/NoMoreWalls/master/list.txt",
],
"ProxyPool": [
"https://raw.githubusercontent.com/snakem982/proxypool/main/source/clash-meta-2.yaml",
"https://raw.githubusercontent.com/snakem982/proxypool/main/source/v2ray-2.txt",
],
}
def write_log(content, level="INFO"):
date_str = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(time.time()))
update_log = f"[{date_str}] [{level}] {content}\n"
print(update_log)
print(update_log, end="")
Path("log").mkdir(exist_ok=True)
with open(f'./log/{time.strftime("%Y-%m", time.localtime(time.time()))}-update.log', 'a', encoding="utf-8") as f:
f.write(update_log)
@ -42,7 +48,7 @@ def _extract_urls(summary):
return urls, decoded
_NODE_SCHEME_RE = re.compile(r"(?:vmess|vless|trojan|ss|ssr|hysteria2?|tuic)://")
_NODE_SCHEME_RE = re.compile(r"(?:vmess|vless|trojan|ss|ssr|hysteria2?|hy2|tuic|anytls|mieru|https?|socks[45]?)://")
def _b64decode(text):
@ -51,10 +57,13 @@ def _b64decode(text):
return ""
try:
# binascii.Error 是 ValueError 的子类,统一捕获即可
raw = base64.b64decode(compact + "=" * (-len(compact) % 4))
raw = base64.b64decode(compact + "=" * (-len(compact) % 4), validate=True)
except ValueError:
return ""
return raw.decode("utf-8", "ignore")
try:
return raw.decode("utf-8")
except UnicodeDecodeError:
return ""
def _detect_kind(text):
@ -62,39 +71,33 @@ def _detect_kind(text):
sample = text.strip()
if not sample:
return None
# clash 配置为 YAML,包含 proxies/proxy-groups 字段
if re.search(r"^(?:proxies|proxy-groups)\s*:", sample, re.MULTILINE):
return "clash"
# v2ray 订阅为节点 URI 列表,可能是明文或 base64 编码
if _NODE_SCHEME_RE.search(sample):
return "v2ray"
decoded = _b64decode(sample)
if decoded and _NODE_SCHEME_RE.search(decoded):
try:
data = yaml.safe_load(sample)
except yaml.YAMLError:
return None
proxies = data.get("proxies") if isinstance(data, dict) else None
if isinstance(proxies, list) and proxies and all(
isinstance(proxy, dict) and isinstance(proxy.get("name"), str) and proxy["name"]
and proxy.get("type") and proxy.get("server") for proxy in proxies
):
return "clash"
return None
decoded = sample if _NODE_SCHEME_RE.match(sample) else _b64decode(sample)
nodes = [node.strip() for node in decoded.splitlines() if node.strip()]
if nodes and all(_NODE_SCHEME_RE.match(node) and not re.search(r"\s", node) for node in nodes):
try:
for node in nodes:
if node.startswith(("http://", "https://", "socks://", "socks4://", "socks5://")):
parsed = urlsplit(node)
if not parsed.hostname or not parsed.port:
return None
except ValueError:
return None
return "v2ray"
return None
def _download_with_retry(urls):
if not urls:
return None, None
for url in urls:
for _ in range(3):
try:
req = requests.request(
"GET",
url,
verify=False,
timeout=20,
headers={"User-Agent": "Mozilla/5.0"},
)
except requests.RequestException as e:
print(f"请求 {url} 失败: {e}")
continue
if req.status_code in ok_code:
return req, url
return None, urls[0]
def _build_session():
session = requests.Session()
retry = Retry(
@ -124,106 +127,109 @@ def _classify_subscriptions(session, urls):
if "v2ray" in found and "clash" in found:
break
try:
req = session.get(url, verify=False, timeout=20)
req = session.get(url, timeout=20)
except requests.RequestException as e:
write_log(f"请求失败:{url} - {e}", "WARN")
write_log(f"候选订阅请求失败:{type(e).__name__}", "WARN")
continue
if req.status_code not in ok_code:
write_log(f"请求失败:{url} - {req.status_code}", "WARN")
write_log(f"候选订阅 HTTP {req.status_code}", "WARN")
continue
kind = _detect_kind(req.text)
if kind and kind not in found:
found[kind] = (req, url)
write_log(f"识别到 {kind} 订阅:{url}", "INFO")
write_log(f"识别到有效的 {kind} 订阅", "INFO")
return found
def _merge_clash(sources):
# 保留第一个可用来源的规则,将所有来源节点加入已有的节点选择组。
config = yaml.safe_load(sources[0][1])
proxies, identities, names, renamed = [], {}, set(), {}
names.update(group["name"] for group in config.get("proxy-groups", []))
names.update(["DIRECT", "REJECT", "REJECT-DROP", "PASS", "COMPATIBLE", "GLOBAL"])
for index, (source, content) in enumerate(sources):
for proxy in yaml.safe_load(content)["proxies"]:
identity = json.dumps({k: v for k, v in proxy.items() if k != "name"}, sort_keys=True, default=str)
if identity not in identities:
name = proxy["name"]
if name in names:
name = f"{source} | {name}"
original = name
suffix = 2
while name in names:
name = f"{original} ({suffix})"
suffix += 1
identities[identity] = name
names.add(name)
proxies.append({**proxy, "name": name})
if index == 0:
renamed[proxy["name"]] = identities[identity]
for group in config.get("proxy-groups", []):
members = group.get("proxies", [])
if any(member in renamed for member in members):
group["proxies"] = list(dict.fromkeys(
[renamed.get(member, member) for member in members] + [p["name"] for p in proxies]
))
config["proxies"] = proxies
return yaml.safe_dump(config, allow_unicode=True, sort_keys=False)
def get_subscribe_url():
dirs = './subscribe'
if not os.path.exists(dirs):
os.makedirs(dirs)
log_dir = "./log"
if not os.path.exists(log_dir):
os.makedirs(log_dir)
collected = {"clash": [], "v2ray": []}
with _build_session() as session:
try:
rss = session.get('https://www.cfmem.com/feeds/posts/default?alt=rss', timeout=20)
rss.raise_for_status()
root = ET.fromstring(rss.content)
for item in root.findall("./channel/item"):
urls, _ = _extract_urls(item.findtext("description") or "")
found = _classify_subscriptions(session, urls)
if found:
for kind, (response, _) in found.items():
collected[kind].append(("长风分享", response.text))
break
except (requests.RequestException, ET.ParseError) as exc:
write_log(f"长风分享采集失败:{type(exc).__name__},继续其他来源", "WARN")
for source, urls in DIRECT_SOURCES.items():
found = _classify_subscriptions(session, urls)
for kind, (response, _) in found.items():
collected[kind].append((source, response.text))
write_log(f"{source}:获取 {', '.join(found) or '无有效订阅'}")
update_list = []
session = _build_session()
try:
rss_req = session.get(
'https://www.cfmem.com/feeds/posts/default?alt=rss',
timeout=20,
)
except requests.RequestException as ex:
write_log(f"更新失败!拉取 RSS 异常: {ex}", "ERROR")
return
if rss_req.status_code not in ok_code:
write_log(f"更新失败!无法拉取原网站内容 - {rss_req.status_code}", "ERROR")
return
try:
root = ET.fromstring(rss_req.text)
except ET.ParseError as ex:
write_log(f"更新失败!RSS 解析失败: {ex}", "ERROR")
return
item = root.find("./channel/item")
if item is None:
write_log("更新失败!RSS 中未找到可用条目", "ERROR")
return
summary = item.findtext("description")
if not summary:
write_log("暂时没有可用的订阅更新", "WARN")
return
urls, _ = _extract_urls(summary)
# 链接已无固定后缀,需下载内容后再判断是 v2ray 还是 clash
classified = _classify_subscriptions(session, urls)
# 获取普通订阅链接
v2ray_entry = classified.get("v2ray")
if v2ray_entry:
v2ray_req, _ = v2ray_entry
update_list.append(f"v2ray: {v2ray_req.status_code}")
with open(dirs + '/v2ray.txt', 'w', encoding="utf-8") as f:
f.write(v2ray_req.text)
else:
cache_file = dirs + '/v2ray.txt'
if os.path.exists(cache_file) and os.path.getsize(cache_file) > 0:
update_list.append("v2ray: cache")
write_log("未获取到 v2ray 订阅,已保留本地缓存", "WARN")
else:
write_log("未获取到 v2ray 订阅", "WARN")
# 获取clash订阅链接
clash_entry = classified.get("clash")
if clash_entry:
clash_req, _ = clash_entry
update_list.append(f"clash: {clash_req.status_code}")
with open(dirs + '/clash.yml', 'w', encoding="utf-8") as f:
f.write(clash_req.content.decode("utf-8"))
else:
cache_file = dirs + '/clash.yml'
if os.path.exists(cache_file) and os.path.getsize(cache_file) > 0:
update_list.append("clash: cache")
write_log("未获取到 clash 订阅,已保留本地缓存", "WARN")
else:
write_log("未获取到 clash 订阅", "WARN")
if update_list:
file_pat = re.compile(r"v2ray\.txt|clash\.yml")
if file_pat.search(os.popen("git status").read()):
write_log(f"更新成功:{update_list}", "INFO")
else:
write_log(f"订阅暂未更新", "WARN")
else:
write_log(f"未能获取新的更新内容", "WARN")
if not all(collected.values()):
write_log("未获取到完整的两种订阅,保留旧文件", "ERROR")
return False
nodes = []
for _, content in collected["v2ray"]:
text = content.strip() if _NODE_SCHEME_RE.match(content.strip()) else _b64decode(content)
nodes.extend(node.strip() for node in text.splitlines() if node.strip())
# 保持主项目原有的明文 URI 输出,兼容现有订阅入口。
outputs = {"clash.yml": _merge_clash(collected["clash"]),
"v2ray.txt": "\n".join(dict.fromkeys(nodes)) + "\n"}
for filename, content in outputs.items():
expected = "clash" if filename.endswith(".yml") else "v2ray"
if _detect_kind(content) != expected:
raise ValueError(f"合并后的 {expected} 订阅格式无效")
Path("subscribe").mkdir(exist_ok=True)
changed = []
for filename, content in outputs.items():
target = Path("subscribe") / filename
if not target.exists() or target.read_text(encoding="utf-8") != content:
temporary = target.with_suffix(target.suffix + ".tmp")
temporary.write_text(content, encoding="utf-8")
temporary.replace(target)
changed.append(filename)
write_log("已更新:" + ", ".join(changed) if changed else "订阅内容未变化")
return True
def main():
get_subscribe_url()
return 0 if get_subscribe_url() else 1
# 主函数入口
if __name__ == '__main__':
main()
try:
sys.exit(main())
except (OSError, ValueError, yaml.YAMLError) as exc:
write_log(f"采集或写入失败:{type(exc).__name__}", "ERROR")
sys.exit(1)

View File

@ -1,5 +1,4 @@
requests
feedparser
matplotlib
pandas
requests==2.34.2
matplotlib==3.10.9
pandas==2.3.3
PyYAML==6.0.3

60
tests/test_health.py Normal file
View File

@ -0,0 +1,60 @@
import base64
import json
from pathlib import Path
import socket
import tempfile
import unittest
from unittest.mock import Mock, patch
import yaml
import check_nodes
class HealthTests(unittest.TestCase):
def test_endpoints_and_udp_are_not_misclassified(self):
vmess = base64.b64encode(json.dumps({'add': 'example.com', 'port': '443'}).encode()).decode()
ss = base64.b64encode(b'aes-256-gcm:password@example.com:8443').decode()
ssr = base64.urlsafe_b64encode(b'example.com:443:origin:aes-256-cfb:plain:cGFzcw/?x=y').decode()
for node, expected in [({'type': 'ss', 'server': 'example.com', 'port': 443}, ('example.com',443)),
('vmess://' + vmess, ('example.com',443)), ('ss://' + ss, ('example.com',8443)),
('ssr://' + ssr, ('example.com',443)), ('vless://id@[2606:4700:4700::1111]:443', ('2606:4700:4700::1111',443)),
('hysteria2://password@example.com:443',None),
({'type':'tuic', 'server':'example.com', 'port':443},None),
('vless://id@example.com:443?type=quic',None), ('vmess://invalid',None)]:
with self.subTest(node=node):
self.assertEqual(check_nodes.endpoint(node), expected)
def test_private_addresses_are_not_connected(self):
with patch.object(socket, 'getaddrinfo', return_value=[(socket.AF_INET,socket.SOCK_STREAM,6,'',('127.0.0.1',443))]), patch.object(socket, 'socket') as connect:
self.assertEqual(check_nodes.tcp_status(('local.example',443)), 'untested')
connect.assert_not_called()
with patch.object(socket, 'getaddrinfo', side_effect=socket.gaierror):
self.assertEqual(check_nodes.tcp_status(('bad.example',443)), 'unreachable')
def test_tcp_success_and_timeout(self):
resolved = [(socket.AF_INET,socket.SOCK_STREAM,6,'',('1.1.1.1',443))]
with patch.object(socket, 'getaddrinfo', return_value=resolved), patch.object(socket, 'socket') as factory:
self.assertEqual(check_nodes.tcp_status(('public.example',443)), 'reachable')
factory.return_value.__enter__.return_value.connect.assert_called_once_with(('1.1.1.1',443))
factory.return_value.__enter__.return_value.connect.side_effect = socket.timeout
self.assertEqual(check_nodes.tcp_status(('public.example',443)), 'unreachable')
def test_classification_is_complete_and_keeps_original(self):
config = {'proxies': [
{'name':'up','type':'ss','server':'up.example','port':443},
{'name':'down','type':'ss','server':'down.example','port':443},
{'name':'udp','type':'tuic','server':'udp.example','port':443}],
'proxy-groups':[{'name':'select','type':'select','proxies':['up','down','udp']}]}
uris = 'ss://YWVzOnBhc3M=@up.example:443\ntuic://id@udp.example:443\n'
with tempfile.TemporaryDirectory() as directory:
path=Path(directory);source=yaml.safe_dump(config)
(path/'clash.yml').write_text(source);(path/'v2ray.txt').write_text(uris)
with patch.object(check_nodes, 'tcp_status', side_effect=lambda a: 'reachable' if a[0]=='up.example' else 'unreachable') as probe:
result=check_nodes.classify(path)
self.assertEqual(probe.call_count,2)
self.assertEqual(result['counts']['untested'],{'clash':1,'v2ray':1})
self.assertEqual((path/'clash.yml').read_text(),source)
self.assertEqual((path/'v2ray.txt').read_text(),uris)
for status,name in [('reachable','up'),('unreachable','down'),('untested','udp')]:
output=yaml.safe_load((path/status/'clash.yml').read_text())
self.assertEqual(output['proxy-groups'][0]['proxies'],[name])

View File

@ -0,0 +1,89 @@
import base64
from contextlib import nullcontext
import yaml
import os
from pathlib import Path
import tempfile
import unittest
from unittest.mock import Mock, patch
import main
CLASH = 'proxies:\n - {name: test, type: ss, server: example.com, port: 443}\n'
NODES = 'vmess://example\nvless://example'
class SubscriptionTests(unittest.TestCase):
def test_real_formats_and_error_pages(self):
for text, expected in [(CLASH, 'clash'), (NODES, 'v2ray'),
(base64.b64encode(NODES.encode()).decode().rstrip('='), 'v2ray'),
('https://example.com:443#proxy', 'v2ray'), ('socks5://example.com:1080', 'v2ray'),
('https://example.com/article', None), ('https://example.com:bad', None), ('proxies: []', None), ('proxies: [', None),
('<html>error vmess://example</html>', None), ('vmess://example\n<html>error</html>', None),
('proxies:\n - broken', None), ('', None)]:
with self.subTest(text=text):
self.assertEqual(main._detect_kind(text), expected)
def test_download_uses_tls_and_skips_invalid_content(self):
session = Mock()
session.get.side_effect = [Mock(status_code=200, text='<html>error</html>'),
Mock(status_code=200, text=CLASH), Mock(status_code=200, text=NODES)]
with patch.object(main, 'write_log'):
found = main._classify_subscriptions(session, ['https://example.com/error', 'https://example.com/c', 'https://example.com/v'])
self.assertEqual(set(found), {'clash', 'v2ray'})
self.assertTrue(all(call.kwargs == {'timeout': 20} for call in session.get.call_args_list))
def test_collection_and_failed_source_keeps_cache(self):
feed = '<rss><channel><item><description>https://example.com/c https://example.com/v</description></item></channel></rss>'
session = Mock()
session.get.side_effect = [Mock(status_code=200, text=feed, content=feed.encode()),
Mock(status_code=200, text=CLASH, content=CLASH.encode()),
Mock(status_code=200, text=NODES)]
with tempfile.TemporaryDirectory() as directory:
previous = os.getcwd(); os.chdir(directory)
try:
with patch.object(main, '_build_session', return_value=nullcontext(session)), patch.object(main, 'DIRECT_SOURCES', {}):
self.assertEqual(main.main(), 0)
self.assertEqual(yaml.safe_load(Path('subscribe/clash.yml').read_text()), yaml.safe_load(CLASH))
self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES)
session.get.side_effect = [Mock(raise_for_status=Mock(side_effect=main.requests.HTTPError))]
with patch.object(main, '_build_session', return_value=nullcontext(session)), patch.object(main, 'DIRECT_SOURCES', {}):
self.assertEqual(main.main(), 1)
self.assertEqual(yaml.safe_load(Path('subscribe/clash.yml').read_text()), yaml.safe_load(CLASH))
self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES)
finally:
os.chdir(previous)
def test_merge_deduplicates_and_keeps_groups_resolvable(self):
first = CLASH + 'proxy-groups:\n - {name: select, type: select, proxies: [test, DIRECT]}\nrules: ["MATCH,select"]\n'
duplicate = CLASH.replace('name: test', 'name: another')
collision = CLASH.replace('example.com', 'other.example')
data = yaml.safe_load(main._merge_clash([('original', first), ('NoMoreWalls', duplicate), ('ProxyPool', collision)]))
names = [p['name'] for p in data['proxies']]
self.assertEqual(names, ['test', 'ProxyPool | test'])
self.assertEqual(data['proxy-groups'][0]['proxies'], ['test', 'DIRECT', 'ProxyPool | test'])
self.assertEqual(data['rules'], ['MATCH,select'])
def test_direct_sources_work_when_rss_fails(self):
session = Mock()
session.get.side_effect = [main.requests.ConnectionError,
Mock(status_code=200, text=CLASH), Mock(status_code=200, text=NODES),
Mock(status_code=200, text=CLASH), Mock(status_code=200, text=base64.b64encode(NODES.encode()).decode())]
with tempfile.TemporaryDirectory() as directory:
previous = os.getcwd(); os.chdir(directory)
try:
with patch.object(main, '_build_session', return_value=nullcontext(session)):
self.assertEqual(main.main(), 0)
self.assertEqual(len(yaml.safe_load(Path('subscribe/clash.yml').read_text())['proxies']), 1)
self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES)
self.assertEqual(session.get.call_count, 5)
finally:
os.chdir(previous)
def test_github_empty_history_and_http_errors(self):
from get_projaec_info import get_project_info
with patch('get_projaec_info.requests.get', return_value=Mock(json=Mock(return_value=[]))):
self.assertEqual(get_project_info('u','p','star','stargazers','starred_at')['num_list'], [0])
with patch('get_projaec_info.requests.get', return_value=Mock(raise_for_status=Mock(side_effect=main.requests.HTTPError))):
with self.assertRaises(main.requests.HTTPError):
get_project_info('u','p','star','stargazers','starred_at')