mirror of
https://github.com/ermaozi/get_subscribe.git
synced 2026-09-30 04:01:41 +00:00
feat: merge public subscription sources and classify TCP connectivity
This commit is contained in:
parent
243b33ae81
commit
ac1ac622c8
7
.github/workflows/clear.yml
vendored
7
.github/workflows/clear.yml
vendored
@ -3,11 +3,16 @@ on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: '0 2 * * 1'
|
||||
permissions:
|
||||
contents: write
|
||||
concurrency:
|
||||
group: repository-updates
|
||||
cancel-in-progress: false
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- uses: actions/checkout@v2
|
||||
- uses: actions/checkout@v7
|
||||
- name: 删除提交记录
|
||||
run: |
|
||||
git config --global user.name "ermaozi"
|
||||
|
||||
34
.github/workflows/get_projaec_info.yml
vendored
34
.github/workflows/get_projaec_info.yml
vendored
@ -3,36 +3,40 @@ on:
|
||||
workflow_dispatch:
|
||||
schedule:
|
||||
- cron: '0 */12 * * *'
|
||||
permissions:
|
||||
contents: write
|
||||
concurrency:
|
||||
group: repository-updates
|
||||
cancel-in-progress: false
|
||||
env:
|
||||
TZ: Asia/Shanghai
|
||||
jobs:
|
||||
deploy:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: 迁出代码
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v7
|
||||
- name: 安装Python
|
||||
uses: actions/setup-python@v2
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: '3.11.6'
|
||||
- name: 加载缓存
|
||||
uses: actions/cache@v3
|
||||
with:
|
||||
path: ~/.cache/pip
|
||||
key: ${{ runner.os }}-pip-${{ hashFiles('**/run_in_Actions/requirements.txt') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-pip-
|
||||
- name: 设置时区
|
||||
run: sudo timedatectl set-timezone 'Asia/Shanghai'
|
||||
python-version: '3.12'
|
||||
cache: pip
|
||||
cache-dependency-path: requirements.txt
|
||||
- name: 安装依赖
|
||||
run: |
|
||||
pip install -r requirements.txt
|
||||
python -m pip install -r requirements.txt
|
||||
- name: 执行任务
|
||||
env:
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
python get_projaec_info.py --user ermaozi --project get_subscribe --save_path mail/project_info.svg --theme dark --token ${{ secrets.TOKEN }}
|
||||
python get_projaec_info.py --user ermaozi --project get_subscribe --save_path mail/project_info.svg --theme dark
|
||||
- name: 提交更改
|
||||
run: |
|
||||
git config core.ignorecase false
|
||||
git config --local user.email "admin@ermao.net"
|
||||
git config --local user.name "ermaozi"
|
||||
git add .
|
||||
git add mail/project_info.svg
|
||||
git diff --cached --quiet && exit 0
|
||||
git commit -m "更新项目信息"
|
||||
git push
|
||||
|
||||
36
.github/workflows/main.yml
vendored
36
.github/workflows/main.yml
vendored
@ -5,40 +5,46 @@ on:
|
||||
workflow_dispatch:
|
||||
# 定时触发
|
||||
schedule:
|
||||
# 每6小时获取一次
|
||||
# 每12小时获取一次
|
||||
- cron: '0 */12 * * *'
|
||||
|
||||
permissions:
|
||||
contents: write
|
||||
concurrency:
|
||||
group: repository-updates
|
||||
cancel-in-progress: false
|
||||
env:
|
||||
TZ: Asia/Shanghai
|
||||
jobs:
|
||||
deploy:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- name: 迁出代码
|
||||
uses: actions/checkout@v2
|
||||
uses: actions/checkout@v7
|
||||
- name: 安装Python
|
||||
uses: actions/setup-python@v4
|
||||
uses: actions/setup-python@v7
|
||||
with:
|
||||
python-version: '3.10'
|
||||
- name: 加载缓存
|
||||
uses: actions/cache@v3
|
||||
with:
|
||||
path: ~/.cache/pip
|
||||
key: ${{ runner.os }}-pip-${{ hashFiles('**/run_in_Actions/requirements.txt') }}
|
||||
restore-keys: |
|
||||
${{ runner.os }}-pip-
|
||||
- name: 设置时区
|
||||
run: sudo timedatectl set-timezone 'Asia/Shanghai'
|
||||
python-version: '3.12'
|
||||
cache: pip
|
||||
cache-dependency-path: requirements.txt
|
||||
- name: 安装依赖
|
||||
run: |
|
||||
pip install -r requirements.txt
|
||||
python -m pip install -r requirements.txt
|
||||
- name: 回归检查
|
||||
run: python -m unittest discover -s tests
|
||||
- name: 执行任务
|
||||
run: |
|
||||
python main.py
|
||||
- name: TCP 连通性分类
|
||||
run: python check_nodes.py
|
||||
|
||||
- name: 提交更改
|
||||
run: |
|
||||
git config core.ignorecase false
|
||||
git config --local user.email "admin@ermao.net"
|
||||
git config --local user.name "ermaozi"
|
||||
git add .
|
||||
git add subscribe log
|
||||
git diff --cached --quiet && exit 0
|
||||
git commit -m "$(date '+%Y-%m-%d %H:%M:%S')更新订阅链接"
|
||||
git push
|
||||
|
||||
5
.gitignore
vendored
5
.gitignore
vendored
@ -1 +1,4 @@
|
||||
venv/
|
||||
venv/
|
||||
.venv/
|
||||
__pycache__/
|
||||
*.pyc
|
||||
|
||||
42
README.md
42
README.md
@ -53,3 +53,45 @@ https://www.ermao.net/sub/v2ray/ermao.net
|
||||
我搜罗的一些比较便宜好用的机场,觉得免费订阅不好使的朋友们可以在这里面找找。
|
||||
|
||||
[https://www.ermao.net/posts/vpn](https://www.ermao.net/posts/vpn)
|
||||
|
||||
## 采集与发布架构
|
||||
|
||||
本仓库通过 GitHub Actions 每 12 小时采集长风分享、
|
||||
[NoMoreWalls](https://github.com/peasoft/NoMoreWalls) 和
|
||||
[ProxyPool](https://github.com/snakem982/proxypool) 的公开订阅。
|
||||
BestClash 不在采集来源中。
|
||||
|
||||
Clash 按节点配置去重(忽略名称),重名节点自动改名,并加入原有节点选择组;
|
||||
保留第一个可用来源的分流规则。V2Ray 兼容明文和 Base64 来源,按完整 URI 去重,
|
||||
统一保存为主项目现有的明文节点列表。一个来源失败会继续其他来源,
|
||||
缺少任一种有效订阅时保留原有文件并令任务失败。格式检查不代表节点连通或速度保证。
|
||||
|
||||
当前公开 `/sub/` 地址由独立 Cloudflare Worker `sub` 提供,Worker 自行采集并写入 R2,
|
||||
不是从本仓库读取文件。因此本仓库增加来源不会自动同步到线上 Worker;
|
||||
GitHub 版本可通过仓库中的 `subscribe/clash.yml` 和 `subscribe/v2ray.txt` 获取。
|
||||
|
||||
使用 Python 3.12 执行本地检查:
|
||||
|
||||
```sh
|
||||
python -m pip install -r requirements.txt
|
||||
python -m unittest discover -s tests -v
|
||||
python main.py
|
||||
```
|
||||
|
||||
保留 `SUBSCRIBE_PROXY` 代理环境变量。GitHub 工作流使用内置 `GITHUB_TOKEN`,
|
||||
无需额外的 `TOKEN` Secret。写入工作流共享并发组,避免同时更新分支;
|
||||
原有每周清理提交历史行为保持不变。频繁手动触发时,GitHub 可能替换仍在排队的任务。
|
||||
|
||||
|
||||
## 连通性分类
|
||||
|
||||
每次自动采集后运行 `python check_nodes.py`,保留 `subscribe/` 下的完整订阅,
|
||||
另生成以下三个目录(各含 `clash.yml` 和 `v2ray.txt`):
|
||||
|
||||
- `subscribe/reachable/`:从本次运行机器可以建立 TCP 连接的节点。
|
||||
- `subscribe/unreachable/`:TCP 连接或 DNS 解析失败的节点。
|
||||
- `subscribe/untested/`:UDP/QUIC 协议、无法解析地址或非公网地址,不判定为不可用。
|
||||
|
||||
`subscribe/health.json` 保存检测时间和分类数量。同一主机端口只检查一次,
|
||||
最多并发 24 个端点,每个端点的 TCP 连接总预算为 3 秒(不含系统 DNS 解析时间)。
|
||||
这不是代理认证、出口访问或速度测试;可连接不等于代理可用,失败也可能是运行机器的网络限制。
|
||||
|
||||
123
check_nodes.py
Normal file
123
check_nodes.py
Normal file
@ -0,0 +1,123 @@
|
||||
"""按 TCP 握手结果分类订阅;UDP/QUIC 和无法解析的格式不作失效判断。"""
|
||||
import base64
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from datetime import datetime, timezone
|
||||
import ipaddress
|
||||
import json
|
||||
from pathlib import Path
|
||||
import socket
|
||||
import time
|
||||
from urllib.parse import parse_qs, urlsplit
|
||||
|
||||
import yaml
|
||||
|
||||
TIMEOUT = 3
|
||||
WORKERS = 24
|
||||
STATUSES = ('reachable', 'unreachable', 'untested')
|
||||
UDP_TYPES = {'hysteria', 'hysteria2', 'hy2', 'tuic', 'wireguard', 'mieru'}
|
||||
|
||||
|
||||
def _decode(value):
|
||||
return base64.urlsafe_b64decode(value + '=' * (-len(value) % 4)).decode('utf-8')
|
||||
|
||||
|
||||
def endpoint(node):
|
||||
"""返回 TCP 主机和端口;不支持或使用 UDP 的节点返回 None。"""
|
||||
try:
|
||||
if isinstance(node, dict):
|
||||
if node.get('type') in UDP_TYPES or node.get('network') in {'quic', 'kcp'}:
|
||||
return None
|
||||
host, port = node.get('server'), node.get('port')
|
||||
else:
|
||||
scheme, body = node.split('://', 1)
|
||||
if scheme in UDP_TYPES:
|
||||
return None
|
||||
if scheme == 'vmess':
|
||||
data = json.loads(_decode(body.split('#', 1)[0]))
|
||||
if not isinstance(data, dict) or data.get('net') in {'quic', 'kcp'}:
|
||||
return None
|
||||
host, port = data.get('add'), data.get('port')
|
||||
elif scheme == 'ssr':
|
||||
host, port, *_ = _decode(body.split('#', 1)[0]).split('/?', 1)[0].rsplit(':', 5)
|
||||
else:
|
||||
if scheme == 'ss' and '@' not in body:
|
||||
node = 'ss://' + _decode(body.split('#', 1)[0])
|
||||
parsed = urlsplit(node)
|
||||
if any(v in {'quic', 'kcp'} for v in parse_qs(parsed.query).get('type', [])):
|
||||
return None
|
||||
host, port = parsed.hostname, parsed.port
|
||||
port = int(port)
|
||||
if not isinstance(host, str) or not host or not 1 <= port <= 65535:
|
||||
return None
|
||||
return host.strip('[]').lower().rstrip('.'), port
|
||||
except (ValueError, TypeError, KeyError, UnicodeError):
|
||||
return None
|
||||
|
||||
|
||||
def tcp_status(address):
|
||||
# ponytail: 只测 TCP 握手;需要验证认证和代理出口时再引入代理内核。
|
||||
host, port = address
|
||||
try:
|
||||
resolved = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM)
|
||||
except OSError:
|
||||
return 'unreachable'
|
||||
public = []
|
||||
for family, socktype, proto, _, sockaddr in resolved:
|
||||
# 来源不可信,不探测宿主机内网、回环或云实例元数据地址。
|
||||
if ipaddress.ip_address(sockaddr[0]).is_global:
|
||||
public.append((family, socktype, proto, sockaddr))
|
||||
if not public:
|
||||
return 'untested'
|
||||
deadline = time.monotonic() + TIMEOUT
|
||||
for family, socktype, proto, sockaddr in public:
|
||||
remaining = deadline - time.monotonic()
|
||||
if remaining <= 0:
|
||||
break
|
||||
try:
|
||||
with socket.socket(family, socktype, proto) as connection:
|
||||
connection.settimeout(remaining)
|
||||
connection.connect(sockaddr)
|
||||
return 'reachable'
|
||||
except OSError:
|
||||
continue
|
||||
return 'unreachable'
|
||||
|
||||
|
||||
def classify(directory='subscribe'):
|
||||
directory = Path(directory)
|
||||
config = yaml.safe_load((directory / 'clash.yml').read_text(encoding='utf-8'))
|
||||
uris = (directory / 'v2ray.txt').read_text(encoding='utf-8').splitlines()
|
||||
proxies = config['proxies']
|
||||
proxy_addresses = [endpoint(proxy) for proxy in proxies]
|
||||
uri_addresses = [endpoint(uri) for uri in uris]
|
||||
addresses = list(dict.fromkeys(a for a in proxy_addresses + uri_addresses if a is not None))
|
||||
with ThreadPoolExecutor(max_workers=WORKERS) as pool:
|
||||
results = dict(zip(addresses, pool.map(tcp_status, addresses)))
|
||||
summary = {'checked_at': datetime.now(timezone.utc).isoformat(), 'method': 'TCP connect',
|
||||
'timeout_seconds': TIMEOUT, 'unique_endpoints': len(addresses), 'counts': {}}
|
||||
all_names = {proxy['name'] for proxy in proxies}
|
||||
for status in STATUSES:
|
||||
selected = [p for p, address in zip(proxies, proxy_addresses) if results.get(address, 'untested') == status]
|
||||
selected_uris = [uri for uri, address in zip(uris, uri_addresses) if results.get(address, 'untested') == status]
|
||||
removed = all_names - {proxy['name'] for proxy in selected}
|
||||
filtered = {**config, 'proxies': selected}
|
||||
groups = []
|
||||
for group in config.get('proxy-groups', []):
|
||||
group = dict(group)
|
||||
if 'proxies' in group:
|
||||
group['proxies'] = [name for name in group['proxies'] if name not in removed] or ['REJECT']
|
||||
groups.append(group)
|
||||
if 'proxy-groups' in config:
|
||||
filtered['proxy-groups'] = groups
|
||||
target = directory / status
|
||||
target.mkdir(exist_ok=True)
|
||||
(target / 'clash.yml').write_text(yaml.safe_dump(filtered, allow_unicode=True, sort_keys=False), encoding='utf-8')
|
||||
(target / 'v2ray.txt').write_text('\n'.join(selected_uris) + ('\n' if selected_uris else ''), encoding='utf-8')
|
||||
summary['counts'][status] = {'clash': len(selected), 'v2ray': len(selected_uris)}
|
||||
(directory / 'health.json').write_text(json.dumps(summary, ensure_ascii=False, indent=2) + '\n', encoding='utf-8')
|
||||
print(json.dumps(summary, ensure_ascii=False))
|
||||
return summary
|
||||
|
||||
|
||||
if __name__ == '__main__':
|
||||
classify()
|
||||
@ -1,4 +1,5 @@
|
||||
import argparse
|
||||
import os
|
||||
import re
|
||||
|
||||
import matplotlib.pyplot as plt
|
||||
@ -19,11 +20,12 @@ def get_project_info(user, project, name, item, date_key, token=""):
|
||||
})
|
||||
data_list = []
|
||||
page = 0
|
||||
date_pat = re.compile("\d{4}-\d{2}-\d{2}")
|
||||
date_pat = re.compile(r"\d{4}-\d{2}-\d{2}")
|
||||
while True:
|
||||
page += 1
|
||||
url = f"https://api.github.com/repos/{user}/{project}/{item}?page={page}"
|
||||
req = requests.get(url, headers=header)
|
||||
url = f"https://api.github.com/repos/{user}/{project}/{item}?page={page}&per_page=100"
|
||||
req = requests.get(url, headers=header, timeout=30)
|
||||
req.raise_for_status()
|
||||
datas = req.json()
|
||||
if not datas:
|
||||
break
|
||||
@ -31,7 +33,7 @@ def get_project_info(user, project, name, item, date_key, token=""):
|
||||
|
||||
date_dic = {}
|
||||
|
||||
start_date = min(data_list)
|
||||
start_date = min(data_list) if data_list else date.today().isoformat()
|
||||
end_date = date.today()
|
||||
for date_str in data_list:
|
||||
if not date_dic.get(date_str):
|
||||
@ -92,6 +94,7 @@ def create_svg(project, datas, save_path, theme=""):
|
||||
ax.grid(True, linestyle='-.')
|
||||
|
||||
plt.savefig(save_path)
|
||||
plt.close(fig)
|
||||
|
||||
|
||||
def main(user, project, save_path, theme="", token=""):
|
||||
@ -108,6 +111,6 @@ if __name__ == "__main__":
|
||||
parser.add_argument("--project", type=str)
|
||||
parser.add_argument("--save_path", type=str)
|
||||
parser.add_argument("--theme", type=str, default="")
|
||||
parser.add_argument("--token", type=str, default="")
|
||||
parser.add_argument("--token", type=str, default=os.environ.get("GH_TOKEN", ""))
|
||||
args = parser.parse_args()
|
||||
main(args.user, args.project, args.save_path, args.theme, args.token)
|
||||
|
||||
260
main.py
260
main.py
@ -1,33 +1,39 @@
|
||||
import base64
|
||||
import os
|
||||
import re
|
||||
import smtplib
|
||||
import sys
|
||||
import time
|
||||
import html
|
||||
import json
|
||||
from pathlib import Path
|
||||
from urllib.parse import urlsplit
|
||||
import xml.etree.ElementTree as ET
|
||||
from email.mime.text import MIMEText
|
||||
from email.utils import formataddr
|
||||
|
||||
import requests
|
||||
import yaml
|
||||
from requests.adapters import HTTPAdapter
|
||||
from urllib3.util.retry import Retry
|
||||
|
||||
requests.packages.urllib3.disable_warnings()
|
||||
|
||||
|
||||
ok_code = [200, 201, 202, 203, 204, 205, 206]
|
||||
|
||||
# 邮箱域名过滤列表
|
||||
blackhole_list = ["cnr.cn", "cyberpolice.cn", "gov.cn", "samr.gov.cn", "12321.cn"
|
||||
"miit.gov.cn", "chinatcc.gov.cn"]
|
||||
|
||||
ok_code = [200]
|
||||
DIRECT_SOURCES = {
|
||||
"NoMoreWalls": [
|
||||
"https://raw.githubusercontent.com/peasoft/NoMoreWalls/master/list.meta.yml",
|
||||
"https://raw.githubusercontent.com/peasoft/NoMoreWalls/master/list.txt",
|
||||
],
|
||||
"ProxyPool": [
|
||||
"https://raw.githubusercontent.com/snakem982/proxypool/main/source/clash-meta-2.yaml",
|
||||
"https://raw.githubusercontent.com/snakem982/proxypool/main/source/v2ray-2.txt",
|
||||
],
|
||||
}
|
||||
|
||||
def write_log(content, level="INFO"):
|
||||
|
||||
date_str = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(time.time()))
|
||||
update_log = f"[{date_str}] [{level}] {content}\n"
|
||||
print(update_log)
|
||||
print(update_log, end="")
|
||||
Path("log").mkdir(exist_ok=True)
|
||||
with open(f'./log/{time.strftime("%Y-%m", time.localtime(time.time()))}-update.log', 'a', encoding="utf-8") as f:
|
||||
f.write(update_log)
|
||||
|
||||
@ -42,7 +48,7 @@ def _extract_urls(summary):
|
||||
return urls, decoded
|
||||
|
||||
|
||||
_NODE_SCHEME_RE = re.compile(r"(?:vmess|vless|trojan|ss|ssr|hysteria2?|tuic)://")
|
||||
_NODE_SCHEME_RE = re.compile(r"(?:vmess|vless|trojan|ss|ssr|hysteria2?|hy2|tuic|anytls|mieru|https?|socks[45]?)://")
|
||||
|
||||
|
||||
def _b64decode(text):
|
||||
@ -51,10 +57,13 @@ def _b64decode(text):
|
||||
return ""
|
||||
try:
|
||||
# binascii.Error 是 ValueError 的子类,统一捕获即可
|
||||
raw = base64.b64decode(compact + "=" * (-len(compact) % 4))
|
||||
raw = base64.b64decode(compact + "=" * (-len(compact) % 4), validate=True)
|
||||
except ValueError:
|
||||
return ""
|
||||
return raw.decode("utf-8", "ignore")
|
||||
try:
|
||||
return raw.decode("utf-8")
|
||||
except UnicodeDecodeError:
|
||||
return ""
|
||||
|
||||
|
||||
def _detect_kind(text):
|
||||
@ -62,39 +71,33 @@ def _detect_kind(text):
|
||||
sample = text.strip()
|
||||
if not sample:
|
||||
return None
|
||||
# clash 配置为 YAML,包含 proxies/proxy-groups 字段
|
||||
if re.search(r"^(?:proxies|proxy-groups)\s*:", sample, re.MULTILINE):
|
||||
return "clash"
|
||||
# v2ray 订阅为节点 URI 列表,可能是明文或 base64 编码
|
||||
if _NODE_SCHEME_RE.search(sample):
|
||||
return "v2ray"
|
||||
decoded = _b64decode(sample)
|
||||
if decoded and _NODE_SCHEME_RE.search(decoded):
|
||||
try:
|
||||
data = yaml.safe_load(sample)
|
||||
except yaml.YAMLError:
|
||||
return None
|
||||
proxies = data.get("proxies") if isinstance(data, dict) else None
|
||||
if isinstance(proxies, list) and proxies and all(
|
||||
isinstance(proxy, dict) and isinstance(proxy.get("name"), str) and proxy["name"]
|
||||
and proxy.get("type") and proxy.get("server") for proxy in proxies
|
||||
):
|
||||
return "clash"
|
||||
return None
|
||||
decoded = sample if _NODE_SCHEME_RE.match(sample) else _b64decode(sample)
|
||||
nodes = [node.strip() for node in decoded.splitlines() if node.strip()]
|
||||
if nodes and all(_NODE_SCHEME_RE.match(node) and not re.search(r"\s", node) for node in nodes):
|
||||
try:
|
||||
for node in nodes:
|
||||
if node.startswith(("http://", "https://", "socks://", "socks4://", "socks5://")):
|
||||
parsed = urlsplit(node)
|
||||
if not parsed.hostname or not parsed.port:
|
||||
return None
|
||||
except ValueError:
|
||||
return None
|
||||
return "v2ray"
|
||||
return None
|
||||
|
||||
|
||||
def _download_with_retry(urls):
|
||||
if not urls:
|
||||
return None, None
|
||||
for url in urls:
|
||||
for _ in range(3):
|
||||
try:
|
||||
req = requests.request(
|
||||
"GET",
|
||||
url,
|
||||
verify=False,
|
||||
timeout=20,
|
||||
headers={"User-Agent": "Mozilla/5.0"},
|
||||
)
|
||||
except requests.RequestException as e:
|
||||
print(f"请求 {url} 失败: {e}")
|
||||
continue
|
||||
if req.status_code in ok_code:
|
||||
return req, url
|
||||
return None, urls[0]
|
||||
|
||||
|
||||
def _build_session():
|
||||
session = requests.Session()
|
||||
retry = Retry(
|
||||
@ -124,106 +127,109 @@ def _classify_subscriptions(session, urls):
|
||||
if "v2ray" in found and "clash" in found:
|
||||
break
|
||||
try:
|
||||
req = session.get(url, verify=False, timeout=20)
|
||||
req = session.get(url, timeout=20)
|
||||
except requests.RequestException as e:
|
||||
write_log(f"请求失败:{url} - {e}", "WARN")
|
||||
write_log(f"候选订阅请求失败:{type(e).__name__}", "WARN")
|
||||
continue
|
||||
if req.status_code not in ok_code:
|
||||
write_log(f"请求失败:{url} - {req.status_code}", "WARN")
|
||||
write_log(f"候选订阅 HTTP {req.status_code}", "WARN")
|
||||
continue
|
||||
kind = _detect_kind(req.text)
|
||||
if kind and kind not in found:
|
||||
found[kind] = (req, url)
|
||||
write_log(f"识别到 {kind} 订阅:{url}", "INFO")
|
||||
write_log(f"识别到有效的 {kind} 订阅", "INFO")
|
||||
return found
|
||||
|
||||
def _merge_clash(sources):
|
||||
# 保留第一个可用来源的规则,将所有来源节点加入已有的节点选择组。
|
||||
config = yaml.safe_load(sources[0][1])
|
||||
proxies, identities, names, renamed = [], {}, set(), {}
|
||||
names.update(group["name"] for group in config.get("proxy-groups", []))
|
||||
names.update(["DIRECT", "REJECT", "REJECT-DROP", "PASS", "COMPATIBLE", "GLOBAL"])
|
||||
for index, (source, content) in enumerate(sources):
|
||||
for proxy in yaml.safe_load(content)["proxies"]:
|
||||
identity = json.dumps({k: v for k, v in proxy.items() if k != "name"}, sort_keys=True, default=str)
|
||||
if identity not in identities:
|
||||
name = proxy["name"]
|
||||
if name in names:
|
||||
name = f"{source} | {name}"
|
||||
original = name
|
||||
suffix = 2
|
||||
while name in names:
|
||||
name = f"{original} ({suffix})"
|
||||
suffix += 1
|
||||
identities[identity] = name
|
||||
names.add(name)
|
||||
proxies.append({**proxy, "name": name})
|
||||
if index == 0:
|
||||
renamed[proxy["name"]] = identities[identity]
|
||||
for group in config.get("proxy-groups", []):
|
||||
members = group.get("proxies", [])
|
||||
if any(member in renamed for member in members):
|
||||
group["proxies"] = list(dict.fromkeys(
|
||||
[renamed.get(member, member) for member in members] + [p["name"] for p in proxies]
|
||||
))
|
||||
config["proxies"] = proxies
|
||||
return yaml.safe_dump(config, allow_unicode=True, sort_keys=False)
|
||||
|
||||
|
||||
def get_subscribe_url():
|
||||
dirs = './subscribe'
|
||||
if not os.path.exists(dirs):
|
||||
os.makedirs(dirs)
|
||||
log_dir = "./log"
|
||||
if not os.path.exists(log_dir):
|
||||
os.makedirs(log_dir)
|
||||
collected = {"clash": [], "v2ray": []}
|
||||
with _build_session() as session:
|
||||
try:
|
||||
rss = session.get('https://www.cfmem.com/feeds/posts/default?alt=rss', timeout=20)
|
||||
rss.raise_for_status()
|
||||
root = ET.fromstring(rss.content)
|
||||
for item in root.findall("./channel/item"):
|
||||
urls, _ = _extract_urls(item.findtext("description") or "")
|
||||
found = _classify_subscriptions(session, urls)
|
||||
if found:
|
||||
for kind, (response, _) in found.items():
|
||||
collected[kind].append(("长风分享", response.text))
|
||||
break
|
||||
except (requests.RequestException, ET.ParseError) as exc:
|
||||
write_log(f"长风分享采集失败:{type(exc).__name__},继续其他来源", "WARN")
|
||||
for source, urls in DIRECT_SOURCES.items():
|
||||
found = _classify_subscriptions(session, urls)
|
||||
for kind, (response, _) in found.items():
|
||||
collected[kind].append((source, response.text))
|
||||
write_log(f"{source}:获取 {', '.join(found) or '无有效订阅'}")
|
||||
|
||||
update_list = []
|
||||
session = _build_session()
|
||||
try:
|
||||
rss_req = session.get(
|
||||
'https://www.cfmem.com/feeds/posts/default?alt=rss',
|
||||
timeout=20,
|
||||
)
|
||||
except requests.RequestException as ex:
|
||||
write_log(f"更新失败!拉取 RSS 异常: {ex}", "ERROR")
|
||||
return
|
||||
|
||||
if rss_req.status_code not in ok_code:
|
||||
write_log(f"更新失败!无法拉取原网站内容 - {rss_req.status_code}", "ERROR")
|
||||
return
|
||||
|
||||
try:
|
||||
root = ET.fromstring(rss_req.text)
|
||||
except ET.ParseError as ex:
|
||||
write_log(f"更新失败!RSS 解析失败: {ex}", "ERROR")
|
||||
return
|
||||
|
||||
item = root.find("./channel/item")
|
||||
if item is None:
|
||||
write_log("更新失败!RSS 中未找到可用条目", "ERROR")
|
||||
return
|
||||
|
||||
summary = item.findtext("description")
|
||||
if not summary:
|
||||
write_log("暂时没有可用的订阅更新", "WARN")
|
||||
return
|
||||
|
||||
urls, _ = _extract_urls(summary)
|
||||
|
||||
# 链接已无固定后缀,需下载内容后再判断是 v2ray 还是 clash
|
||||
classified = _classify_subscriptions(session, urls)
|
||||
|
||||
# 获取普通订阅链接
|
||||
v2ray_entry = classified.get("v2ray")
|
||||
if v2ray_entry:
|
||||
v2ray_req, _ = v2ray_entry
|
||||
update_list.append(f"v2ray: {v2ray_req.status_code}")
|
||||
with open(dirs + '/v2ray.txt', 'w', encoding="utf-8") as f:
|
||||
f.write(v2ray_req.text)
|
||||
else:
|
||||
cache_file = dirs + '/v2ray.txt'
|
||||
if os.path.exists(cache_file) and os.path.getsize(cache_file) > 0:
|
||||
update_list.append("v2ray: cache")
|
||||
write_log("未获取到 v2ray 订阅,已保留本地缓存", "WARN")
|
||||
else:
|
||||
write_log("未获取到 v2ray 订阅", "WARN")
|
||||
|
||||
# 获取clash订阅链接
|
||||
clash_entry = classified.get("clash")
|
||||
if clash_entry:
|
||||
clash_req, _ = clash_entry
|
||||
update_list.append(f"clash: {clash_req.status_code}")
|
||||
with open(dirs + '/clash.yml', 'w', encoding="utf-8") as f:
|
||||
f.write(clash_req.content.decode("utf-8"))
|
||||
else:
|
||||
cache_file = dirs + '/clash.yml'
|
||||
if os.path.exists(cache_file) and os.path.getsize(cache_file) > 0:
|
||||
update_list.append("clash: cache")
|
||||
write_log("未获取到 clash 订阅,已保留本地缓存", "WARN")
|
||||
else:
|
||||
write_log("未获取到 clash 订阅", "WARN")
|
||||
if update_list:
|
||||
file_pat = re.compile(r"v2ray\.txt|clash\.yml")
|
||||
if file_pat.search(os.popen("git status").read()):
|
||||
write_log(f"更新成功:{update_list}", "INFO")
|
||||
else:
|
||||
write_log(f"订阅暂未更新", "WARN")
|
||||
else:
|
||||
write_log(f"未能获取新的更新内容", "WARN")
|
||||
if not all(collected.values()):
|
||||
write_log("未获取到完整的两种订阅,保留旧文件", "ERROR")
|
||||
return False
|
||||
nodes = []
|
||||
for _, content in collected["v2ray"]:
|
||||
text = content.strip() if _NODE_SCHEME_RE.match(content.strip()) else _b64decode(content)
|
||||
nodes.extend(node.strip() for node in text.splitlines() if node.strip())
|
||||
# 保持主项目原有的明文 URI 输出,兼容现有订阅入口。
|
||||
outputs = {"clash.yml": _merge_clash(collected["clash"]),
|
||||
"v2ray.txt": "\n".join(dict.fromkeys(nodes)) + "\n"}
|
||||
for filename, content in outputs.items():
|
||||
expected = "clash" if filename.endswith(".yml") else "v2ray"
|
||||
if _detect_kind(content) != expected:
|
||||
raise ValueError(f"合并后的 {expected} 订阅格式无效")
|
||||
Path("subscribe").mkdir(exist_ok=True)
|
||||
changed = []
|
||||
for filename, content in outputs.items():
|
||||
target = Path("subscribe") / filename
|
||||
if not target.exists() or target.read_text(encoding="utf-8") != content:
|
||||
temporary = target.with_suffix(target.suffix + ".tmp")
|
||||
temporary.write_text(content, encoding="utf-8")
|
||||
temporary.replace(target)
|
||||
changed.append(filename)
|
||||
write_log("已更新:" + ", ".join(changed) if changed else "订阅内容未变化")
|
||||
return True
|
||||
|
||||
|
||||
def main():
|
||||
get_subscribe_url()
|
||||
return 0 if get_subscribe_url() else 1
|
||||
|
||||
|
||||
# 主函数入口
|
||||
if __name__ == '__main__':
|
||||
main()
|
||||
try:
|
||||
sys.exit(main())
|
||||
except (OSError, ValueError, yaml.YAMLError) as exc:
|
||||
write_log(f"采集或写入失败:{type(exc).__name__}", "ERROR")
|
||||
sys.exit(1)
|
||||
|
||||
@ -1,5 +1,4 @@
|
||||
requests
|
||||
feedparser
|
||||
matplotlib
|
||||
pandas
|
||||
|
||||
requests==2.34.2
|
||||
matplotlib==3.10.9
|
||||
pandas==2.3.3
|
||||
PyYAML==6.0.3
|
||||
|
||||
60
tests/test_health.py
Normal file
60
tests/test_health.py
Normal file
@ -0,0 +1,60 @@
|
||||
import base64
|
||||
import json
|
||||
from pathlib import Path
|
||||
import socket
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import Mock, patch
|
||||
|
||||
import yaml
|
||||
import check_nodes
|
||||
|
||||
|
||||
class HealthTests(unittest.TestCase):
|
||||
def test_endpoints_and_udp_are_not_misclassified(self):
|
||||
vmess = base64.b64encode(json.dumps({'add': 'example.com', 'port': '443'}).encode()).decode()
|
||||
ss = base64.b64encode(b'aes-256-gcm:password@example.com:8443').decode()
|
||||
ssr = base64.urlsafe_b64encode(b'example.com:443:origin:aes-256-cfb:plain:cGFzcw/?x=y').decode()
|
||||
for node, expected in [({'type': 'ss', 'server': 'example.com', 'port': 443}, ('example.com',443)),
|
||||
('vmess://' + vmess, ('example.com',443)), ('ss://' + ss, ('example.com',8443)),
|
||||
('ssr://' + ssr, ('example.com',443)), ('vless://id@[2606:4700:4700::1111]:443', ('2606:4700:4700::1111',443)),
|
||||
('hysteria2://password@example.com:443',None),
|
||||
({'type':'tuic', 'server':'example.com', 'port':443},None),
|
||||
('vless://id@example.com:443?type=quic',None), ('vmess://invalid',None)]:
|
||||
with self.subTest(node=node):
|
||||
self.assertEqual(check_nodes.endpoint(node), expected)
|
||||
|
||||
def test_private_addresses_are_not_connected(self):
|
||||
with patch.object(socket, 'getaddrinfo', return_value=[(socket.AF_INET,socket.SOCK_STREAM,6,'',('127.0.0.1',443))]), patch.object(socket, 'socket') as connect:
|
||||
self.assertEqual(check_nodes.tcp_status(('local.example',443)), 'untested')
|
||||
connect.assert_not_called()
|
||||
with patch.object(socket, 'getaddrinfo', side_effect=socket.gaierror):
|
||||
self.assertEqual(check_nodes.tcp_status(('bad.example',443)), 'unreachable')
|
||||
|
||||
def test_tcp_success_and_timeout(self):
|
||||
resolved = [(socket.AF_INET,socket.SOCK_STREAM,6,'',('1.1.1.1',443))]
|
||||
with patch.object(socket, 'getaddrinfo', return_value=resolved), patch.object(socket, 'socket') as factory:
|
||||
self.assertEqual(check_nodes.tcp_status(('public.example',443)), 'reachable')
|
||||
factory.return_value.__enter__.return_value.connect.assert_called_once_with(('1.1.1.1',443))
|
||||
factory.return_value.__enter__.return_value.connect.side_effect = socket.timeout
|
||||
self.assertEqual(check_nodes.tcp_status(('public.example',443)), 'unreachable')
|
||||
|
||||
def test_classification_is_complete_and_keeps_original(self):
|
||||
config = {'proxies': [
|
||||
{'name':'up','type':'ss','server':'up.example','port':443},
|
||||
{'name':'down','type':'ss','server':'down.example','port':443},
|
||||
{'name':'udp','type':'tuic','server':'udp.example','port':443}],
|
||||
'proxy-groups':[{'name':'select','type':'select','proxies':['up','down','udp']}]}
|
||||
uris = 'ss://YWVzOnBhc3M=@up.example:443\ntuic://id@udp.example:443\n'
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
path=Path(directory);source=yaml.safe_dump(config)
|
||||
(path/'clash.yml').write_text(source);(path/'v2ray.txt').write_text(uris)
|
||||
with patch.object(check_nodes, 'tcp_status', side_effect=lambda a: 'reachable' if a[0]=='up.example' else 'unreachable') as probe:
|
||||
result=check_nodes.classify(path)
|
||||
self.assertEqual(probe.call_count,2)
|
||||
self.assertEqual(result['counts']['untested'],{'clash':1,'v2ray':1})
|
||||
self.assertEqual((path/'clash.yml').read_text(),source)
|
||||
self.assertEqual((path/'v2ray.txt').read_text(),uris)
|
||||
for status,name in [('reachable','up'),('unreachable','down'),('untested','udp')]:
|
||||
output=yaml.safe_load((path/status/'clash.yml').read_text())
|
||||
self.assertEqual(output['proxy-groups'][0]['proxies'],[name])
|
||||
89
tests/test_subscriptions.py
Normal file
89
tests/test_subscriptions.py
Normal file
@ -0,0 +1,89 @@
|
||||
import base64
|
||||
from contextlib import nullcontext
|
||||
import yaml
|
||||
import os
|
||||
from pathlib import Path
|
||||
import tempfile
|
||||
import unittest
|
||||
from unittest.mock import Mock, patch
|
||||
|
||||
import main
|
||||
|
||||
CLASH = 'proxies:\n - {name: test, type: ss, server: example.com, port: 443}\n'
|
||||
NODES = 'vmess://example\nvless://example'
|
||||
|
||||
|
||||
class SubscriptionTests(unittest.TestCase):
|
||||
def test_real_formats_and_error_pages(self):
|
||||
for text, expected in [(CLASH, 'clash'), (NODES, 'v2ray'),
|
||||
(base64.b64encode(NODES.encode()).decode().rstrip('='), 'v2ray'),
|
||||
('https://example.com:443#proxy', 'v2ray'), ('socks5://example.com:1080', 'v2ray'),
|
||||
('https://example.com/article', None), ('https://example.com:bad', None), ('proxies: []', None), ('proxies: [', None),
|
||||
('<html>error vmess://example</html>', None), ('vmess://example\n<html>error</html>', None),
|
||||
('proxies:\n - broken', None), ('', None)]:
|
||||
with self.subTest(text=text):
|
||||
self.assertEqual(main._detect_kind(text), expected)
|
||||
|
||||
def test_download_uses_tls_and_skips_invalid_content(self):
|
||||
session = Mock()
|
||||
session.get.side_effect = [Mock(status_code=200, text='<html>error</html>'),
|
||||
Mock(status_code=200, text=CLASH), Mock(status_code=200, text=NODES)]
|
||||
with patch.object(main, 'write_log'):
|
||||
found = main._classify_subscriptions(session, ['https://example.com/error', 'https://example.com/c', 'https://example.com/v'])
|
||||
self.assertEqual(set(found), {'clash', 'v2ray'})
|
||||
self.assertTrue(all(call.kwargs == {'timeout': 20} for call in session.get.call_args_list))
|
||||
|
||||
def test_collection_and_failed_source_keeps_cache(self):
|
||||
feed = '<rss><channel><item><description>https://example.com/c https://example.com/v</description></item></channel></rss>'
|
||||
session = Mock()
|
||||
session.get.side_effect = [Mock(status_code=200, text=feed, content=feed.encode()),
|
||||
Mock(status_code=200, text=CLASH, content=CLASH.encode()),
|
||||
Mock(status_code=200, text=NODES)]
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
previous = os.getcwd(); os.chdir(directory)
|
||||
try:
|
||||
with patch.object(main, '_build_session', return_value=nullcontext(session)), patch.object(main, 'DIRECT_SOURCES', {}):
|
||||
self.assertEqual(main.main(), 0)
|
||||
self.assertEqual(yaml.safe_load(Path('subscribe/clash.yml').read_text()), yaml.safe_load(CLASH))
|
||||
self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES)
|
||||
session.get.side_effect = [Mock(raise_for_status=Mock(side_effect=main.requests.HTTPError))]
|
||||
with patch.object(main, '_build_session', return_value=nullcontext(session)), patch.object(main, 'DIRECT_SOURCES', {}):
|
||||
self.assertEqual(main.main(), 1)
|
||||
self.assertEqual(yaml.safe_load(Path('subscribe/clash.yml').read_text()), yaml.safe_load(CLASH))
|
||||
self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES)
|
||||
finally:
|
||||
os.chdir(previous)
|
||||
|
||||
def test_merge_deduplicates_and_keeps_groups_resolvable(self):
|
||||
first = CLASH + 'proxy-groups:\n - {name: select, type: select, proxies: [test, DIRECT]}\nrules: ["MATCH,select"]\n'
|
||||
duplicate = CLASH.replace('name: test', 'name: another')
|
||||
collision = CLASH.replace('example.com', 'other.example')
|
||||
data = yaml.safe_load(main._merge_clash([('original', first), ('NoMoreWalls', duplicate), ('ProxyPool', collision)]))
|
||||
names = [p['name'] for p in data['proxies']]
|
||||
self.assertEqual(names, ['test', 'ProxyPool | test'])
|
||||
self.assertEqual(data['proxy-groups'][0]['proxies'], ['test', 'DIRECT', 'ProxyPool | test'])
|
||||
self.assertEqual(data['rules'], ['MATCH,select'])
|
||||
|
||||
def test_direct_sources_work_when_rss_fails(self):
|
||||
session = Mock()
|
||||
session.get.side_effect = [main.requests.ConnectionError,
|
||||
Mock(status_code=200, text=CLASH), Mock(status_code=200, text=NODES),
|
||||
Mock(status_code=200, text=CLASH), Mock(status_code=200, text=base64.b64encode(NODES.encode()).decode())]
|
||||
with tempfile.TemporaryDirectory() as directory:
|
||||
previous = os.getcwd(); os.chdir(directory)
|
||||
try:
|
||||
with patch.object(main, '_build_session', return_value=nullcontext(session)):
|
||||
self.assertEqual(main.main(), 0)
|
||||
self.assertEqual(len(yaml.safe_load(Path('subscribe/clash.yml').read_text())['proxies']), 1)
|
||||
self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES)
|
||||
self.assertEqual(session.get.call_count, 5)
|
||||
finally:
|
||||
os.chdir(previous)
|
||||
|
||||
def test_github_empty_history_and_http_errors(self):
|
||||
from get_projaec_info import get_project_info
|
||||
with patch('get_projaec_info.requests.get', return_value=Mock(json=Mock(return_value=[]))):
|
||||
self.assertEqual(get_project_info('u','p','star','stargazers','starred_at')['num_list'], [0])
|
||||
with patch('get_projaec_info.requests.get', return_value=Mock(raise_for_status=Mock(side_effect=main.requests.HTTPError))):
|
||||
with self.assertRaises(main.requests.HTTPError):
|
||||
get_project_info('u','p','star','stargazers','starred_at')
|
||||
Loading…
Reference in New Issue
Block a user