From ac1ac622c8f6d5cffb3e124bb483b42132bfd56c Mon Sep 17 00:00:00 2001 From: ermaozi Date: Tue, 8 Sep 2026 07:37:26 +0800 Subject: [PATCH] feat: merge public subscription sources and classify TCP connectivity --- .github/workflows/clear.yml | 7 +- .github/workflows/get_projaec_info.yml | 34 ++-- .github/workflows/main.yml | 36 ++-- .gitignore | 5 +- README.md | 42 ++++ check_nodes.py | 123 ++++++++++++ get_projaec_info.py | 13 +- main.py | 260 +++++++++++++------------ requirements.txt | 9 +- tests/test_health.py | 60 ++++++ tests/test_subscriptions.py | 89 +++++++++ 11 files changed, 509 insertions(+), 169 deletions(-) create mode 100644 check_nodes.py create mode 100644 tests/test_health.py create mode 100644 tests/test_subscriptions.py diff --git a/.github/workflows/clear.yml b/.github/workflows/clear.yml index 57e2002..7a11e67 100644 --- a/.github/workflows/clear.yml +++ b/.github/workflows/clear.yml @@ -3,11 +3,16 @@ on: workflow_dispatch: schedule: - cron: '0 2 * * 1' +permissions: + contents: write +concurrency: + group: repository-updates + cancel-in-progress: false jobs: build: runs-on: ubuntu-latest steps: - - uses: actions/checkout@v2 + - uses: actions/checkout@v7 - name: 删除提交记录 run: | git config --global user.name "ermaozi" diff --git a/.github/workflows/get_projaec_info.yml b/.github/workflows/get_projaec_info.yml index 093c331..a4f6ed1 100644 --- a/.github/workflows/get_projaec_info.yml +++ b/.github/workflows/get_projaec_info.yml @@ -3,36 +3,40 @@ on: workflow_dispatch: schedule: - cron: '0 */12 * * *' +permissions: + contents: write +concurrency: + group: repository-updates + cancel-in-progress: false +env: + TZ: Asia/Shanghai jobs: deploy: runs-on: ubuntu-latest + timeout-minutes: 20 steps: - name: 迁出代码 - uses: actions/checkout@v2 + uses: actions/checkout@v7 - name: 安装Python - uses: actions/setup-python@v2 + uses: actions/setup-python@v7 with: - python-version: '3.11.6' - - name: 加载缓存 - uses: actions/cache@v3 - with: - path: ~/.cache/pip - key: ${{ runner.os }}-pip-${{ hashFiles('**/run_in_Actions/requirements.txt') }} - restore-keys: | - ${{ runner.os }}-pip- - - name: 设置时区 - run: sudo timedatectl set-timezone 'Asia/Shanghai' + python-version: '3.12' + cache: pip + cache-dependency-path: requirements.txt - name: 安装依赖 run: | - pip install -r requirements.txt + python -m pip install -r requirements.txt - name: 执行任务 + env: + GH_TOKEN: ${{ github.token }} run: | - python get_projaec_info.py --user ermaozi --project get_subscribe --save_path mail/project_info.svg --theme dark --token ${{ secrets.TOKEN }} + python get_projaec_info.py --user ermaozi --project get_subscribe --save_path mail/project_info.svg --theme dark - name: 提交更改 run: | git config core.ignorecase false git config --local user.email "admin@ermao.net" git config --local user.name "ermaozi" - git add . + git add mail/project_info.svg + git diff --cached --quiet && exit 0 git commit -m "更新项目信息" git push diff --git a/.github/workflows/main.yml b/.github/workflows/main.yml index 2b93424..77761dd 100644 --- a/.github/workflows/main.yml +++ b/.github/workflows/main.yml @@ -5,40 +5,46 @@ on: workflow_dispatch: # 定时触发 schedule: - # 每6小时获取一次 + # 每12小时获取一次 - cron: '0 */12 * * *' +permissions: + contents: write +concurrency: + group: repository-updates + cancel-in-progress: false +env: + TZ: Asia/Shanghai jobs: deploy: runs-on: ubuntu-latest + timeout-minutes: 20 steps: - name: 迁出代码 - uses: actions/checkout@v2 + uses: actions/checkout@v7 - name: 安装Python - uses: actions/setup-python@v4 + uses: actions/setup-python@v7 with: - python-version: '3.10' - - name: 加载缓存 - uses: actions/cache@v3 - with: - path: ~/.cache/pip - key: ${{ runner.os }}-pip-${{ hashFiles('**/run_in_Actions/requirements.txt') }} - restore-keys: | - ${{ runner.os }}-pip- - - name: 设置时区 - run: sudo timedatectl set-timezone 'Asia/Shanghai' + python-version: '3.12' + cache: pip + cache-dependency-path: requirements.txt - name: 安装依赖 run: | - pip install -r requirements.txt + python -m pip install -r requirements.txt + - name: 回归检查 + run: python -m unittest discover -s tests - name: 执行任务 run: | python main.py + - name: TCP 连通性分类 + run: python check_nodes.py - name: 提交更改 run: | git config core.ignorecase false git config --local user.email "admin@ermao.net" git config --local user.name "ermaozi" - git add . + git add subscribe log + git diff --cached --quiet && exit 0 git commit -m "$(date '+%Y-%m-%d %H:%M:%S')更新订阅链接" git push diff --git a/.gitignore b/.gitignore index eba74f4..1e248ad 100644 --- a/.gitignore +++ b/.gitignore @@ -1 +1,4 @@ -venv/ \ No newline at end of file +venv/ +.venv/ +__pycache__/ +*.pyc diff --git a/README.md b/README.md index a1e8169..6a46261 100644 --- a/README.md +++ b/README.md @@ -53,3 +53,45 @@ https://www.ermao.net/sub/v2ray/ermao.net 我搜罗的一些比较便宜好用的机场,觉得免费订阅不好使的朋友们可以在这里面找找。 [https://www.ermao.net/posts/vpn](https://www.ermao.net/posts/vpn) + +## 采集与发布架构 + +本仓库通过 GitHub Actions 每 12 小时采集长风分享、 +[NoMoreWalls](https://github.com/peasoft/NoMoreWalls) 和 +[ProxyPool](https://github.com/snakem982/proxypool) 的公开订阅。 +BestClash 不在采集来源中。 + +Clash 按节点配置去重(忽略名称),重名节点自动改名,并加入原有节点选择组; +保留第一个可用来源的分流规则。V2Ray 兼容明文和 Base64 来源,按完整 URI 去重, +统一保存为主项目现有的明文节点列表。一个来源失败会继续其他来源, +缺少任一种有效订阅时保留原有文件并令任务失败。格式检查不代表节点连通或速度保证。 + +当前公开 `/sub/` 地址由独立 Cloudflare Worker `sub` 提供,Worker 自行采集并写入 R2, +不是从本仓库读取文件。因此本仓库增加来源不会自动同步到线上 Worker; +GitHub 版本可通过仓库中的 `subscribe/clash.yml` 和 `subscribe/v2ray.txt` 获取。 + +使用 Python 3.12 执行本地检查: + +```sh +python -m pip install -r requirements.txt +python -m unittest discover -s tests -v +python main.py +``` + +保留 `SUBSCRIBE_PROXY` 代理环境变量。GitHub 工作流使用内置 `GITHUB_TOKEN`, +无需额外的 `TOKEN` Secret。写入工作流共享并发组,避免同时更新分支; +原有每周清理提交历史行为保持不变。频繁手动触发时,GitHub 可能替换仍在排队的任务。 + + +## 连通性分类 + +每次自动采集后运行 `python check_nodes.py`,保留 `subscribe/` 下的完整订阅, +另生成以下三个目录(各含 `clash.yml` 和 `v2ray.txt`): + +- `subscribe/reachable/`:从本次运行机器可以建立 TCP 连接的节点。 +- `subscribe/unreachable/`:TCP 连接或 DNS 解析失败的节点。 +- `subscribe/untested/`:UDP/QUIC 协议、无法解析地址或非公网地址,不判定为不可用。 + +`subscribe/health.json` 保存检测时间和分类数量。同一主机端口只检查一次, +最多并发 24 个端点,每个端点的 TCP 连接总预算为 3 秒(不含系统 DNS 解析时间)。 +这不是代理认证、出口访问或速度测试;可连接不等于代理可用,失败也可能是运行机器的网络限制。 diff --git a/check_nodes.py b/check_nodes.py new file mode 100644 index 0000000..99344a5 --- /dev/null +++ b/check_nodes.py @@ -0,0 +1,123 @@ +"""按 TCP 握手结果分类订阅;UDP/QUIC 和无法解析的格式不作失效判断。""" +import base64 +from concurrent.futures import ThreadPoolExecutor +from datetime import datetime, timezone +import ipaddress +import json +from pathlib import Path +import socket +import time +from urllib.parse import parse_qs, urlsplit + +import yaml + +TIMEOUT = 3 +WORKERS = 24 +STATUSES = ('reachable', 'unreachable', 'untested') +UDP_TYPES = {'hysteria', 'hysteria2', 'hy2', 'tuic', 'wireguard', 'mieru'} + + +def _decode(value): + return base64.urlsafe_b64decode(value + '=' * (-len(value) % 4)).decode('utf-8') + + +def endpoint(node): + """返回 TCP 主机和端口;不支持或使用 UDP 的节点返回 None。""" + try: + if isinstance(node, dict): + if node.get('type') in UDP_TYPES or node.get('network') in {'quic', 'kcp'}: + return None + host, port = node.get('server'), node.get('port') + else: + scheme, body = node.split('://', 1) + if scheme in UDP_TYPES: + return None + if scheme == 'vmess': + data = json.loads(_decode(body.split('#', 1)[0])) + if not isinstance(data, dict) or data.get('net') in {'quic', 'kcp'}: + return None + host, port = data.get('add'), data.get('port') + elif scheme == 'ssr': + host, port, *_ = _decode(body.split('#', 1)[0]).split('/?', 1)[0].rsplit(':', 5) + else: + if scheme == 'ss' and '@' not in body: + node = 'ss://' + _decode(body.split('#', 1)[0]) + parsed = urlsplit(node) + if any(v in {'quic', 'kcp'} for v in parse_qs(parsed.query).get('type', [])): + return None + host, port = parsed.hostname, parsed.port + port = int(port) + if not isinstance(host, str) or not host or not 1 <= port <= 65535: + return None + return host.strip('[]').lower().rstrip('.'), port + except (ValueError, TypeError, KeyError, UnicodeError): + return None + + +def tcp_status(address): + # ponytail: 只测 TCP 握手;需要验证认证和代理出口时再引入代理内核。 + host, port = address + try: + resolved = socket.getaddrinfo(host, port, type=socket.SOCK_STREAM) + except OSError: + return 'unreachable' + public = [] + for family, socktype, proto, _, sockaddr in resolved: + # 来源不可信,不探测宿主机内网、回环或云实例元数据地址。 + if ipaddress.ip_address(sockaddr[0]).is_global: + public.append((family, socktype, proto, sockaddr)) + if not public: + return 'untested' + deadline = time.monotonic() + TIMEOUT + for family, socktype, proto, sockaddr in public: + remaining = deadline - time.monotonic() + if remaining <= 0: + break + try: + with socket.socket(family, socktype, proto) as connection: + connection.settimeout(remaining) + connection.connect(sockaddr) + return 'reachable' + except OSError: + continue + return 'unreachable' + + +def classify(directory='subscribe'): + directory = Path(directory) + config = yaml.safe_load((directory / 'clash.yml').read_text(encoding='utf-8')) + uris = (directory / 'v2ray.txt').read_text(encoding='utf-8').splitlines() + proxies = config['proxies'] + proxy_addresses = [endpoint(proxy) for proxy in proxies] + uri_addresses = [endpoint(uri) for uri in uris] + addresses = list(dict.fromkeys(a for a in proxy_addresses + uri_addresses if a is not None)) + with ThreadPoolExecutor(max_workers=WORKERS) as pool: + results = dict(zip(addresses, pool.map(tcp_status, addresses))) + summary = {'checked_at': datetime.now(timezone.utc).isoformat(), 'method': 'TCP connect', + 'timeout_seconds': TIMEOUT, 'unique_endpoints': len(addresses), 'counts': {}} + all_names = {proxy['name'] for proxy in proxies} + for status in STATUSES: + selected = [p for p, address in zip(proxies, proxy_addresses) if results.get(address, 'untested') == status] + selected_uris = [uri for uri, address in zip(uris, uri_addresses) if results.get(address, 'untested') == status] + removed = all_names - {proxy['name'] for proxy in selected} + filtered = {**config, 'proxies': selected} + groups = [] + for group in config.get('proxy-groups', []): + group = dict(group) + if 'proxies' in group: + group['proxies'] = [name for name in group['proxies'] if name not in removed] or ['REJECT'] + groups.append(group) + if 'proxy-groups' in config: + filtered['proxy-groups'] = groups + target = directory / status + target.mkdir(exist_ok=True) + (target / 'clash.yml').write_text(yaml.safe_dump(filtered, allow_unicode=True, sort_keys=False), encoding='utf-8') + (target / 'v2ray.txt').write_text('\n'.join(selected_uris) + ('\n' if selected_uris else ''), encoding='utf-8') + summary['counts'][status] = {'clash': len(selected), 'v2ray': len(selected_uris)} + (directory / 'health.json').write_text(json.dumps(summary, ensure_ascii=False, indent=2) + '\n', encoding='utf-8') + print(json.dumps(summary, ensure_ascii=False)) + return summary + + +if __name__ == '__main__': + classify() diff --git a/get_projaec_info.py b/get_projaec_info.py index ee8236d..871650c 100644 --- a/get_projaec_info.py +++ b/get_projaec_info.py @@ -1,4 +1,5 @@ import argparse +import os import re import matplotlib.pyplot as plt @@ -19,11 +20,12 @@ def get_project_info(user, project, name, item, date_key, token=""): }) data_list = [] page = 0 - date_pat = re.compile("\d{4}-\d{2}-\d{2}") + date_pat = re.compile(r"\d{4}-\d{2}-\d{2}") while True: page += 1 - url = f"https://api.github.com/repos/{user}/{project}/{item}?page={page}" - req = requests.get(url, headers=header) + url = f"https://api.github.com/repos/{user}/{project}/{item}?page={page}&per_page=100" + req = requests.get(url, headers=header, timeout=30) + req.raise_for_status() datas = req.json() if not datas: break @@ -31,7 +33,7 @@ def get_project_info(user, project, name, item, date_key, token=""): date_dic = {} - start_date = min(data_list) + start_date = min(data_list) if data_list else date.today().isoformat() end_date = date.today() for date_str in data_list: if not date_dic.get(date_str): @@ -92,6 +94,7 @@ def create_svg(project, datas, save_path, theme=""): ax.grid(True, linestyle='-.') plt.savefig(save_path) + plt.close(fig) def main(user, project, save_path, theme="", token=""): @@ -108,6 +111,6 @@ if __name__ == "__main__": parser.add_argument("--project", type=str) parser.add_argument("--save_path", type=str) parser.add_argument("--theme", type=str, default="") - parser.add_argument("--token", type=str, default="") + parser.add_argument("--token", type=str, default=os.environ.get("GH_TOKEN", "")) args = parser.parse_args() main(args.user, args.project, args.save_path, args.theme, args.token) diff --git a/main.py b/main.py index c538d30..b6f8d66 100644 --- a/main.py +++ b/main.py @@ -1,33 +1,39 @@ import base64 import os import re -import smtplib import sys import time import html +import json +from pathlib import Path +from urllib.parse import urlsplit import xml.etree.ElementTree as ET -from email.mime.text import MIMEText -from email.utils import formataddr import requests +import yaml from requests.adapters import HTTPAdapter from urllib3.util.retry import Retry -requests.packages.urllib3.disable_warnings() -ok_code = [200, 201, 202, 203, 204, 205, 206] - -# 邮箱域名过滤列表 -blackhole_list = ["cnr.cn", "cyberpolice.cn", "gov.cn", "samr.gov.cn", "12321.cn" - "miit.gov.cn", "chinatcc.gov.cn"] - +ok_code = [200] +DIRECT_SOURCES = { + "NoMoreWalls": [ + "https://raw.githubusercontent.com/peasoft/NoMoreWalls/master/list.meta.yml", + "https://raw.githubusercontent.com/peasoft/NoMoreWalls/master/list.txt", + ], + "ProxyPool": [ + "https://raw.githubusercontent.com/snakem982/proxypool/main/source/clash-meta-2.yaml", + "https://raw.githubusercontent.com/snakem982/proxypool/main/source/v2ray-2.txt", + ], +} def write_log(content, level="INFO"): date_str = time.strftime('%Y-%m-%d %H:%M:%S', time.localtime(time.time())) update_log = f"[{date_str}] [{level}] {content}\n" - print(update_log) + print(update_log, end="") + Path("log").mkdir(exist_ok=True) with open(f'./log/{time.strftime("%Y-%m", time.localtime(time.time()))}-update.log', 'a', encoding="utf-8") as f: f.write(update_log) @@ -42,7 +48,7 @@ def _extract_urls(summary): return urls, decoded -_NODE_SCHEME_RE = re.compile(r"(?:vmess|vless|trojan|ss|ssr|hysteria2?|tuic)://") +_NODE_SCHEME_RE = re.compile(r"(?:vmess|vless|trojan|ss|ssr|hysteria2?|hy2|tuic|anytls|mieru|https?|socks[45]?)://") def _b64decode(text): @@ -51,10 +57,13 @@ def _b64decode(text): return "" try: # binascii.Error 是 ValueError 的子类,统一捕获即可 - raw = base64.b64decode(compact + "=" * (-len(compact) % 4)) + raw = base64.b64decode(compact + "=" * (-len(compact) % 4), validate=True) except ValueError: return "" - return raw.decode("utf-8", "ignore") + try: + return raw.decode("utf-8") + except UnicodeDecodeError: + return "" def _detect_kind(text): @@ -62,39 +71,33 @@ def _detect_kind(text): sample = text.strip() if not sample: return None - # clash 配置为 YAML,包含 proxies/proxy-groups 字段 if re.search(r"^(?:proxies|proxy-groups)\s*:", sample, re.MULTILINE): - return "clash" - # v2ray 订阅为节点 URI 列表,可能是明文或 base64 编码 - if _NODE_SCHEME_RE.search(sample): - return "v2ray" - decoded = _b64decode(sample) - if decoded and _NODE_SCHEME_RE.search(decoded): + try: + data = yaml.safe_load(sample) + except yaml.YAMLError: + return None + proxies = data.get("proxies") if isinstance(data, dict) else None + if isinstance(proxies, list) and proxies and all( + isinstance(proxy, dict) and isinstance(proxy.get("name"), str) and proxy["name"] + and proxy.get("type") and proxy.get("server") for proxy in proxies + ): + return "clash" + return None + decoded = sample if _NODE_SCHEME_RE.match(sample) else _b64decode(sample) + nodes = [node.strip() for node in decoded.splitlines() if node.strip()] + if nodes and all(_NODE_SCHEME_RE.match(node) and not re.search(r"\s", node) for node in nodes): + try: + for node in nodes: + if node.startswith(("http://", "https://", "socks://", "socks4://", "socks5://")): + parsed = urlsplit(node) + if not parsed.hostname or not parsed.port: + return None + except ValueError: + return None return "v2ray" return None -def _download_with_retry(urls): - if not urls: - return None, None - for url in urls: - for _ in range(3): - try: - req = requests.request( - "GET", - url, - verify=False, - timeout=20, - headers={"User-Agent": "Mozilla/5.0"}, - ) - except requests.RequestException as e: - print(f"请求 {url} 失败: {e}") - continue - if req.status_code in ok_code: - return req, url - return None, urls[0] - - def _build_session(): session = requests.Session() retry = Retry( @@ -124,106 +127,109 @@ def _classify_subscriptions(session, urls): if "v2ray" in found and "clash" in found: break try: - req = session.get(url, verify=False, timeout=20) + req = session.get(url, timeout=20) except requests.RequestException as e: - write_log(f"请求失败:{url} - {e}", "WARN") + write_log(f"候选订阅请求失败:{type(e).__name__}", "WARN") continue if req.status_code not in ok_code: - write_log(f"请求失败:{url} - {req.status_code}", "WARN") + write_log(f"候选订阅 HTTP {req.status_code}", "WARN") continue kind = _detect_kind(req.text) if kind and kind not in found: found[kind] = (req, url) - write_log(f"识别到 {kind} 订阅:{url}", "INFO") + write_log(f"识别到有效的 {kind} 订阅", "INFO") return found +def _merge_clash(sources): + # 保留第一个可用来源的规则,将所有来源节点加入已有的节点选择组。 + config = yaml.safe_load(sources[0][1]) + proxies, identities, names, renamed = [], {}, set(), {} + names.update(group["name"] for group in config.get("proxy-groups", [])) + names.update(["DIRECT", "REJECT", "REJECT-DROP", "PASS", "COMPATIBLE", "GLOBAL"]) + for index, (source, content) in enumerate(sources): + for proxy in yaml.safe_load(content)["proxies"]: + identity = json.dumps({k: v for k, v in proxy.items() if k != "name"}, sort_keys=True, default=str) + if identity not in identities: + name = proxy["name"] + if name in names: + name = f"{source} | {name}" + original = name + suffix = 2 + while name in names: + name = f"{original} ({suffix})" + suffix += 1 + identities[identity] = name + names.add(name) + proxies.append({**proxy, "name": name}) + if index == 0: + renamed[proxy["name"]] = identities[identity] + for group in config.get("proxy-groups", []): + members = group.get("proxies", []) + if any(member in renamed for member in members): + group["proxies"] = list(dict.fromkeys( + [renamed.get(member, member) for member in members] + [p["name"] for p in proxies] + )) + config["proxies"] = proxies + return yaml.safe_dump(config, allow_unicode=True, sort_keys=False) + + def get_subscribe_url(): - dirs = './subscribe' - if not os.path.exists(dirs): - os.makedirs(dirs) - log_dir = "./log" - if not os.path.exists(log_dir): - os.makedirs(log_dir) + collected = {"clash": [], "v2ray": []} + with _build_session() as session: + try: + rss = session.get('https://www.cfmem.com/feeds/posts/default?alt=rss', timeout=20) + rss.raise_for_status() + root = ET.fromstring(rss.content) + for item in root.findall("./channel/item"): + urls, _ = _extract_urls(item.findtext("description") or "") + found = _classify_subscriptions(session, urls) + if found: + for kind, (response, _) in found.items(): + collected[kind].append(("长风分享", response.text)) + break + except (requests.RequestException, ET.ParseError) as exc: + write_log(f"长风分享采集失败:{type(exc).__name__},继续其他来源", "WARN") + for source, urls in DIRECT_SOURCES.items(): + found = _classify_subscriptions(session, urls) + for kind, (response, _) in found.items(): + collected[kind].append((source, response.text)) + write_log(f"{source}:获取 {', '.join(found) or '无有效订阅'}") - update_list = [] - session = _build_session() - try: - rss_req = session.get( - 'https://www.cfmem.com/feeds/posts/default?alt=rss', - timeout=20, - ) - except requests.RequestException as ex: - write_log(f"更新失败!拉取 RSS 异常: {ex}", "ERROR") - return - - if rss_req.status_code not in ok_code: - write_log(f"更新失败!无法拉取原网站内容 - {rss_req.status_code}", "ERROR") - return - - try: - root = ET.fromstring(rss_req.text) - except ET.ParseError as ex: - write_log(f"更新失败!RSS 解析失败: {ex}", "ERROR") - return - - item = root.find("./channel/item") - if item is None: - write_log("更新失败!RSS 中未找到可用条目", "ERROR") - return - - summary = item.findtext("description") - if not summary: - write_log("暂时没有可用的订阅更新", "WARN") - return - - urls, _ = _extract_urls(summary) - - # 链接已无固定后缀,需下载内容后再判断是 v2ray 还是 clash - classified = _classify_subscriptions(session, urls) - - # 获取普通订阅链接 - v2ray_entry = classified.get("v2ray") - if v2ray_entry: - v2ray_req, _ = v2ray_entry - update_list.append(f"v2ray: {v2ray_req.status_code}") - with open(dirs + '/v2ray.txt', 'w', encoding="utf-8") as f: - f.write(v2ray_req.text) - else: - cache_file = dirs + '/v2ray.txt' - if os.path.exists(cache_file) and os.path.getsize(cache_file) > 0: - update_list.append("v2ray: cache") - write_log("未获取到 v2ray 订阅,已保留本地缓存", "WARN") - else: - write_log("未获取到 v2ray 订阅", "WARN") - - # 获取clash订阅链接 - clash_entry = classified.get("clash") - if clash_entry: - clash_req, _ = clash_entry - update_list.append(f"clash: {clash_req.status_code}") - with open(dirs + '/clash.yml', 'w', encoding="utf-8") as f: - f.write(clash_req.content.decode("utf-8")) - else: - cache_file = dirs + '/clash.yml' - if os.path.exists(cache_file) and os.path.getsize(cache_file) > 0: - update_list.append("clash: cache") - write_log("未获取到 clash 订阅,已保留本地缓存", "WARN") - else: - write_log("未获取到 clash 订阅", "WARN") - if update_list: - file_pat = re.compile(r"v2ray\.txt|clash\.yml") - if file_pat.search(os.popen("git status").read()): - write_log(f"更新成功:{update_list}", "INFO") - else: - write_log(f"订阅暂未更新", "WARN") - else: - write_log(f"未能获取新的更新内容", "WARN") + if not all(collected.values()): + write_log("未获取到完整的两种订阅,保留旧文件", "ERROR") + return False + nodes = [] + for _, content in collected["v2ray"]: + text = content.strip() if _NODE_SCHEME_RE.match(content.strip()) else _b64decode(content) + nodes.extend(node.strip() for node in text.splitlines() if node.strip()) + # 保持主项目原有的明文 URI 输出,兼容现有订阅入口。 + outputs = {"clash.yml": _merge_clash(collected["clash"]), + "v2ray.txt": "\n".join(dict.fromkeys(nodes)) + "\n"} + for filename, content in outputs.items(): + expected = "clash" if filename.endswith(".yml") else "v2ray" + if _detect_kind(content) != expected: + raise ValueError(f"合并后的 {expected} 订阅格式无效") + Path("subscribe").mkdir(exist_ok=True) + changed = [] + for filename, content in outputs.items(): + target = Path("subscribe") / filename + if not target.exists() or target.read_text(encoding="utf-8") != content: + temporary = target.with_suffix(target.suffix + ".tmp") + temporary.write_text(content, encoding="utf-8") + temporary.replace(target) + changed.append(filename) + write_log("已更新:" + ", ".join(changed) if changed else "订阅内容未变化") + return True def main(): - get_subscribe_url() + return 0 if get_subscribe_url() else 1 # 主函数入口 if __name__ == '__main__': - main() + try: + sys.exit(main()) + except (OSError, ValueError, yaml.YAMLError) as exc: + write_log(f"采集或写入失败:{type(exc).__name__}", "ERROR") + sys.exit(1) diff --git a/requirements.txt b/requirements.txt index c55719b..5d7b24c 100644 --- a/requirements.txt +++ b/requirements.txt @@ -1,5 +1,4 @@ -requests -feedparser -matplotlib -pandas - +requests==2.34.2 +matplotlib==3.10.9 +pandas==2.3.3 +PyYAML==6.0.3 diff --git a/tests/test_health.py b/tests/test_health.py new file mode 100644 index 0000000..528a78f --- /dev/null +++ b/tests/test_health.py @@ -0,0 +1,60 @@ +import base64 +import json +from pathlib import Path +import socket +import tempfile +import unittest +from unittest.mock import Mock, patch + +import yaml +import check_nodes + + +class HealthTests(unittest.TestCase): + def test_endpoints_and_udp_are_not_misclassified(self): + vmess = base64.b64encode(json.dumps({'add': 'example.com', 'port': '443'}).encode()).decode() + ss = base64.b64encode(b'aes-256-gcm:password@example.com:8443').decode() + ssr = base64.urlsafe_b64encode(b'example.com:443:origin:aes-256-cfb:plain:cGFzcw/?x=y').decode() + for node, expected in [({'type': 'ss', 'server': 'example.com', 'port': 443}, ('example.com',443)), + ('vmess://' + vmess, ('example.com',443)), ('ss://' + ss, ('example.com',8443)), + ('ssr://' + ssr, ('example.com',443)), ('vless://id@[2606:4700:4700::1111]:443', ('2606:4700:4700::1111',443)), + ('hysteria2://password@example.com:443',None), + ({'type':'tuic', 'server':'example.com', 'port':443},None), + ('vless://id@example.com:443?type=quic',None), ('vmess://invalid',None)]: + with self.subTest(node=node): + self.assertEqual(check_nodes.endpoint(node), expected) + + def test_private_addresses_are_not_connected(self): + with patch.object(socket, 'getaddrinfo', return_value=[(socket.AF_INET,socket.SOCK_STREAM,6,'',('127.0.0.1',443))]), patch.object(socket, 'socket') as connect: + self.assertEqual(check_nodes.tcp_status(('local.example',443)), 'untested') + connect.assert_not_called() + with patch.object(socket, 'getaddrinfo', side_effect=socket.gaierror): + self.assertEqual(check_nodes.tcp_status(('bad.example',443)), 'unreachable') + + def test_tcp_success_and_timeout(self): + resolved = [(socket.AF_INET,socket.SOCK_STREAM,6,'',('1.1.1.1',443))] + with patch.object(socket, 'getaddrinfo', return_value=resolved), patch.object(socket, 'socket') as factory: + self.assertEqual(check_nodes.tcp_status(('public.example',443)), 'reachable') + factory.return_value.__enter__.return_value.connect.assert_called_once_with(('1.1.1.1',443)) + factory.return_value.__enter__.return_value.connect.side_effect = socket.timeout + self.assertEqual(check_nodes.tcp_status(('public.example',443)), 'unreachable') + + def test_classification_is_complete_and_keeps_original(self): + config = {'proxies': [ + {'name':'up','type':'ss','server':'up.example','port':443}, + {'name':'down','type':'ss','server':'down.example','port':443}, + {'name':'udp','type':'tuic','server':'udp.example','port':443}], + 'proxy-groups':[{'name':'select','type':'select','proxies':['up','down','udp']}]} + uris = 'ss://YWVzOnBhc3M=@up.example:443\ntuic://id@udp.example:443\n' + with tempfile.TemporaryDirectory() as directory: + path=Path(directory);source=yaml.safe_dump(config) + (path/'clash.yml').write_text(source);(path/'v2ray.txt').write_text(uris) + with patch.object(check_nodes, 'tcp_status', side_effect=lambda a: 'reachable' if a[0]=='up.example' else 'unreachable') as probe: + result=check_nodes.classify(path) + self.assertEqual(probe.call_count,2) + self.assertEqual(result['counts']['untested'],{'clash':1,'v2ray':1}) + self.assertEqual((path/'clash.yml').read_text(),source) + self.assertEqual((path/'v2ray.txt').read_text(),uris) + for status,name in [('reachable','up'),('unreachable','down'),('untested','udp')]: + output=yaml.safe_load((path/status/'clash.yml').read_text()) + self.assertEqual(output['proxy-groups'][0]['proxies'],[name]) diff --git a/tests/test_subscriptions.py b/tests/test_subscriptions.py new file mode 100644 index 0000000..263d9a1 --- /dev/null +++ b/tests/test_subscriptions.py @@ -0,0 +1,89 @@ +import base64 +from contextlib import nullcontext +import yaml +import os +from pathlib import Path +import tempfile +import unittest +from unittest.mock import Mock, patch + +import main + +CLASH = 'proxies:\n - {name: test, type: ss, server: example.com, port: 443}\n' +NODES = 'vmess://example\nvless://example' + + +class SubscriptionTests(unittest.TestCase): + def test_real_formats_and_error_pages(self): + for text, expected in [(CLASH, 'clash'), (NODES, 'v2ray'), + (base64.b64encode(NODES.encode()).decode().rstrip('='), 'v2ray'), + ('https://example.com:443#proxy', 'v2ray'), ('socks5://example.com:1080', 'v2ray'), + ('https://example.com/article', None), ('https://example.com:bad', None), ('proxies: []', None), ('proxies: [', None), + ('error vmess://example', None), ('vmess://example\nerror', None), + ('proxies:\n - broken', None), ('', None)]: + with self.subTest(text=text): + self.assertEqual(main._detect_kind(text), expected) + + def test_download_uses_tls_and_skips_invalid_content(self): + session = Mock() + session.get.side_effect = [Mock(status_code=200, text='error'), + Mock(status_code=200, text=CLASH), Mock(status_code=200, text=NODES)] + with patch.object(main, 'write_log'): + found = main._classify_subscriptions(session, ['https://example.com/error', 'https://example.com/c', 'https://example.com/v']) + self.assertEqual(set(found), {'clash', 'v2ray'}) + self.assertTrue(all(call.kwargs == {'timeout': 20} for call in session.get.call_args_list)) + + def test_collection_and_failed_source_keeps_cache(self): + feed = 'https://example.com/c https://example.com/v' + session = Mock() + session.get.side_effect = [Mock(status_code=200, text=feed, content=feed.encode()), + Mock(status_code=200, text=CLASH, content=CLASH.encode()), + Mock(status_code=200, text=NODES)] + with tempfile.TemporaryDirectory() as directory: + previous = os.getcwd(); os.chdir(directory) + try: + with patch.object(main, '_build_session', return_value=nullcontext(session)), patch.object(main, 'DIRECT_SOURCES', {}): + self.assertEqual(main.main(), 0) + self.assertEqual(yaml.safe_load(Path('subscribe/clash.yml').read_text()), yaml.safe_load(CLASH)) + self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES) + session.get.side_effect = [Mock(raise_for_status=Mock(side_effect=main.requests.HTTPError))] + with patch.object(main, '_build_session', return_value=nullcontext(session)), patch.object(main, 'DIRECT_SOURCES', {}): + self.assertEqual(main.main(), 1) + self.assertEqual(yaml.safe_load(Path('subscribe/clash.yml').read_text()), yaml.safe_load(CLASH)) + self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES) + finally: + os.chdir(previous) + + def test_merge_deduplicates_and_keeps_groups_resolvable(self): + first = CLASH + 'proxy-groups:\n - {name: select, type: select, proxies: [test, DIRECT]}\nrules: ["MATCH,select"]\n' + duplicate = CLASH.replace('name: test', 'name: another') + collision = CLASH.replace('example.com', 'other.example') + data = yaml.safe_load(main._merge_clash([('original', first), ('NoMoreWalls', duplicate), ('ProxyPool', collision)])) + names = [p['name'] for p in data['proxies']] + self.assertEqual(names, ['test', 'ProxyPool | test']) + self.assertEqual(data['proxy-groups'][0]['proxies'], ['test', 'DIRECT', 'ProxyPool | test']) + self.assertEqual(data['rules'], ['MATCH,select']) + + def test_direct_sources_work_when_rss_fails(self): + session = Mock() + session.get.side_effect = [main.requests.ConnectionError, + Mock(status_code=200, text=CLASH), Mock(status_code=200, text=NODES), + Mock(status_code=200, text=CLASH), Mock(status_code=200, text=base64.b64encode(NODES.encode()).decode())] + with tempfile.TemporaryDirectory() as directory: + previous = os.getcwd(); os.chdir(directory) + try: + with patch.object(main, '_build_session', return_value=nullcontext(session)): + self.assertEqual(main.main(), 0) + self.assertEqual(len(yaml.safe_load(Path('subscribe/clash.yml').read_text())['proxies']), 1) + self.assertEqual(Path('subscribe/v2ray.txt').read_text().strip(), NODES) + self.assertEqual(session.get.call_count, 5) + finally: + os.chdir(previous) + + def test_github_empty_history_and_http_errors(self): + from get_projaec_info import get_project_info + with patch('get_projaec_info.requests.get', return_value=Mock(json=Mock(return_value=[]))): + self.assertEqual(get_project_info('u','p','star','stargazers','starred_at')['num_list'], [0]) + with patch('get_projaec_info.requests.get', return_value=Mock(raise_for_status=Mock(side_effect=main.requests.HTTPError))): + with self.assertRaises(main.requests.HTTPError): + get_project_info('u','p','star','stargazers','starred_at')