Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
244b7c4199 | ||
|
|
a8ef887c56 | ||
|
|
8fd5c52488 | ||
|
|
32e8f8ff0b | ||
|
|
c562579ef3 | ||
|
|
6f89acbae7 | ||
|
|
88a82c7297 | ||
|
|
a8f2b0f068 | ||
|
|
0b32a6d58b | ||
|
|
893fdc500c | ||
|
|
02c266188f | ||
|
|
024e9308c9 | ||
|
|
eab1db2c3f | ||
|
|
22d7161c08 | ||
|
|
24c5a1615d | ||
|
|
4492f0599c | ||
|
|
bc2b53069a | ||
|
|
c257d6c5cc | ||
|
|
225cce5397 | ||
|
|
84d2766592 | ||
|
|
33f940c2fb | ||
|
|
a57717dabb | ||
|
|
129e31fbdc | ||
|
|
4f0618515e | ||
|
|
fd698162c9 | ||
|
|
ffa78d67bc | ||
|
|
61e51495c3 | ||
|
|
bfe71cd1dd | ||
|
|
9b3644becb | ||
|
|
0a8bdbaf46 | ||
|
|
ebe2e7807b | ||
|
|
c1e1b17a61 | ||
|
|
782770f306 | ||
|
|
8980a25a59 |
+112
-19
@@ -32,6 +32,23 @@
|
||||
# -> rolling `latest` pre-release (always-fresh feed). Publish uses the Gitea
|
||||
# API via curl (ci/gitea-release.sh) — no external action needed.
|
||||
#
|
||||
# PACKAGE VERSIONING (bug B4)
|
||||
# PKG_VERSION/PKG_RELEASE are NOT hand-written in the Makefiles any more. They
|
||||
# used to be, and nobody bumped them: v0.2.2…v0.2.6 all shipped as
|
||||
# `shaterd 0.2.0-r3` with different binaries inside, so `apk update` never saw
|
||||
# a new version and routers could not be updated at all. Now `ci/version.sh`
|
||||
# derives them from the git tag ONCE per job (the "Compute version" step,
|
||||
# exported via $GITHUB_ENV):
|
||||
# tag `vX.Y.Z` -> X.Y.Z-r1
|
||||
# anything else -> <nearest tag>-r<commits since it + 1>
|
||||
# and hands them to the SDK builds as SHATER_PKG_VERSION/SHATER_PKG_RELEASE;
|
||||
# $SHATER_VERSION (the same numbers, plus the short sha off-tag) is stamped
|
||||
# into the binary's constant.Version. ci/sdk-build*.sh then ASSERT that the
|
||||
# built .ipk/.apk really carry that version, so the failure can never be
|
||||
# silent again. This is also why both build jobs check out with fetch-depth: 0
|
||||
# — `git describe` needs tags and ancestry. `byedpi` is excluded: it keeps
|
||||
# upstream ByeDPI's own PKG_VERSION (see openwrt/byedpi/Makefile).
|
||||
#
|
||||
# APK LANE (25.12+, ADDITIVE — T2)
|
||||
# The fleet is migrating to BananaWRT 25.12-mtk-vendor (= ImmortalWrt 25.12
|
||||
# base), where opkg is replaced by Alpine apk (.apk, binary packages.adb
|
||||
@@ -70,6 +87,26 @@
|
||||
# - apt .deb archives for the apk lane's debian:bookworm host-deps
|
||||
# (.cache/apt) — key = hash of ci/sdk-build-apk.sh (the apt list is in it).
|
||||
# - usign binary (.cache/tools) — static helper, fixed key.
|
||||
# - SDK feeds/ git checkouts (.cache/feeds) — the single biggest recurring
|
||||
# cost: `scripts/feeds update -a` cloned base+packages+luci+routing+
|
||||
# telephony EVERY run (~7 min/job; github.com is ~1 MB/s from this
|
||||
# runner — run 51 evidence). The feeds dir is symlinked into the SDK
|
||||
# container from the workspace cache; `feeds update` on an existing clone
|
||||
# is a fast fetch+checkout of the pinned revs. Correctness-safe: update
|
||||
# always checks out feeds.conf's pins, and ci/sdk-build*.sh wipes the
|
||||
# cache + re-clones fresh if update ever fails on a cached checkout.
|
||||
# Key = lane + SDK release (shared across the two arch jobs of a lane —
|
||||
# same release pins identical feed revs; the sequential runner means the
|
||||
# second arch restores what the first saved). restore-keys lets an SDK
|
||||
# version bump start from the old clones (git fetch delta, not re-clone).
|
||||
# Act_runner facts this design leans on (verified in run 51 logs):
|
||||
# - the cache backend works: restores/saves confirmed, hashFiles() works;
|
||||
# - docker images (openwrt/sdk, debian:bookworm, runner-images) live on the
|
||||
# PERSISTENT host daemon — "Image is up to date" each run, no re-download;
|
||||
# - each actions/cache SAVE is followed by an exact 3-minute act_runner
|
||||
# stall (node process lingers; hit→no-save→no stall). Steady state saves
|
||||
# nothing, so adding cache entries is fine, but keys that change every
|
||||
# run (e.g. github.sha) would cost +3 min/entry/run — do NOT do that.
|
||||
|
||||
name: release
|
||||
|
||||
@@ -98,8 +135,14 @@ jobs:
|
||||
- { arch: x86_64, sdk: x86_64-24.10.4 } # testbed VM (generic x86-64)
|
||||
- { arch: aarch64_cortex-a53, sdk: mediatek-filogic-24.10.4 } # BPI-R3 + BPI-R4 (mediatek/filogic)
|
||||
steps:
|
||||
# fetch-depth: 0 — the package version is DERIVED from the git tag
|
||||
# (ci/version.sh: nearest `vX.Y.Z` + commits since it). The default
|
||||
# shallow checkout has neither tags nor ancestry, so `git describe` would
|
||||
# fail and every dispatch build would fall back to 0.0.0.
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
# scripts/build-shaterd.sh builds the engine via a go.mod
|
||||
# `replace => ./submodules/wireguard-go` (AmneziaWG fork), so that submodule
|
||||
@@ -109,6 +152,13 @@ jobs:
|
||||
- name: Init wireguard-go submodule (awg)
|
||||
run: git submodule update --init --depth 1 submodules/wireguard-go
|
||||
|
||||
# THE version step (bug B4). One computation, used by both the binary
|
||||
# (constant.Version) and the three tag-versioned packages, exported to
|
||||
# every later step of this job:
|
||||
# tag vX.Y.Z -> X.Y.Z-r1 ; off-tag -> <last tag>-r<commits+1>
|
||||
- name: Compute version from git tag
|
||||
run: bash ci/version.sh --env >> "$GITHUB_ENV"
|
||||
|
||||
# Toolchain for scripts/build-shaterd.sh: Go (daemon), Node (Vite SPA), UPX.
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
@@ -150,6 +200,21 @@ jobs:
|
||||
restore-keys: |
|
||||
dl-
|
||||
|
||||
# feeds git checkouts (see header): both 24.10.4 arch jobs share one entry
|
||||
# (same release = same feeds.conf.default pins), so derive the release
|
||||
# from the matrix sdk tag (x86_64-24.10.4 -> 24.10.4).
|
||||
- name: Compute feeds cache key
|
||||
id: feedskey
|
||||
run: echo "ver=$(echo '${{ matrix.sdk }}' | sed 's/.*-//')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache SDK feeds checkouts
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: .cache/feeds
|
||||
key: feeds-opkg-${{ steps.feedskey.outputs.ver }}
|
||||
restore-keys: |
|
||||
feeds-opkg-
|
||||
|
||||
- name: Cache CI tools (usign)
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
@@ -162,25 +227,23 @@ jobs:
|
||||
# Build the SPA-embedded, static-musl, UPX'd shaterd for BOTH arches and
|
||||
# stage dist/shaterd-<a>.upx into openwrt/shaterd/files/. MUST run before
|
||||
# the SDK package build (the openwrt/shaterd package installs the staged
|
||||
# artifact). VERSION is stamped into constant.Version. On an exact
|
||||
# artifact). $SHATER_VERSION (from the version step above) is stamped into
|
||||
# constant.Version, so the binary and the package agree. On an exact
|
||||
# node_modules cache hit, --fast skips the redundant `npm ci`.
|
||||
- name: Build & stage shaterd artifact
|
||||
env:
|
||||
NPM_CACHE_HIT: ${{ steps.npm-cache.outputs.cache-hit }}
|
||||
run: |
|
||||
set -eu
|
||||
if [ "${GITHUB_REF#refs/tags/}" != "$GITHUB_REF" ]; then
|
||||
V="${GITHUB_REF#refs/tags/}"
|
||||
else
|
||||
V="v0.2.0-dev"
|
||||
fi
|
||||
FAST=""
|
||||
if [ "${NPM_CACHE_HIT:-}" = "true" ]; then FAST="--fast"; fi
|
||||
echo "shaterd version: $V (npm cache hit: ${NPM_CACHE_HIT:-false})"
|
||||
bash scripts/build-shaterd.sh "$V" $FAST
|
||||
echo "shaterd version: $SHATER_VERSION / package ${SHATER_PKG_VERSION}-r${SHATER_PKG_RELEASE} (npm cache hit: ${NPM_CACHE_HIT:-false})"
|
||||
bash scripts/build-shaterd.sh $FAST
|
||||
|
||||
# Compile the 4 packages through the arch-matched OpenWrt SDK and produce a
|
||||
# signed per-arch opkg feed (Packages + Packages.gz + Packages.sig + .ipk).
|
||||
# SHATER_PKG_VERSION/SHATER_PKG_RELEASE reach the package Makefiles through
|
||||
# the SDK container; ci/sdk-build.sh asserts the .ipk really carry them.
|
||||
- name: Build signed feed (SDK)
|
||||
env:
|
||||
KEY_BUILD: ${{ secrets.KEY_BUILD }}
|
||||
@@ -218,8 +281,12 @@ jobs:
|
||||
- arch: aarch64_cortex-a53 # BPI-R3 mini (BananaWRT 25.12-mtk-vendor) + BPI-R4
|
||||
sdk_url: https://downloads.immortalwrt.org/releases/25.12.1/targets/mediatek/filogic/immortalwrt-sdk-25.12.1-mediatek-filogic_gcc-14.3.0_musl.Linux-x86_64.tar.zst
|
||||
steps:
|
||||
# fetch-depth: 0 — see the opkg lane: the package version comes from
|
||||
# `git describe`, which needs tags + ancestry.
|
||||
- name: Checkout
|
||||
uses: actions/checkout@v4
|
||||
with:
|
||||
fetch-depth: 0
|
||||
|
||||
# scripts/build-shaterd.sh builds the AmneziaWG-patched wireguard-go via a
|
||||
# go.mod `replace => ./submodules/wireguard-go`, so that submodule must be
|
||||
@@ -228,6 +295,11 @@ jobs:
|
||||
- name: Init wireguard-go submodule (awg)
|
||||
run: git submodule update --init --depth 1 submodules/wireguard-go
|
||||
|
||||
# Same single version computation as the opkg lane — both lanes MUST agree
|
||||
# on the version, they package the identical tree.
|
||||
- name: Compute version from git tag
|
||||
run: bash ci/version.sh --env >> "$GITHUB_ENV"
|
||||
|
||||
- name: Set up Go
|
||||
uses: actions/setup-go@v5
|
||||
with:
|
||||
@@ -281,7 +353,9 @@ jobs:
|
||||
# mirror) — so even with a dead cache server this never wedges the run.
|
||||
- name: Compute SDK cache key
|
||||
id: sdkkey
|
||||
run: echo "tarball=$(basename '${{ matrix.sdk_url }}')" >> "$GITHUB_OUTPUT"
|
||||
run: |
|
||||
echo "tarball=$(basename '${{ matrix.sdk_url }}')" >> "$GITHUB_OUTPUT"
|
||||
echo "relver=$(echo '${{ matrix.sdk_url }}' | sed -n 's#.*/releases/\([^/]*\)/.*#\1#p')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache ImmortalWrt SDK tarball
|
||||
uses: actions/cache@v3.3.2
|
||||
@@ -289,6 +363,16 @@ jobs:
|
||||
path: .cache/sdk
|
||||
key: sdk-${{ steps.sdkkey.outputs.tarball }}
|
||||
|
||||
# feeds git checkouts (see header) — one entry shared by both apk arch
|
||||
# jobs of one ImmortalWrt release (identical feeds.conf.default pins).
|
||||
- name: Cache SDK feeds checkouts
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: .cache/feeds
|
||||
key: feeds-apk-${{ steps.sdkkey.outputs.relver }}
|
||||
restore-keys: |
|
||||
feeds-apk-
|
||||
|
||||
- name: Install UPX
|
||||
run: sudo apt-get update -qq && sudo apt-get install -y -qq upx-ucl
|
||||
|
||||
@@ -299,15 +383,10 @@ jobs:
|
||||
NPM_CACHE_HIT: ${{ steps.npm-cache.outputs.cache-hit }}
|
||||
run: |
|
||||
set -eu
|
||||
if [ "${GITHUB_REF#refs/tags/}" != "$GITHUB_REF" ]; then
|
||||
V="${GITHUB_REF#refs/tags/}"
|
||||
else
|
||||
V="v0.2.0-dev"
|
||||
fi
|
||||
FAST=""
|
||||
if [ "${NPM_CACHE_HIT:-}" = "true" ]; then FAST="--fast"; fi
|
||||
echo "shaterd version: $V (npm cache hit: ${NPM_CACHE_HIT:-false})"
|
||||
bash scripts/build-shaterd.sh "$V" $FAST
|
||||
echo "shaterd version: $SHATER_VERSION / package ${SHATER_PKG_VERSION}-r${SHATER_PKG_RELEASE} (npm cache hit: ${NPM_CACHE_HIT:-false})"
|
||||
bash scripts/build-shaterd.sh $FAST
|
||||
|
||||
# Compile the 4 packages as .apk through the ImmortalWrt 25.12 SDK and
|
||||
# sign the per-arch packages.adb with the EC key (secret KEY_APK).
|
||||
@@ -415,7 +494,7 @@ jobs:
|
||||
luci-app-shater (arch=all).
|
||||
Targets: x86_64 (testbed) and aarch64_cortex-a53 (BPI-R3 + BPI-R4, mediatek/filogic).
|
||||
|
||||
── Add as an opkg feed (recommended — then `opkg upgrade` just works) ──
|
||||
── Add as an opkg feed (recommended — then updating is one command) ──
|
||||
This release is itself a SIGNED package feed; opkg filters by
|
||||
architecture, so the same lines work on every device:
|
||||
wget -O /etc/opkg/keys/5ac4b177689cb8e0 https://git.qomar.pw/omar/shater/releases/download/latest/shater-feed.pub
|
||||
@@ -425,6 +504,10 @@ jobs:
|
||||
The public-key install is one-time; after it, `opkg update/upgrade`
|
||||
verify the signature with check_signature left on. Full guide: docs-shater/INSTALL.md.
|
||||
|
||||
── Update (name our packages — never a bare `opkg upgrade`) ──
|
||||
opkg update
|
||||
opkg upgrade shaterd shater-core luci-app-shater byedpi
|
||||
|
||||
── Or install the loose .ipk directly / from the tarball feed ──
|
||||
wget -O /tmp/f.tgz <this release>/shater-feed-aarch64_cortex-a53.tar.gz
|
||||
mkdir -p /tmp/shater && tar -C /tmp/shater -xzf /tmp/f.tgz
|
||||
@@ -440,6 +523,10 @@ jobs:
|
||||
release-apk:
|
||||
name: release apk
|
||||
needs: build-apk
|
||||
# Publish whatever arch feeds succeeded — do NOT block the aarch64 release
|
||||
# when an unrelated arch (e.g. x86_64) fails. download-artifact only fetches
|
||||
# artifacts that exist, and the publish loop skips missing apkfeed-* dirs.
|
||||
if: ${{ !cancelled() }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
@@ -480,13 +567,19 @@ jobs:
|
||||
Packages: shaterd + byedpi (per-arch), shater-core + luci-app-shater (arch=all).
|
||||
The index \`packages.adb\` is EC-signed; trust anchor \`shater-apk.pem\` (also in \`dist/\`).
|
||||
|
||||
── Add as an apk repository (auto-updates via \`apk upgrade\`) ──
|
||||
── Add as an apk repository ──
|
||||
wget -O /etc/apk/keys/shater-apk.pem https://git.qomar.pw/omar/shater/releases/download/$TAG/shater-apk.pem
|
||||
echo \"https://git.qomar.pw/omar/shater/releases/download/apk-latest-\$(cat /etc/apk/arch)/packages.adb\" > /etc/apk/repositories.d/shater.list
|
||||
apk update
|
||||
apk add luci-app-shater # pulls shater-core + shaterd too
|
||||
apk add byedpi # optional: ByeDPI desync egress
|
||||
Update: apk update && apk upgrade shaterd shater-core luci-app-shater byedpi
|
||||
── Update — ALWAYS name the packages, NEVER a bare \`apk upgrade\` ──
|
||||
apk update
|
||||
apk upgrade shaterd shater-core luci-app-shater byedpi
|
||||
A bare \`apk upgrade\` reconciles EVERY installed package against every
|
||||
configured repo and can downgrade unrelated system packages; naming them
|
||||
upgrades only those (apk-tools 3: \"If list of packages is provided, only
|
||||
those packages are upgraded along with needed dependencies\").
|
||||
Full guide: docs-shater/INSTALL.md §6. The opkg/24.10 feed lives in the \`latest\` release."
|
||||
echo "[release-apk] publishing $TAG from $d"
|
||||
TAG="$TAG" NAME="shater apk $VER ($arch)" BODY="$BODY" \
|
||||
|
||||
+20
-14
@@ -1,28 +1,34 @@
|
||||
#!/usr/bin/env bash
|
||||
# mod from https://gist.github.com/pldubouilh/c5703052986bfdd404005951dee54683
|
||||
|
||||
set -e -o pipefail
|
||||
set -euo pipefail
|
||||
|
||||
ARCH=$1
|
||||
DEB_SRC=$2
|
||||
OUT_IPK=$3
|
||||
|
||||
PROJECT=$(dirname "$0")/../..
|
||||
TMP_PATH=`mktemp -d`
|
||||
cp $2 $TMP_PATH
|
||||
pushd $TMP_PATH
|
||||
TMP_PATH=$(mktemp -d)
|
||||
trap 'rm -rf "$TMP_PATH"' EXIT
|
||||
|
||||
DEB_NAME=`ls *.deb`
|
||||
ar x $DEB_NAME
|
||||
cp "$DEB_SRC" "$TMP_PATH"/
|
||||
pushd "$TMP_PATH" >/dev/null
|
||||
|
||||
# Derive the name from the file we copied — do not glob-parse `ls *.deb`.
|
||||
DEB_NAME=$(basename "$DEB_SRC")
|
||||
ar x "$DEB_NAME"
|
||||
|
||||
mkdir control
|
||||
pushd control
|
||||
pushd control >/dev/null
|
||||
tar xf ../control.tar.gz
|
||||
rm md5sums
|
||||
sed "s/Architecture:\\ \w*/Architecture:\\ $1/g" ./control -i
|
||||
rm -f md5sums
|
||||
sed "s/Architecture:\\ \w*/Architecture:\\ $ARCH/g" ./control -i
|
||||
cat control
|
||||
tar czf ../control.tar.gz ./*
|
||||
popd
|
||||
popd >/dev/null
|
||||
|
||||
DEB_NAME=${DEB_NAME%.deb}
|
||||
tar czf $DEB_NAME.ipk control.tar.gz data.tar.gz debian-binary
|
||||
popd
|
||||
tar czf "$DEB_NAME.ipk" control.tar.gz data.tar.gz debian-binary
|
||||
popd >/dev/null
|
||||
|
||||
cp $TMP_PATH/$DEB_NAME.ipk $3
|
||||
rm -r $TMP_PATH
|
||||
cp "$TMP_PATH/$DEB_NAME.ipk" "$OUT_IPK"
|
||||
|
||||
@@ -36,8 +36,15 @@ nul
|
||||
# playwright MCP screenshots/snapshots
|
||||
.playwright-mcp/
|
||||
|
||||
# working-session screenshots dropped in the repo root (not shipped docs)
|
||||
/*.png
|
||||
|
||||
# throwaway build binaries / scratch staged under tmp/
|
||||
/tmp/
|
||||
|
||||
# -- upstream sing-box-lx -----------------------------------------
|
||||
/.idea/
|
||||
.idea/
|
||||
/vendor/
|
||||
/*.json
|
||||
/*.srs
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
## Правила делегирования
|
||||
|
||||
1. ЛЮБАЯ реализация (код, тесты, конфиги, рефакторинг, отладка) выполняется
|
||||
субагентами через инструмент Agent с `model: "opus"`. Сам ты правишь файлы
|
||||
субагентами через инструмент Agent с `model: "fable"`. Сам ты правишь файлы
|
||||
только в одном случае: тривиальная правка в 1–2 строки, где постановка
|
||||
задачи дороже самой правки.
|
||||
|
||||
@@ -27,7 +27,7 @@
|
||||
названия скиллов прямо в текст задания.
|
||||
|
||||
4. Независимые задачи запускай ПАРАЛЛЕЛЬНО — несколько вызовов Agent в одном
|
||||
сообщении, каждый с `model: "opus"`. Зависимые — последовательно, передавая
|
||||
сообщении, каждый с `model: "fable"`. Зависимые — последовательно, передавая
|
||||
в следующее ТЗ результаты предыдущего.
|
||||
|
||||
5. Приёмка: результат каждого субагента ты проверяешь сам (читаешь diff
|
||||
|
||||
+109
@@ -0,0 +1,109 @@
|
||||
<!-- Language: [Русский](README.md) · **English** -->
|
||||
|
||||
# shater
|
||||
|
||||
**A self-hosted internet-control appliance for OpenWrt routers.** One box turns a
|
||||
home or office network into a transparent VPN gateway, a network-wide
|
||||
ad/tracker/malware blocker, per-device parental control, and a live traffic
|
||||
dashboard — all local, all configured from a rich built-in web panel.
|
||||
|
||||
> The primary README is Russian — [README.md](README.md). This is a condensed
|
||||
> English mirror.
|
||||
|
||||
[](LICENSE)
|
||||

|
||||
|
||||
## What it is
|
||||
|
||||
shater is a network proxy stack for **OpenWrt / ImmortalWrt / BananaWRT** routers
|
||||
(Banana Pi BPI-R3, BPI-R4 and compatible). It transparently routes all LAN traffic
|
||||
through a proxy (split by domain/geo/client), filters DNS, gathers statistics, and
|
||||
is managed from a built-in web panel.
|
||||
|
||||
The engine is a **fork of [sing-box](https://github.com/SagerNet/sing-box) via
|
||||
[sing-box-lx](https://github.com/Leadaxe/sing-box-lx)**, compiled into a single Go
|
||||
binary `shaterd` together with the control plane, DNS filter, stats aggregator and
|
||||
the web panel itself. Broad protocol set: VLESS/VMess/Trojan/Shadowsocks,
|
||||
Reality/XTLS, WireGuard, **AmneziaWG 2.0**, Hysteria2, TUIC, XHTTP, MASQUE/CONNECT-IP.
|
||||
|
||||
A thin **LuCI launcher** (mini-dashboard + "Open panel" button) hands the browser a
|
||||
single-use token into the standalone SPA the daemon serves on its own port
|
||||
(default `:8088`).
|
||||
|
||||
## Highlights
|
||||
|
||||
- Transparent **TPROXY** data plane (TCP + UDP), SNI/Host/QUIC sniffing, no DNS leaks.
|
||||
- First-match routing by source / destination / list / geo / client → outbound /
|
||||
selector / chain / direct / block; node groups with balancer/observatory;
|
||||
multi-hop chains; per-rule egress.
|
||||
- **Fail-closed kill-switch** (dead group → block, never a silent direct leak); own
|
||||
`inet shater` nft table; atomic apply with `nft -c` validation and commit-confirm
|
||||
auto-rollback.
|
||||
- **DNS filtering & blocklists** with flexible sources (inline / file / url /
|
||||
geosite), compiled `.srs` matcher; Block-DoH/DoT to stop filter bypass.
|
||||
- Subscriptions (Clash / sing-box / Xray-JSON) and manual nodes; node health board.
|
||||
- Per-device control (proxy/blocklist toggles, exit country, per-device block/allow,
|
||||
schedules) and per-domain/client/device statistics from in-process DNS events.
|
||||
|
||||
Full list with MVP/T1/T2 tags — [`docs-shater/FEATURES.md`](docs-shater/FEATURES.md).
|
||||
|
||||
## Install
|
||||
|
||||
Two signed feeds. Pick by the router's OpenWrt version. Verbatim commands and the
|
||||
manual `.ipk`/`.apk` install are in [`docs-shater/INSTALL.md`](docs-shater/INSTALL.md).
|
||||
|
||||
**opkg (OpenWrt 24.10):**
|
||||
|
||||
```sh
|
||||
wget -O /etc/opkg/keys/5ac4b177689cb8e0 \
|
||||
https://git.qomar.pw/omar/shater/releases/download/latest/shater-feed.pub
|
||||
echo "src/gz shater https://git.qomar.pw/omar/shater/releases/download/latest" \
|
||||
>> /etc/opkg/customfeeds.conf
|
||||
opkg update && opkg install luci-app-shater # -> shater-core -> shaterd
|
||||
```
|
||||
|
||||
**apk (OpenWrt / ImmortalWrt / BananaWRT 25.12+):**
|
||||
|
||||
```sh
|
||||
wget -O /etc/apk/keys/shater-apk.pem \
|
||||
"https://git.qomar.pw/omar/shater/releases/download/apk-latest-$(cat /etc/apk/arch)/shater-apk.pem"
|
||||
echo "https://git.qomar.pw/omar/shater/releases/download/apk-latest-$(cat /etc/apk/arch)/packages.adb" \
|
||||
> /etc/apk/repositories.d/shater.list
|
||||
apk update && apk add luci-app-shater # -> shater-core -> shaterd
|
||||
```
|
||||
|
||||
shater ships **inert** (globals off) so install never breaks connectivity. After
|
||||
configuring nodes/rules: `uci set shater.globals.enabled=1 && uci commit shater`,
|
||||
then `shaterd apply` and `shaterd confirm`.
|
||||
|
||||
## Build from source
|
||||
|
||||
`scripts/build-shaterd.sh [VERSION] [--fast]` builds the SPA (Vite), embeds it via
|
||||
`//go:embed`, cross-builds musl-static `{amd64, arm64}` and UPX-packs the artifact
|
||||
into `openwrt/shaterd/files/`. Details in
|
||||
[`docs-shater/INSTALL.md`](docs-shater/INSTALL.md).
|
||||
|
||||
## Repository layout
|
||||
|
||||
| Path | What |
|
||||
|------|------|
|
||||
| `shater/` | Go control plane, DNS filter, stats aggregator, engine host |
|
||||
| `panel/` | Admin SPA (Vite + React + TS) and its Go server |
|
||||
| `openwrt/` | Packages: `shaterd`, `shater-core`, `luci-app-shater`, `byedpi` |
|
||||
| `docs-shater/` | Product documentation |
|
||||
| `scripts/`, `ci/`, `.gitea/workflows/` | Build script, feed/release scripts, CI |
|
||||
| `SPECS/`, `docs-lx/` | Engine-fork constitution/specs and feature-config reference |
|
||||
| `docs/`, `mkdocs.yml` | **Upstream** sing-box docs (mkdocs) — kept as-is |
|
||||
| `adapter/ cmd/ dns/ route/ option/ protocol/ transport/ …` | sing-box-lx engine tree |
|
||||
|
||||
## CI, upstream & license
|
||||
|
||||
CI (`.gitea/workflows/release.yml`) builds all 4 packages and publishes signed
|
||||
feeds: opkg (usign, key `5ac4b177689cb8e0`) and apk (EC key `shater-apk.pem`). A
|
||||
`vX.Y.Z` tag → versioned release; `workflow_dispatch` → rolling `latest`.
|
||||
|
||||
The engine is the **sing-box-lx** fork — a thin downstream of upstream sing-box that
|
||||
lives by **rebase, never merge**; its constitution is
|
||||
[`SPECS/CONSTITUTION.md`](SPECS/CONSTITUTION.md). Licensed under
|
||||
[GPL-3.0](LICENSE), like upstream sing-box. Unofficial fork, not affiliated with
|
||||
SagerNet.
|
||||
@@ -1,71 +1,334 @@
|
||||
<!-- Язык: **Русский** · [English](README.en.md) -->
|
||||
|
||||
# shater
|
||||
|
||||
**A self-hosted internet-control appliance for OpenWrt.** One box turns your
|
||||
network into a transparent VPN gateway, a network-wide ad/tracker/malware blocker,
|
||||
per-device parental control, and a live traffic dashboard — configured from a rich
|
||||
web admin panel, all local.
|
||||
**Управляемый интернет-шлюз для роутеров на OpenWrt.** Одна коробка превращает
|
||||
домашнюю или офисную сеть в прозрачный VPN-шлюз, сетевой блокировщик рекламы,
|
||||
трекеров и вредоносных доменов, средство родительского контроля по устройствам
|
||||
и живую панель аналитики трафика — всё локально, всё self-hosted, всё
|
||||
настраивается из богатой веб-панели.
|
||||
|
||||
[](LICENSE)
|
||||

|
||||

|
||||

|
||||
|
||||
> ⚠️ **v0.2 is under active development on a new foundation.** The previous,
|
||||
> complete and VM-verified xray-based version lives on the **[`v0.1`](../../src/branch/v0.1)**
|
||||
> branch and still installs from the signed feed.
|
||||
---
|
||||
|
||||
## What v0.2 is
|
||||
## Что это
|
||||
|
||||
shater v0.2 is built as a **fork of [sing-box](https://github.com/SagerNet/sing-box)
|
||||
(via [sing-box-lx](https://github.com/Leadaxe/sing-box-lx))** with our whole
|
||||
product embedded in the one binary: the proxy engine, a control plane, a DNS
|
||||
filter, and a full admin panel. Riding sing-box gives a broad, up-to-date protocol
|
||||
set — VLESS/VMess/Trojan/Shadowsocks, Reality, **AmneziaWG 2.0**, Hysteria2, TUIC —
|
||||
without reinventing the anti-DPI arms race.
|
||||
**shater** — это сетевой прокси-стек для роутеров на **OpenWrt / ImmortalWrt /
|
||||
BananaWRT** (Banana Pi BPI-R3, BPI-R4 и совместимые). Он прозрачно (без настройки
|
||||
клиентов) заворачивает весь LAN-трафик через прокси с маршрутизацией по домену,
|
||||
гео и клиенту, фильтрует DNS, собирает статистику и управляется из встроенной
|
||||
веб-панели.
|
||||
|
||||
The UI is split for both integration and a great experience: a **thin LuCI app**
|
||||
(a small dashboard + an "Open panel" button) hands a short-lived token to a
|
||||
**standalone admin panel** the daemon serves on its own port — so panel auth is
|
||||
bootstrapped from LuCI's existing login, and the real UX is a modern SPA we fully
|
||||
own.
|
||||
Ядро — **форк движка [sing-box](https://github.com/SagerNet/sing-box) через
|
||||
[sing-box-lx](https://github.com/Leadaxe/sing-box-lx)** — вкомпилировано в один
|
||||
Go-бинарь `shaterd` вместе с control-plane, DNS-фильтром, агрегатором статистики и
|
||||
самой веб-панелью. За счёт sing-box поддерживается широкий и актуальный набор
|
||||
протоколов: VLESS/VMess/Trojan/Shadowsocks, Reality/XTLS, WireGuard,
|
||||
**AmneziaWG 2.0**, Hysteria2, TUIC, XHTTP, MASQUE/CONNECT-IP.
|
||||
|
||||
## Highlights (planned)
|
||||
Интеграция в OpenWrt — тонкий **LuCI-лаунчер**: мини-дашборд и кнопка «Открыть
|
||||
панель», которая по одноразовому токену передаёт браузер в полноценную SPA-панель,
|
||||
поднятую демоном на собственном порту (по умолчанию `:8088`).
|
||||
|
||||
- Transparent TPROXY proxy (TCP+UDP), split by domain/geo/client, no DNS leaks.
|
||||
- Broad protocols incl. **AmneziaWG 2.0**, Reality, Hysteria2, TUIC.
|
||||
- Network-wide **DNS blocklists** with flexible sources (inline / file / url /
|
||||
geosite) and an efficient matcher for million-entry lists.
|
||||
- **Per-domain, per-client, per-device statistics** — fed by the engine's DNS
|
||||
events in-process (no log scraping).
|
||||
- **Per-device control**: block a site for one device or everyone; per-device
|
||||
exit/proxy toggles; schedules; alerts.
|
||||
- Fail-closed kill-switch, atomic apply with commit-confirm rollback, signed opkg
|
||||
feed.
|
||||
---
|
||||
|
||||
See **[`docs-shater/FEATURES.md`](docs-shater/FEATURES.md)** for the full list.
|
||||
## Ключевые возможности
|
||||
|
||||
## Documentation
|
||||
**Прозрачный прокси и маршрутизация**
|
||||
- TPROXY data-plane для нескольких LAN-интерфейсов (TCP + UDP), сниффинг
|
||||
SNI/Host/QUIC, без утечек DNS.
|
||||
- Правила маршрутизации по источнику (IP/CIDR/MAC/интерфейс/зона), назначению
|
||||
(domain/suffix/keyword/geosite), спискам, порту, протоколу →
|
||||
outbound / selector / chain / direct / block.
|
||||
- Группы узлов с балансировщиком/обсерваторией (least-ping / failover /
|
||||
round-robin), **мульти-хоп цепочки** и выбор egress по правилу.
|
||||
|
||||
| Doc | What |
|
||||
|-----|------|
|
||||
| [`docs-shater/CONTEXT.md`](docs-shater/CONTEXT.md) | **Start here** — project context, v0.1→v0.2 history, decisions in brief, testbed/infra |
|
||||
| [`docs-shater/ROADMAP.md`](docs-shater/ROADMAP.md) | Phased plan (Phase 1 = fork + embedding prototype) |
|
||||
| [`docs-shater/FEATURES.md`](docs-shater/FEATURES.md) | Full feature list with MVP/T1/T2 tags |
|
||||
| [`docs-shater/ARCHITECTURE.md`](docs-shater/ARCHITECTURE.md) | One-binary design, auth handoff, data/DNS/apply flow (diagrams) |
|
||||
| [`docs-shater/DECISIONS.md`](docs-shater/DECISIONS.md) | Why sing-box, why fork, why the panel split, license, etc. |
|
||||
| [`docs-shater/DESIGN.md`](docs-shater/DESIGN.md) | Admin-panel visual system — the "Faceplate" direction, tokens, components, north-star prototype |
|
||||
**Надёжность («железно»)**
|
||||
- **Fail-closed kill-switch**: мёртвая группа → block, а не тихая утечка мимо
|
||||
прокси; собственная nft-таблица `inet shater` и свои марки/таблицы, fw4 не
|
||||
трогаем.
|
||||
- Атомарный apply с валидацией движком и `nft -c`, **commit-confirm** с
|
||||
авто-откатом к последней рабочей конфигурации.
|
||||
- Идемпотентный reconcile из hotplug/boot под flock; management-bypass
|
||||
(SSH/LuCI/LAN) всегда в обход.
|
||||
|
||||
## Status
|
||||
**DNS, фильтрация, блокировки**
|
||||
- Перехват `:53`, DNS движка sing-box в процессе; резолверы DoH/DoT/plain/FakeIP,
|
||||
выбор резолвера по домену.
|
||||
- **Блок-листы с гибкими источниками**: `inline` / `file` / `url` (авто-обновление) /
|
||||
категория `geosite`; hosts-файл, plain-список или AdBlock-стиль `||domain^`
|
||||
компилируются в локальный `.srs`. Эффективный компилированный матчер вместо
|
||||
dnsmasq-мегасписков.
|
||||
- **Block-DoH/DoT** — не даёт устройствам обходить фильтр через свой шифрованный DNS.
|
||||
|
||||
Foundation reset complete: v0.1 preserved on its branch, `main` reset for v0.2.
|
||||
Next is **Phase 1** — fork sing-box-lx into `main` and stand up the embedding
|
||||
prototype (prove AmneziaWG 2.0, measure binary size). Follow `docs-shater/ROADMAP.md`.
|
||||
**Подписки и узлы**
|
||||
- Подписки (VLESS/VMess/Trojan/SS/WG/AmneziaWG), форматы Clash/sing-box/Xray-JSON,
|
||||
интервал обновления + вручную + на загрузке; стабильная идентичность узла между
|
||||
обновлениями; квоты/срок из `subscription-userinfo`.
|
||||
- Ручные узлы: share-ссылки, импорт файла, `wg-quick`/AmneziaWG `.conf`.
|
||||
- Health board: TCP + реальная проба через прокси-путь, exit-IP, «протестировать
|
||||
все».
|
||||
|
||||
## Hardware
|
||||
**Контроль по устройствам**
|
||||
- Авто-обнаружение устройств (dhcp.leases + `ip neigh`), имена, живой статус/трафик.
|
||||
- Тумблеры на устройство: прокси on/off, блок-листы on/off, страна/узел выхода;
|
||||
блок/allow домена для одного устройства или для всех; расписания.
|
||||
|
||||
`aarch64_cortex-a53` covers Banana Pi **BPI-R3** (MT7986/Filogic 830) and **BPI-R4**
|
||||
(MT7988/Filogic 880), both the OpenWrt `mediatek/filogic` target. `x86_64` is the
|
||||
QEMU test VM.
|
||||
**Статистика и видимость**
|
||||
- Топ доменов (запрошенные/заблокированные), allowed-vs-blocked, разбивка по
|
||||
устройствам, таймлайны — из DNS-событий движка в процессе (без скрейпинга логов).
|
||||
- Трафик по клиенту/узлу/правилу (байты) из nft-счётчиков; живой query-log.
|
||||
|
||||
## License
|
||||
**Панель и профили**
|
||||
- Встроенная SPA-панель (собственный порт, вшита в бинарь): overview, узлы и
|
||||
подписки, правила маршрутизации, DNS/блок-листы, устройства, apply/rollback.
|
||||
- Именованные профили/сцены и WAN-профили (условные оверрайды).
|
||||
|
||||
[GPL-3.0](LICENSE) (sing-box is GPL-3.0). See `docs-shater/DECISIONS.md` D6.
|
||||
Полный список с тегами MVP/T1/T2 — [`docs-shater/FEATURES.md`](docs-shater/FEATURES.md).
|
||||
|
||||
---
|
||||
|
||||
## Архитектура
|
||||
|
||||
Один бинарь `shaterd` держит движок, control-plane, DNS-фильтр и веб-сервер панели
|
||||
в одном процессе; OpenWrt-обвязка (тонкий LuCI + procd/system glue) оборачивает его.
|
||||
Конфиг — UCI desired-state; демон рендерит его в конфиг движка и применяет;
|
||||
телеметрия течёт обратно в панель.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph BIN["shaterd — один бинарь (форк sing-box-lx)"]
|
||||
ENG["движок sing-box\nпротоколы · Reality · AmneziaWG 2.0 · DNS · routing · stats"]
|
||||
CTRL["control-plane (shater/)\nUCI-модель · генерация конфига · apply/rollback · nft/routing"]
|
||||
FILT["DNS-фильтр + блок-листы + политика по устройствам (shater/)"]
|
||||
STAT["агрегатор статистики (shater/)"]
|
||||
PANEL["веб-сервер панели + вшитая SPA (свой порт, токен-auth)"]
|
||||
end
|
||||
subgraph WRT["OpenWrt-обвязка (openwrt/)"]
|
||||
LUCI["тонкий LuCI — мини-дашборд + кнопка «Открыть панель»"]
|
||||
PROCD["procd init · hotplug · uci-defaults · fw4/routing"]
|
||||
end
|
||||
LUCI -->|"ubus: mint token"| PANEL
|
||||
PROCD --> BIN
|
||||
CTRL --> ENG
|
||||
FILT --> ENG
|
||||
ENG --> STAT
|
||||
STAT --> PANEL
|
||||
```
|
||||
|
||||
Путь трафика: LAN-клиент → `nft tproxy` (mark → tproxy-порт) → tproxy-inbound
|
||||
sing-box (сниффинг SNI/Host/QUIC) → маршрут по правилу → outbound/selector/chain
|
||||
(проксировано) · direct (flow-offload) · block. Подробные диаграммы (auth-handoff,
|
||||
data-plane, DNS-flow, apply-flow) — в [`docs-shater/ARCHITECTURE.md`](docs-shater/ARCHITECTURE.md).
|
||||
|
||||
---
|
||||
|
||||
## Установка
|
||||
|
||||
shater поставляется двумя подписанными фидами. Выберите по версии OpenWrt на роутере:
|
||||
|
||||
- **OpenWrt 24.10** → фид **opkg** (`.ipk`, `Packages.gz`, ключ usign).
|
||||
- **OpenWrt / ImmortalWrt / BananaWRT 25.12+** → фид **apk** (`.apk`, `packages.adb`,
|
||||
EC-ключ).
|
||||
|
||||
Пакеты ставятся по зависимостям: `shaterd` → `shater-core` → `luci-app-shater`
|
||||
(+ опциональный `byedpi`). `shaterd` подтягивается автоматически как зависимость.
|
||||
|
||||
### Путь A — фид opkg (OpenWrt 24.10)
|
||||
|
||||
```sh
|
||||
# 1) доверяем ключу фида — ИМЯ файла обязано равняться отпечатку usign-ключа.
|
||||
wget -O /etc/opkg/keys/5ac4b177689cb8e0 \
|
||||
https://git.qomar.pw/omar/shater/releases/download/latest/shater-feed.pub
|
||||
|
||||
# 2) добавляем фид (один URL обслуживает все арки).
|
||||
echo "src/gz shater https://git.qomar.pw/omar/shater/releases/download/latest" \
|
||||
>> /etc/opkg/customfeeds.conf
|
||||
|
||||
# 3) обновляемся и ставим (shaterd подтянется как зависимость).
|
||||
opkg update
|
||||
opkg install luci-app-shater # -> shater-core -> shaterd
|
||||
opkg install byedpi # опционально: ByeDPI desync-egress
|
||||
```
|
||||
|
||||
Обновление — **только наши пакеты, никогда голый `opkg upgrade`** (без аргументов
|
||||
он тянет обновления и на системные пакеты, это классический способ окирпичить
|
||||
роутер):
|
||||
|
||||
```sh
|
||||
opkg update
|
||||
opkg upgrade shaterd shater-core luci-app-shater byedpi
|
||||
```
|
||||
|
||||
### Путь B — фид apk (OpenWrt / ImmortalWrt / BananaWRT 25.12+)
|
||||
|
||||
`/etc/apk/arch` сам выбирает нужный per-arch релиз (apk-релизы раздельны по арке):
|
||||
|
||||
```sh
|
||||
# 1) доверяем ключу apk-фида (любое имя *.pem под /etc/apk/keys подходит).
|
||||
wget -O /etc/apk/keys/shater-apk.pem \
|
||||
"https://git.qomar.pw/omar/shater/releases/download/apk-latest-$(cat /etc/apk/arch)/shater-apk.pem"
|
||||
|
||||
# 2) добавляем репозиторий — строка указывает на сам ФАЙЛ-ИНДЕКС packages.adb.
|
||||
echo "https://git.qomar.pw/omar/shater/releases/download/apk-latest-$(cat /etc/apk/arch)/packages.adb" \
|
||||
> /etc/apk/repositories.d/shater.list
|
||||
|
||||
# 3) обновляемся и ставим (shaterd подтянется как зависимость).
|
||||
apk update
|
||||
apk add luci-app-shater # -> shater-core -> shaterd
|
||||
apk add byedpi # опционально: ByeDPI desync-egress
|
||||
```
|
||||
|
||||
Обновление — **перечисляйте пакеты явно, голый `apk upgrade` не запускайте**: без
|
||||
аргументов apk пересобирает состояние ВСЕХ установленных пакетов по ВСЕМ
|
||||
подключённым репозиториям и может задеть (в т.ч. откатить) посторонние системные
|
||||
пакеты.
|
||||
|
||||
```sh
|
||||
apk update
|
||||
apk upgrade shaterd shater-core luci-app-shater byedpi
|
||||
# эквивалент, дополнительно закрепляющий пакеты в world:
|
||||
# apk add -u shaterd shater-core luci-app-shater byedpi
|
||||
```
|
||||
|
||||
Документация apk-tools 3 про `apk upgrade`: *«If list of packages is provided,
|
||||
only those packages are upgraded along with needed dependencies»*. Проверить
|
||||
установленные версии: `apk list -I shaterd shater-core luci-app-shater byedpi`.
|
||||
|
||||
> Версии пакетов CI берёт из git-тега (`vX.Y.Z` → `X.Y.Z-r1`, сборка вне тега →
|
||||
> `X.Y.Z-r<коммитов+1>`), поэтому каждая новая сборка действительно видна
|
||||
> менеджеру пакетов как новая. Подробности — `docs-shater/INSTALL.md` §2.1.
|
||||
|
||||
> Полные инструкции — раздельная установка из `.ipk`/`.apk` вручную, закрепление
|
||||
> версии (`vX.Y.Z` / `apk-vX.Y.Z-<arch>`), совместимость с BananaWRT
|
||||
> `25.12-mtk-vendor` — в [`docs-shater/INSTALL.md`](docs-shater/INSTALL.md).
|
||||
|
||||
### Включение
|
||||
|
||||
shater ставится **инертным** (globals выключены), чтобы установка не рвала связь.
|
||||
Настройте узлы/правила (через панель или `uci`), затем включите и примените:
|
||||
|
||||
```sh
|
||||
uci set shater.globals.enabled=1
|
||||
uci commit shater
|
||||
shaterd apply # apply + вооружить commit-confirm на живом демоне
|
||||
shaterd confirm # подтвердить (отменяет авто-откат)
|
||||
```
|
||||
|
||||
`/etc/init.d/shater enable && /etc/init.d/shater start` поднимает демона под procd.
|
||||
Кнопка «Открыть панель» в LuCI чеканит одноразовый токен и передаёт браузер в
|
||||
панель (`:8088` по умолчанию).
|
||||
|
||||
---
|
||||
|
||||
## Сборка из исходников
|
||||
|
||||
Ship-артефакт — бинарь `shaterd` со вшитой SPA. Собирается вне дерева SDK скриптом
|
||||
`scripts/build-shaterd.sh`:
|
||||
|
||||
```sh
|
||||
scripts/build-shaterd.sh [VERSION] [--fast]
|
||||
```
|
||||
|
||||
Что он делает: (1) собирает панель — `cd panel && npm ci && npm run build` (Vite →
|
||||
`panel/dist`); (2) копирует `panel/dist/*` в `shater/panel/webroot/`, откуда
|
||||
`//go:embed` вшивает **реальную** SPA в бинарь; (3) кросс-собирает под `{amd64,
|
||||
arm64}` с musl-static набором тегов (`CGO_ENABLED=0 GOOS=linux`), stripped/trimmed;
|
||||
(4) прогоняет UPX `--lzma --best` (~42 МБ → ~8–11 МБ); (5) стейджит артефакт в
|
||||
`openwrt/shaterd/files/` для пакета.
|
||||
|
||||
Затем OpenWrt-пакеты из `openwrt/` собираются каноническим путём SDK. Детали
|
||||
(набор build-тегов, почему `shaterd` — prebuilt-пакет, порядок CI) — в
|
||||
[`docs-shater/INSTALL.md`](docs-shater/INSTALL.md).
|
||||
|
||||
---
|
||||
|
||||
## Структура репозитория
|
||||
|
||||
Репозиторий — это оверлей продукта **shater** поверх дерева форка движка
|
||||
**sing-box-lx** (конфликт-фри: движок в апстрим-каталогах, продукт в своих).
|
||||
|
||||
| Путь | Что это |
|
||||
|------|---------|
|
||||
| `shater/` | Go: control-plane, DNS-фильтр, агрегатор статистики, хост движка |
|
||||
| `panel/` | Админ-SPA (Vite + React + TS) и её Go-сервер |
|
||||
| `openwrt/` | Пакеты: `shaterd`, `shater-core`, `luci-app-shater`, `byedpi` |
|
||||
| `docs-shater/` | Документация продукта (см. таблицу ниже) |
|
||||
| `scripts/` | `build-shaterd.sh` — сборка ship-артефакта |
|
||||
| `ci/` | Скрипты сборки фидов и релизов (SDK, usign/EC, Gitea API) |
|
||||
| `.gitea/workflows/` | `release.yml` — CI: сборка пакетов + подписанные фиды opkg/apk |
|
||||
| `SPECS/` | Конституция форка движка и спеки (Spec Kit) |
|
||||
| `docs-lx/` | Справочник конфигурации фич движка (`lx-config.md`, `.ru.md`) |
|
||||
| `lx-test/`, `submodules/` | Примеры конфигов движка и submodule AmneziaWG-рантайма |
|
||||
| `docs/`, `mkdocs.yml` | **Апстрим** документация sing-box (mkdocs) — как есть |
|
||||
| `adapter/ cmd/ dns/ route/ option/ protocol/ transport/ …` | Дерево движка sing-box-lx |
|
||||
|
||||
---
|
||||
|
||||
## CI и релизы
|
||||
|
||||
CI на **Gitea Actions** (`.gitea/workflows/release.yml`) собирает все 4 пакета и
|
||||
публикует **подписанные фиды**:
|
||||
|
||||
- **opkg (24.10):** один комбинированный релиз, подписан usign-ключом (публичный
|
||||
`dist/shater-feed.pub`, отпечаток `5ac4b177689cb8e0`; секрет — в Gitea-secret
|
||||
`KEY_BUILD`).
|
||||
- **apk (25.12+):** параллельная линия, **по релизу на арку**, подписан EC-ключом
|
||||
(`dist/shater-apk.pem`; секрет — `KEY_APK`).
|
||||
|
||||
Триггеры: push тега **`vX.Y.Z`** → версионный релиз; `workflow_dispatch` →
|
||||
плавающий `latest`/`apk-latest-<arch>` (всегда свежий фид). Публикация — через
|
||||
Gitea API (`ci/gitea-release.sh`). Ключи **никогда не перегенерируются** — это
|
||||
инвалидировало бы доверие на всех развёрнутых роутерах.
|
||||
|
||||
---
|
||||
|
||||
## Связь с upstream и движок
|
||||
|
||||
shater вкомпилирует **форк движка sing-box-lx** — тонкий downstream апстрима
|
||||
[SagerNet/sing-box](https://github.com/SagerNet/sing-box), добавляющий набор
|
||||
клиентских фич (XHTTP, AmneziaWG 2.0, MASQUE, расширения наблюдаемости) за
|
||||
build-тегами и живущий **ребейзом на каждый upstream-тег, а не merge**. Форк
|
||||
разрабатывается по Spec Kit; неизменяемые принципы — в
|
||||
[`SPECS/CONSTITUTION.md`](SPECS/CONSTITUTION.md), справочник фич движка — в
|
||||
[`docs-lx/lx-config.ru.md`](docs-lx/lx-config.ru.md).
|
||||
|
||||
История: **v0.1** (движок на xray-core, полностью рабочая и VM-проверенная версия)
|
||||
сохранена на ветке **[`v0.1`](../../src/branch/v0.1)**. v0.2 схлопнула runtime в
|
||||
один форкнутый бинарь.
|
||||
|
||||
---
|
||||
|
||||
## Документация
|
||||
|
||||
| Документ | О чём |
|
||||
|----------|-------|
|
||||
| [`docs-shater/CONTEXT.md`](docs-shater/CONTEXT.md) | **Начните здесь** — контекст проекта, история v0.1→v0.2, testbed/инфра |
|
||||
| [`docs-shater/INSTALL.md`](docs-shater/INSTALL.md) | Сборка ship-артефакта и установка обоих фидов (opkg/apk) |
|
||||
| [`docs-shater/ARCHITECTURE.md`](docs-shater/ARCHITECTURE.md) | One-binary дизайн, auth-handoff, data/DNS/apply-потоки (диаграммы) |
|
||||
| [`docs-shater/FEATURES.md`](docs-shater/FEATURES.md) | Полный список фич с тегами MVP/T1/T2 |
|
||||
| [`docs-shater/ROADMAP.md`](docs-shater/ROADMAP.md) | Фазовый план |
|
||||
| [`docs-shater/DECISIONS.md`](docs-shater/DECISIONS.md) | Почему sing-box, почему форк, split панели, лицензия |
|
||||
| [`docs-shater/DESIGN.md`](docs-shater/DESIGN.md) | Визуальная система панели — «Faceplate», токены, компоненты |
|
||||
| [`docs-shater/PORTING.md`](docs-shater/PORTING.md) | Порт проверенных кусков из v0.1 |
|
||||
|
||||
Индекс папки — [`docs-shater/README.md`](docs-shater/README.md).
|
||||
|
||||
---
|
||||
|
||||
## Оборудование
|
||||
|
||||
Арка `aarch64_cortex-a53` покрывает Banana Pi **BPI-R3** (MT7986/Filogic 830) и
|
||||
**BPI-R4** (MT7988/Filogic 880) — оба таргет OpenWrt `mediatek/filogic`. `x86_64` —
|
||||
QEMU-стенд для тестов.
|
||||
|
||||
---
|
||||
|
||||
## Лицензия
|
||||
|
||||
[GPL-3.0](LICENSE) — как у upstream sing-box. Подробности — в
|
||||
[`docs-shater/DECISIONS.md`](docs-shater/DECISIONS.md) (D6). Неофициальный форк, не
|
||||
аффилирован с SagerNet.
|
||||
|
||||
+15
-235
@@ -1,239 +1,19 @@
|
||||
[English](README.md) · **Русский**
|
||||
# shater — этот файл переехал
|
||||
|
||||
# sing-box-lx
|
||||
Лицо этого репозитория — продукт **shater** (управляемый интернет-шлюз для
|
||||
роутеров на OpenWrt). Основной README на русском — **[README.md](README.md)**;
|
||||
краткая английская версия — **[README.en.md](README.en.md)**.
|
||||
|
||||
> **Тонкий downstream-форк [SagerNet/sing-box](https://github.com/SagerNet/sing-box).**
|
||||
> Небольшой набор клиентских фич поверх upstream — транспорт **XHTTP**, **AmneziaWG 2.0**, **MASQUE** (CONNECT-IP / Cloudflare WARP), расширения **наблюдаемости** (CommandClient) и балансировка нагрузки **round_robin** — каждая за своим build-tag.
|
||||
> Набор может расти, философия — нет: жить ребейзом на каждый upstream-тег, а не отдельной жизнью.
|
||||
Раньше здесь лежал README форка движка **sing-box-lx**, который shater
|
||||
вкомпилирует в свой бинарь. Документация именно движка-форка живёт в его слое:
|
||||
|
||||
> 📄 README самого upstream sing-box — **[на GitHub](https://github.com/SagerNet/sing-box/blob/main/README.md)** (всегда актуальный).
|
||||
- **[docs-lx/lx-config.ru.md](docs-lx/lx-config.ru.md)** — справочник конфигурации
|
||||
фич движка (XHTTP, AmneziaWG 2.0, MASQUE).
|
||||
- **[SPECS/CONSTITUTION.md](SPECS/CONSTITUTION.md)** — конституция тонкого форка
|
||||
(принципы, build-tag изоляция, ребейз-модель).
|
||||
- **[SPECS/README.md](SPECS/README.md)** — формат задач Spec Kit.
|
||||
- Апстрим-README самого sing-box —
|
||||
[на GitHub](https://github.com/Leadaxe/sing-box-lx).
|
||||
|
||||
Это не отдельный проект и не «улучшенный sing-box». Это upstream sing-box **плюс несколько фич**, реализованных так, чтобы их можно было переносить на новые версии sing-box годами, почти без конфликтов. Со временем фич может становиться больше — другие протоколы, новые возможности, — но каждая обязана жить по тем же правилам тонкого форка ([CONSTITUTION](SPECS/CONSTITUTION.md)).
|
||||
|
||||
---
|
||||
|
||||
## Уникальное позиционирование
|
||||
|
||||
В экосистеме sing-box форки, добавляющие XHTTP/AmneziaWG, делятся на два лагеря — и `sing-box-lx` не входит ни в один:
|
||||
|
||||
| Форк | Фичи | Подход | Синк с upstream |
|
||||
|------|------|--------|-----------------|
|
||||
| **SagerNet/sing-box** (upstream) | базовый | — | — |
|
||||
| **shtorm-7/sing-box-extended** | десятки (WARP, MASQUE, MTProxy, XHTTP, AWG2, …) | «комбайн», правки повсюду | отдельная ветка, без ребейза на теги |
|
||||
| **amnezia-vpn/amnezia-box**, **hoaxisr/amnezia-box** | только AWG | толстый форк, правки in-place | синк по веткам (`dev-next`/`stable-next`) |
|
||||
| **➡ sing-box-lx** (этот репозиторий) | **малый набор (XHTTP, AWG2, наблюдаемость, round_robin)** | **тонкий: новые файлы за build-tag, минимум касаний upstream** | **ребейз атомарных `// lx`-коммитов на upstream-теги** |
|
||||
|
||||
**Чем мы отличаемся:**
|
||||
|
||||
- **Минимальная дивергенция.** Новый код живёт в новых файлах. Существующие upstream-файлы трогаются только в крошечных помеченных швах `// lx:begin … // lx:end`. → дешёвые ребейзы.
|
||||
- **Изоляция за build-tag.** Фичи включаются тегами `with_xhttp` / `with_awg`. Сборка **без** них байт-в-байт повторяет поведение upstream — фичи ничего не ломают по умолчанию.
|
||||
- **Идентичность сохранена.** Go-модуль остаётся `github.com/sagernet/sing-box`, бинарь называется `sing-box`. Суффикс `-lx` есть только в строке версии (`1.13.13-lx.N`).
|
||||
- **Build-tag — родная конвенция sing-box**, а не наше изобретение (`with_quic`, `with_wireguard`, …). Мы просто применяем её с максимальной дисциплиной.
|
||||
|
||||
> Готовые форки-комбайны мы **не тянем как зависимость**, а используем только как референс wire-протокола.
|
||||
|
||||
---
|
||||
|
||||
## Фичи и статус
|
||||
|
||||
| # | Фича | Что это | Статус |
|
||||
|---|------|---------|--------|
|
||||
| **XHTTP** | клиентский транспорт | Xray-совместимый «splithttp» (режимы `auto`/`packet-up`/`stream-up`/`stream-one`) поверх Reality/TLS/h2c | ✅ **проверен живым Xray (3x-ui) сервером** (packet-up/auto): handshake + DNS + HTTPS + скачивание. `stream-one` — известный баг framing |
|
||||
| **AmneziaWG 2.0** | клиентский endpoint | обфускация WireGuard: `Jc/Jmin/Jmax`, `S1–S4`, `H1–H4` + **2.0**: `I1–I5` (CPS — кастомные пакеты-приманки) | ✅ собирается, проходит `check`; зависимость **активирована** ([Leadaxe/wireguard-go-awg2-lx](https://github.com/Leadaxe/wireguard-go-awg2-lx) — sagernet-база + обфускация); **проверено живым AWG2-сервером**: handshake + keepalive + трафик наружу |
|
||||
| **Маскировка `id/ip/ib`** | сахар над AWG | WireSock-стиль: декларативная маскировка поверх `I1` — домен (`id`) + протокол (`ip`: `quic`/`dns`/`stun`/`sip`) + браузер (`ib`), ядро строит клиент-инициированную `I1`-приманку: `quic` = out-of-order фрагментированный Initial (i1+i2), `dns`/`stun`/`sip` = query/Binding-Request/INVITE | ✅ **`ip=quic` device-проверен на реальном LTE/WARP DPI** (~330 мс, упрощает Cloudflare WARP); `dns`/`stun`/`sip` собираются и проходят `check`, но режутся как класс протокола к WARP-edge — для других провайдеров |
|
||||
| **Наблюдаемость** (расширения CommandClient) | live-стрим для UI | нативные расширения libbox gRPC за `with_lx_command`: `URLTestOutbound`, `GetRules`, `GetGroups`, `GetOutbounds`, `GetPool`, плюс `Connection.detourList` (хвост detour'а отдельным полем, SPEC 017) и `SubscribeDNSQueries` — структурный live-поток DNS (домен, qtype, rcode `-1`=ошибка, CNAME-цепочка, привязка к процессу, `dnsServer`/`dnsServerType`/`outbound`, SPEC 018) | ✅ в rc-серии, потребляется **LxBox**. SPEC 014–018: [`014`](SPECS/014-CLASH_API_TO_COMMANDCLIENT_MIGRATION/SPEC.md) · [`015`](SPECS/015-COMMAND_PROTOCOL_RPC_EXTENSIONS/SPEC.md) · [`017`](SPECS/017-CONNECTION_DETOUR_CHAIN/SPEC.md) · [`018`](SPECS/018-DNS_QUERY_STREAM/SPEC.md) |
|
||||
| **round_robin** (балансировка нагрузки) | режим `urltest` | пул-балансировка на `urltest` за `with_lx_command` (для `GetPool`): `mode` `least_test` (дефолт) \| `round_robin`; `balancer{pool (дефолт 3), pool_tolerance (0=держать живые / >0=топ по задержке), sticky_hash}`. Sticky-ключ: пропущен/`[]` → дефолт `["process","domain"]`, `["none"]` → выкл; компоненты `process`/`domain`/`source_ip`/`dest_ip`/`dest_port`. Фиксированные слоты `slot[hash(key)%pool]` (FNV-64a), замена в слоте; `GetPool` отдаёт слоты | ✅ локально равномерно (10/10/10, sticky off); rc.15 починил схлопывание `domain`-ключа (теперь читается `metadata.Domain`, переживающий resolve домен→IP, а не пустой `destination.Fqdn`) — на устройстве равномерность 0.27 → 0.95+. SPEC [`019`](SPECS/019-URLTEST_MODE_STICKY/SPEC.md), конфиг — [docs/.../urltest.md](docs/configuration/outbound/urltest.md) |
|
||||
| **MASQUE** (`type: masque`) | клиентский outbound | CONNECT-IP (RFC 9484) поверх HTTP/3 **или** HTTP/2 для **Cloudflare WARP** (SPEC 021): туннелирует целые IP-пакеты через userspace gVisor-стек; `profile` (`cloudflare`/`standard`), `network` (`h3`/`h2`), pinning ECDSA public key, idle-suspend + самовосстановление. h2 — ручной фреймер поверх `x/net/http2` (без доп. зависимостей); `connect-ip-go` вкопан | ✅ **device-verified на Wi-Fi и LTE** (`warp=on`, реальный трафик на `h3` и `h2`); на сетях, режущих входящий UDP:443, `h3`-handshake виснет — там `network: h2` (TCP:443) |
|
||||
|
||||
Подробные отчёты — в [`SPECS/002-…`](SPECS/002-XHTTP_CLIENT_TRANSPORT/IMPLEMENTATION_REPORT.md), [`SPECS/003-…`](SPECS/003-AWG2_CLIENT_ENDPOINT/IMPLEMENTATION_REPORT.md) и [`SPECS/009-…`](SPECS/009-WIRESOCK_MASQUERADE_PROFILES/IMPLEMENTATION_REPORT.md). Полный справочник конфига — **[docs-lx/lx-config.ru.md](docs-lx/lx-config.ru.md)**.
|
||||
|
||||
> **Не поддерживается (слой Reality, отложено):** post-quantum Reality (`pqv` / ML-DSA-65) и `spiderX` из Xray. Это Xray-специфичные фичи Reality, которых нет в sing-box, а Reality — upstream-слой TLS, который мы держим нетронутым (это не одна из наших фич). Классический X25519 Reality работает; сервер, который **требует** post-quantum Reality, не подключится. Это ограничение sing-box — правильнее решать в upstream (получим на ребейзе).
|
||||
|
||||
---
|
||||
|
||||
## Сборка
|
||||
|
||||
Сборка идёт через отдельный **`Makefile.lx`** (upstream `Makefile` не трогаем):
|
||||
|
||||
```bash
|
||||
git clone --recurse-submodules https://github.com/Leadaxe/sing-box-lx
|
||||
make -f Makefile.lx lx-build
|
||||
# → бинарь ./sing-box с версией вида 1.13.13-lx.1
|
||||
```
|
||||
|
||||
> `--recurse-submodules` обязателен для `with_awg`: рантайм AmneziaWG подключён submodule'ом `submodules/wireguard-go` → [Leadaxe/wireguard-go-awg2-lx](https://github.com/Leadaxe/wireguard-go-awg2-lx).
|
||||
|
||||
Под капотом — стандартный `go build` с набором тегов (единственный источник истины — `make -f Makefile.lx lx-print-tags`):
|
||||
|
||||
```
|
||||
with_gvisor,with_quic,with_dhcp,with_wireguard,with_utls,with_clash_api,with_naive_outbound,with_purego,badlinkname,tfogo_checklinkname0,with_xhttp,with_awg
|
||||
```
|
||||
|
||||
Это клиентский feature-set upstream **минус** серверные/нерелевантные теги — `with_acme` (серверный выпуск сертов), `with_tailscale`, `with_ccm`/`with_ocm` (AI-прокси) — **плюс** `with_purego` (CGO-free кросс-сборка, чтобы `with_naive_outbound`/cronet собирался при `CGO=0` на любом desktop-таргете, кроме Windows 7 / 32-бит legacy-сборки, где naive выкинут — у `cronet-go` нет windows/386) и наши фичи `with_xhttp` / `with_awg`. Всё остальное — ровно как upstream.
|
||||
|
||||
Проверка конфигов:
|
||||
|
||||
```bash
|
||||
./sing-box check -c lx-test/config/xhttp_reality.json
|
||||
./sing-box check -c lx-test/config/awg2_basic.json
|
||||
```
|
||||
|
||||
> `lx-test/config/` — наши примеры (upstream `test/` — отдельный Go-модуль, его не используем).
|
||||
|
||||
**Android (`libbox.aar`).** `make lib_install && make lib_android` собирает gomobile-AAR — `libbox.aar` (SDK 23) + `libbox-legacy.aar` (SDK 21) — с зашитыми `with_xhttp`/`with_awg` (и без `tailscale`), для встраивания в Android-приложение-потребитель (нужны NDK r28 + OpenJDK 17). `Libbox.version()` отдаёт `…-lx.N`.
|
||||
|
||||
---
|
||||
|
||||
## Конфигурация фич
|
||||
|
||||
> Полные таблицы полей, дефолты и `awg-quick`→JSON маппинг — **[docs-lx/lx-config.ru.md](docs-lx/lx-config.ru.md)**. Здесь — кратко.
|
||||
|
||||
### XHTTP (outbound transport)
|
||||
|
||||
```jsonc
|
||||
"transport": {
|
||||
"type": "xhttp",
|
||||
"host": "example.com",
|
||||
"path": "/xhttp",
|
||||
"mode": "auto" // auto | packet-up | stream-up | stream-one
|
||||
}
|
||||
```
|
||||
|
||||
### AmneziaWG 2.0 (endpoint)
|
||||
|
||||
Поля AWG промотированы прямо в `WireGuardEndpointOptions`:
|
||||
|
||||
```jsonc
|
||||
{
|
||||
"type": "wireguard",
|
||||
// … стандартные поля wireguard (private_key, address, peers, …) …
|
||||
"jc": 10, "jmin": 50, "jmax": 100,
|
||||
"s1": 20, "s2": 20, "s3": 60, "s4": 60,
|
||||
"h1": 1, "h2": 2, "h3": 3, "h4": 4,
|
||||
"i1": "<b 0x...><r 12>", "i2": "", "i3": "", "i4": "", "i5": "" // 2.0 CPS
|
||||
}
|
||||
```
|
||||
|
||||
> `I1–I5` — это конфиг (не согласуется по сети), значения должны **совпадать на клиенте и сервере**, регистрозависимы.
|
||||
|
||||
**Сахар-маскировка (`id`/`ip`/`ib`).** Вместо ручного `i1` задаёшь домен, протокол и
|
||||
браузер — ядро само собирает `I1`-приманку (стиль WireSock). Удобно для упрощения
|
||||
коннекта к **Cloudflare WARP**:
|
||||
|
||||
```jsonc
|
||||
{
|
||||
"type": "wireguard",
|
||||
// … стандартные поля wireguard …
|
||||
"id": "www.google.com", "ip": "quic", "ib": "chrome" // quic: id идёт как SNI в ClientHello
|
||||
// или: "ip": "dns", "id": "www.google.com" // dns/sip: id идёт как QNAME/host
|
||||
}
|
||||
```
|
||||
|
||||
`ip` ∈ `quic|dns|stun|sip`; `id` обязателен только для `quic` (SNI); для `dns`/`sip` опционален (без него генерится псевдо-имя), `stun` игнорирует. Где задан — идёт на провод (SNI / QNAME / host)
|
||||
и опционален для `sip` (без него генерится псевдо-host) и `stun`; `ib` ∈ `chrome|firefox|curl`
|
||||
(только quic, эффект минимальный — без JA3-fingerprint). Взаимоисключается с явным `i1`.
|
||||
|
||||
Для **`quic`** ядро генерит out-of-order фрагментированный QUIC Initial (RFC 9001) — реальный
|
||||
ClientHello, нарезанный на CRYPTO-фреймы в перемешанном порядке, так что line-rate DPI парсит
|
||||
мусор и пропускает. Раскладка рандомизируется на каждый вызов (нет межюзерной сигнатуры), и
|
||||
`ip=quic` теперь шлёт **два** независимых Initial (i1+i2) — поток читается как развивающаяся
|
||||
QUIC-сессия. Это **единственный профиль, device-проверенный на реальном LTE/WARP DPI** (~330 мс).
|
||||
`dns`/`stun`/`sip` реализованы как корректные клиент-инициированные запросы, но режутся как класс
|
||||
протокола к WARP-edge (raw DNS/STUN/SIP к дата-центровому IP сам по себе аномален) — сохранены
|
||||
для других провайдеров, чей DPI проверяет лишь корректность пакета. См.
|
||||
[docs-lx/lx-config.ru.md](docs-lx/lx-config.ru.md) и [примеры SPECS/009](SPECS/009-WIRESOCK_MASQUERADE_PROFILES/EXAMPLES.md).
|
||||
|
||||
### MASQUE (outbound — Cloudflare WARP)
|
||||
|
||||
Outbound `masque` туннелирует целые IP-пакеты через **CONNECT-IP (RFC 9484)**, HTTP/3 или HTTP/2,
|
||||
к **Cloudflare WARP**. Не путать с AWG-сахаром *masquerade* `id/ip/ib` выше — разные фичи, одно слово.
|
||||
|
||||
```jsonc
|
||||
{
|
||||
"type": "masque",
|
||||
"tag": "warp",
|
||||
"server": "162.159.198.2",
|
||||
"server_port": 443,
|
||||
"profile": "cloudflare", // cloudflare (WARP) | standard (RFC 9484)
|
||||
"network": "h3", // ТРАНСПОРТ: h3 (QUIC) | h2 (HTTP/2). НЕ tcp/udp — это network_list
|
||||
"sni": "www.microsoft.com", // domain-fronting; endpoint аутентифицируется пиннингом public key, не по SNI
|
||||
"private_key": "<base64 DER EC>",
|
||||
"public_key": "<base64 DER PKIX>",
|
||||
"ip": "172.16.0.2/32", "ipv6": "2606:4700:110:...::/128"
|
||||
}
|
||||
```
|
||||
|
||||
Ключевой материал (`private_key`/`public_key`/`ip`/`ipv6`) берётся готовым из конфига — регистрацию
|
||||
устройства в WARP делает клиент. На сетях, режущих входящий UDP:443, `h3`-handshake виснет —
|
||||
переключите узел на `network: h2` (TCP:443). Полный справочник —
|
||||
[docs-lx/lx-config.ru.md §4](docs-lx/lx-config.ru.md) и [SPECS/021](SPECS/021-MASQUE_CONNECT_IP_OUTBOUND/CONFIG.md).
|
||||
|
||||
---
|
||||
|
||||
## Модель сопровождения
|
||||
|
||||
```
|
||||
upstream tag (vX.Y.Z)
|
||||
│
|
||||
└─► ветка lx = upstream + N атомарных // lx-коммитов
|
||||
├─ FORK_BOOTSTRAP (Makefile.lx, CI, версия)
|
||||
├─ XHTTP client transport
|
||||
├─ AWG2 client endpoint
|
||||
└─ … (новые фичи — такими же атомарными // lx-коммитами)
|
||||
```
|
||||
|
||||
- **Только ребейз, никогда merge.** На новый upstream-тег ветка `lx` ребейзится поверх него.
|
||||
- Каждая фича — атомарный коммит(ы), помеченный `// lx`. Новые файлы конфликтов не дают; швы в upstream-файлах малы и переносятся вручную.
|
||||
- Разработка ведётся по **Spec Kit** (`SPECS/NNN-T-S-NAME/`: SPEC → PLAN → TASKS → IMPLEMENTATION_REPORT).
|
||||
|
||||
### Remotes
|
||||
|
||||
```bash
|
||||
origin git@github.com:Leadaxe/sing-box-lx.git # ветка по умолчанию: lx
|
||||
upstream https://github.com/SagerNet/sing-box.git
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Структура lx-специфики
|
||||
|
||||
| Путь | Назначение |
|
||||
|------|------------|
|
||||
| `Makefile.lx` | сборка с lx-тегами и версией `-lx` |
|
||||
| `.github/workflows/lx-ci.yml` | CI: матрица фич (baseline/xhttp/awg/full) + negative-check + кросс-платформа + android AAR |
|
||||
| `.github/workflows/lx-release.yml` | релиз на `v*-lx.*`: desktop ×6 + `libbox.aar` → GitHub Release |
|
||||
| `SPECS/` | Spec Kit (конституция, задачи, отчёты) |
|
||||
| `lx-test/config/` | примеры конфигов для `sing-box check` |
|
||||
| `transport/v2rayxhttp/` | XHTTP-клиент (новый пакет) |
|
||||
| `transport/wireguard/device_awg.go` | AWG IpcSet-параметры (за `with_awg`) |
|
||||
| `submodules/wireguard-go` | submodule: merged-форк AmneziaWG-рантайма ([Leadaxe/wireguard-go-awg2-lx](https://github.com/Leadaxe/wireguard-go-awg2-lx)) |
|
||||
| `option/v2ray_xhttp.go`, `option/wireguard_awg.go` | опции фич |
|
||||
| `include/v2rayxhttp.go` | регистрация транспорта за build-tag |
|
||||
|
||||
Поиск всех правок upstream-файлов: `grep -rn "// lx"`.
|
||||
|
||||
---
|
||||
|
||||
## Потребитель
|
||||
|
||||
Ядро собирается для десктоп-лаунчера **singbox-launcher** (бандлит `bin/sing-box`). На Android потребитель встраивает **`libbox.aar`** (gomobile) вместо бинаря — конфиг-JSON тот же. Маппинг `type=xhttp` и AWG-полей в визарде — задачи на стороне потребителя, не здесь.
|
||||
|
||||
---
|
||||
|
||||
## Ссылки
|
||||
|
||||
| | |
|
||||
|---|---|
|
||||
| Upstream | [SagerNet/sing-box](https://github.com/SagerNet/sing-box) · [документация](https://sing-box.sagernet.org/) |
|
||||
| Этот форк | [Leadaxe/sing-box-lx](https://github.com/Leadaxe/sing-box-lx) |
|
||||
| AmneziaWG-рантайм | [Leadaxe/wireguard-go-awg2-lx](https://github.com/Leadaxe/wireguard-go-awg2-lx) — sagernet-база + обфускация (3-way merge) |
|
||||
| AmneziaWG upstream | [amnezia-vpn/amneziawg-go](https://github.com/amnezia-vpn/amneziawg-go) · [docs.amnezia.org](https://docs.amnezia.org/documentation/amnezia-wg/) |
|
||||
| XHTTP (исток) | [XTLS/Xray-core](https://github.com/XTLS/Xray-core) — `transport/internet/splithttp` |
|
||||
| Конфиг фич | [docs-lx/lx-config.ru.md](docs-lx/lx-config.ru.md) |
|
||||
| Spec Kit | [SPECS/](SPECS/) — [README](SPECS/README.md) · [CONSTITUTION](SPECS/CONSTITUTION.md) · [IMPLEMENTATION_PROMPT](SPECS/IMPLEMENTATION_PROMPT.md) |
|
||||
|
||||
---
|
||||
|
||||
## Лицензия
|
||||
|
||||
Наследует лицензию upstream sing-box (**GPL-3.0**). Все правки помечены `// lx` и распространяются под той же лицензией. Это неофициальный форк, не аффилирован с SagerNet.
|
||||
> Файл оставлен как указатель, чтобы у репозитория был один основной русский
|
||||
> README (`README.md`), а не два конкурирующих.
|
||||
|
||||
@@ -157,7 +157,9 @@ if need == 0: стоп // все слоты живы, не
|
||||
|
||||
## Ошибка дайла
|
||||
|
||||
Слоты/пул **не трогаем**. По одной ошибке дайла причина неизвестна (мёртвая нода / упавший сайт назначения / своя сеть пропала — неразличимы). Состав пула меняет **только дотест** (честный health-check через `url`). Соединение просто фейлится для этого запроса.
|
||||
Слоты/пул **не трогаем** — инвариант сохранён: по одной ошибке дайла причина неизвестна (мёртвая нода / упавший сайт назначения / своя сеть пропала — неразличимы), состав пула меняет **только дотест** (честный health-check через `url`).
|
||||
|
||||
> **Обновление (план «живой пул», §5.B / health board):** поведение «соединение просто фейлится» заменено. Дайл-провал теперь пишет `MarkFailed` в health board и дайл ретраится следующим кандидатом (≤3 членов, в пределах дедлайна коннекта) — коннект юзера выживает, пока в группе есть живая нода. Слоты при этом по-прежнему физически не двигаются: мёртвый жилец остаётся в своём слоте, но вердикт board (`dead`, TTL) делает его невыбираемым на следующем pick — селекция уходит с мёртвого слота, не дожидаясь дотеста. Sticky, replace-in-slot и never-shrink не затронуты.
|
||||
|
||||
---
|
||||
|
||||
@@ -248,7 +250,7 @@ PoolSlot {
|
||||
1. `balancer`-объект; `mode` снаружи как индикатор. Липкость — плоское `balancer.sticky_hash []string` (не вложенный объект — механизм один, выбирать нечего). Дефолт (омит или `[]`)→`["process","domain"]` (липкость из коробки); выключение через sentinel **`["none"]`**. *(rc.15: изначально планировали `[]`→выкл через nil-vs-`[]`, но `badjson.UnmarshallExcludedContext` ре-маршалит структуру и схлопывает `[]`→nil — различить на проде нельзя, подтверждено живым прогоном. `["none"]` переживает round-trip.)*
|
||||
2. `pool_tolerance` — **наше** поле; апстрим `tolerance` затирается на 50, для round_robin не используется (варн).
|
||||
3. `pool < 0` → ошибка; `pool==0`/опущено → дефолт 3; `round_robin` без `balancer` → дефолты.
|
||||
4. Дайл-ошибка слоты не трогает (причина неизвестна).
|
||||
4. Дайл-ошибка слоты не трогает (причина неизвестна). *(План «живой пул» §5.B: дайл-провал теперь `MarkFailed` в board + ретрай следующим кандидатом; слоты по-прежнему не двигаются — см. «Ошибка дайла».)*
|
||||
5. Пул не пустеет — замена только при наличии живого кандидата.
|
||||
6. **Слоты фиксированы; ноды текут сквозь; delay-ранг — отдельное вычисление для выявления худшей, не перестановка слотов.**
|
||||
7. **sticky slot-hash: `slot[hash(key) % pool]` (`pool` фиксирован) → строгий ноль реконнектов для живого узла, 0 памяти.** Единственный механизм. Заменяет ОБА механизма rc.11. Модуль достаточен, т.к. число слотов не меняется — преимущество jump-hash/rendezvous (плавный ресайз) здесь не нужно.
|
||||
@@ -273,7 +275,7 @@ PoolSlot {
|
||||
- **Слоты фиксированы:** вылет ноды из середины не двигает другие слоты; победитель занимает слот вытесненного.
|
||||
- **slot-hash:** один ключ → стабильный слот при неизменном `pool`; **живой узел в своём слоте держит ВСЕ свои ключи** при замене жильцов в других слотах (ноль реконнектов); замена в слоте `k` трогает только ключи `→ k`; ключ `""` → фиксированный слот.
|
||||
- **Замена 1:1:** нет живых → мёртвая держит слот; `min(pool, nodes)`.
|
||||
- **Дайл-ошибка** не меняет состав пула.
|
||||
- **Дайл-ошибка** не меняет состав пула (но с планом «живой пул» §5.B демотит слот через вердикт board и ретраится — см. «Ошибка дайла»).
|
||||
- **Валидация:** `pool<1`, `balancer`+wrong-mode, неизвестный `sticky_hash`-компонент → ошибки; `tolerance`+round_robin → варн.
|
||||
- **Дефолт sticky_hash:** `nil` (поле опущено) → `["process","domain"]` (липкость есть); `[]` → липкости нет (round_robin по counter); различение nil-vs-`[]` после unmarshal.
|
||||
- **`least_test` дефолт** — без изменений.
|
||||
|
||||
+10
-2
@@ -19,11 +19,19 @@ type ClashServer interface {
|
||||
AddModeUpdateHook(hook *observable.Subscriber[struct{}])
|
||||
}
|
||||
|
||||
// lx:begin health-board
|
||||
// Health board (plan §5.A): a history entry now records both the last success and
|
||||
// the last failure instead of being deleted on failure. `Time` is renamed to
|
||||
// `LastOK`; its JSON tag stays "time" so the Clash API history payload is
|
||||
// unchanged, and `LastFail` is omitted when zero for the same reason.
|
||||
type URLTestHistory struct {
|
||||
Time time.Time `json:"time"`
|
||||
Delay uint16 `json:"delay"`
|
||||
LastOK time.Time `json:"time"`
|
||||
Delay uint16 `json:"delay"`
|
||||
LastFail time.Time `json:"last_fail,omitzero"`
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
|
||||
type V2RayServer interface {
|
||||
LifecycleService
|
||||
StatsService() ConnectionTracker
|
||||
|
||||
+22
-1
@@ -55,6 +55,16 @@ fi
|
||||
|
||||
chmod +x "$REPO"/ci/*.sh 2>/dev/null || true
|
||||
|
||||
# --- 0.4) package version from the git tag ------------------------------------
|
||||
# Same contract as the opkg lane (ci/build-feed.sh): the workflow puts these in
|
||||
# the job env via `ci/version.sh --env >> $GITHUB_ENV`; recompute here when run
|
||||
# standalone. Passed into the container below and re-exported to the
|
||||
# unprivileged build user in ci/sdk-build-apk.sh.
|
||||
if [ -z "${SHATER_PKG_VERSION:-}" ] || [ -z "${SHATER_PKG_RELEASE:-}" ]; then
|
||||
eval "$(sh "$REPO/ci/version.sh" --env)"
|
||||
fi
|
||||
echo "[apk-feed] package version: ${SHATER_PKG_VERSION}-r${SHATER_PKG_RELEASE}"
|
||||
|
||||
# --- 0.5) runner-side caches --------------------------------------------------
|
||||
# All under $REPO/.cache so (a) actions/cache in the workflow can persist them
|
||||
# between runs and (b) the nested container sees them via --volumes-from.
|
||||
@@ -68,6 +78,15 @@ CACHE="$REPO/.cache"
|
||||
mkdir -p "$CACHE/sdk" "$CACHE/dl" "$CACHE/apt"
|
||||
chmod -R a+rwX "$CACHE/dl" "$CACHE/apt" 2>/dev/null || true
|
||||
|
||||
# feeds/ git checkouts (actions/cache key: feeds-apk-<release>) — symlinked
|
||||
# over the SDK's feeds dir inside the container (ci/sdk-build-apk.sh) so
|
||||
# `scripts/feeds update -a` fetches deltas instead of re-cloning the
|
||||
# ImmortalWrt feeds every run. Top-level chmod only: contents are created and
|
||||
# owned by the container's uid-1000 build user (restore preserves ownership).
|
||||
FEEDS_CACHE="$CACHE/feeds/apk"
|
||||
mkdir -p "$FEEDS_CACHE"
|
||||
chmod a+rwX "$CACHE" "$CACHE/feeds" "$FEEDS_CACHE" 2>/dev/null || true
|
||||
|
||||
# Fetch the SDK tarball ON THE RUNNER (restored cache -> own Gitea release-asset
|
||||
# mirror -> upstream with stall-kill + retries) instead of the old bare
|
||||
# `wget` inside the container, which hung whole runs when
|
||||
@@ -84,7 +103,9 @@ docker pull -q debian:bookworm
|
||||
docker run --rm --volumes-from "$(hostname)" \
|
||||
-e ARCH="$ARCH" -e REPO="$REPO" -e OUT="$OUT" -e SDK_URL="$SDK_URL" \
|
||||
-e SDK_TAR="$SDK_TAR" -e DL_DIR="$CACHE/dl" -e APT_CACHE="$CACHE/apt" \
|
||||
-e KEY_APK="${KEY_APK:-}" \
|
||||
-e FEEDS_CACHE="$FEEDS_CACHE" -e KEY_APK="${KEY_APK:-}" \
|
||||
-e SHATER_PKG_VERSION="$SHATER_PKG_VERSION" \
|
||||
-e SHATER_PKG_RELEASE="$SHATER_PKG_RELEASE" \
|
||||
debian:bookworm bash "$REPO/ci/sdk-build-apk.sh"
|
||||
|
||||
# --- 2) sanity: the per-arch apk repo dir must be complete -------------------
|
||||
|
||||
@@ -47,6 +47,18 @@ fi
|
||||
|
||||
chmod +x "$REPO"/ci/*.sh 2>/dev/null || true
|
||||
|
||||
# --- 0.4) package version from the git tag ------------------------------------
|
||||
# The workflow normally puts these in the job env (ci/version.sh --env >>
|
||||
# $GITHUB_ENV); recompute here when this script is run standalone so a manual
|
||||
# `ci/build-feed.sh ...` produces the same versions as CI. They are handed to the
|
||||
# SDK container below and read by openwrt/*/Makefile (bug B4 — versions used to
|
||||
# be hand-written literals that nobody bumped, so v0.2.2…v0.2.6 all shipped as
|
||||
# 0.2.0-r3 and no router could ever see an update).
|
||||
if [ -z "${SHATER_PKG_VERSION:-}" ] || [ -z "${SHATER_PKG_RELEASE:-}" ]; then
|
||||
eval "$(sh "$REPO/ci/version.sh" --env)"
|
||||
fi
|
||||
echo "[feed] package version: ${SHATER_PKG_VERSION}-r${SHATER_PKG_RELEASE}"
|
||||
|
||||
# --- 0.5) persistent dl/ (package source tarballs) ----------------------------
|
||||
# Workspace dir restored/saved by actions/cache in the workflow and shared into
|
||||
# the nested SDK container via --volumes-from; becomes CONFIG_DOWNLOAD_FOLDER
|
||||
@@ -57,6 +69,17 @@ DL_DIR="$REPO/.cache/dl"
|
||||
mkdir -p "$DL_DIR"
|
||||
chmod -R a+rwX "$DL_DIR" 2>/dev/null || true
|
||||
|
||||
# --- 0.6) persistent feeds/ git checkouts -------------------------------------
|
||||
# Workspace dir restored/saved by actions/cache (key: feeds-opkg-<release>) and
|
||||
# symlinked over the SDK's feeds/ inside the container (ci/sdk-build.sh), so
|
||||
# `scripts/feeds update -a` fetches deltas instead of re-cloning base+packages+
|
||||
# luci from scratch (~7 min/run on this runner's slow github.com link).
|
||||
# Top-level chmod only: the contents are created by the container's uid-1000
|
||||
# build user and restored with the same ownership (tar-as-root preserves it).
|
||||
FEEDS_CACHE="$REPO/.cache/feeds/opkg"
|
||||
mkdir -p "$FEEDS_CACHE"
|
||||
chmod a+rwX "$REPO/.cache" "$REPO/.cache/feeds" "$FEEDS_CACHE" 2>/dev/null || true
|
||||
|
||||
# --- 1) SDK package build (4 packages) in the arch-matched SDK image ----------
|
||||
# We drive the `openwrt/sdk` docker image directly (not openwrt/gh-action-sdk):
|
||||
# on a self-hosted Gitea act_runner the marketplace action fetch can be
|
||||
@@ -69,6 +92,9 @@ echo "[feed] SDK build arch=$ARCH image=openwrt/sdk:$SDK_TAG"
|
||||
docker pull "openwrt/sdk:$SDK_TAG"
|
||||
docker run --rm --volumes-from "$(hostname)" \
|
||||
-e ARCH="$ARCH" -e REPO="$REPO" -e OUT="$OUT" -e DL_DIR="$DL_DIR" \
|
||||
-e FEEDS_CACHE="$FEEDS_CACHE" \
|
||||
-e SHATER_PKG_VERSION="$SHATER_PKG_VERSION" \
|
||||
-e SHATER_PKG_RELEASE="$SHATER_PKG_RELEASE" \
|
||||
"openwrt/sdk:$SDK_TAG" \
|
||||
sh "$REPO/ci/sdk-build.sh"
|
||||
|
||||
|
||||
+1
-1
@@ -13,7 +13,7 @@
|
||||
# Feed format: opkg `src/gz` (.ipk + text Packages index, usign signature).
|
||||
# OpenWrt 24.10 (our SDK) still uses opkg; apk arrives at 25.12. The committed
|
||||
# trust anchor dist/shater-feed.pub is a usign (Ed25519) key, matching this.
|
||||
set -e
|
||||
set -euo pipefail
|
||||
OUT="${1:?feed dir required}"; cd "$OUT"
|
||||
: > Packages
|
||||
for ipk in *.ipk; do
|
||||
|
||||
+261
-2
@@ -28,6 +28,11 @@ SDK_URL="${SDK_URL:?SDK_URL env required}"
|
||||
|
||||
echo "[apk-sdk] arch=$ARCH repo=$REPO out=$OUT"
|
||||
echo "[apk-sdk] sdk=$SDK_URL"
|
||||
# Package version derived from the git tag by ci/version.sh (bug B4). Forwarded
|
||||
# to the unprivileged build user on the `su` line at the bottom of this file;
|
||||
# openwrt/{shaterd,shater-core,luci-app-shater}/Makefile pick it up from the
|
||||
# environment. byedpi keeps upstream ByeDPI's own version (see its Makefile).
|
||||
echo "[apk-sdk] package version: ${SHATER_PKG_VERSION:-<unset -> Makefile fallback>}-r${SHATER_PKG_RELEASE:-?}"
|
||||
test -f "$REPO/openwrt/shaterd/Makefile" || {
|
||||
echo "[apk-sdk] ERROR: feed not mounted ($REPO/openwrt/shaterd/Makefile missing)"; ls -la "$REPO" || true; exit 9; }
|
||||
|
||||
@@ -112,11 +117,122 @@ cd "$SDKDIR"
|
||||
cp -f feeds.conf.default feeds.conf
|
||||
grep -q '^src-link shater ' feeds.conf || echo "src-link shater $REPO/openwrt" >> feeds.conf
|
||||
|
||||
# Persistent feeds checkouts: $FEEDS_CACHE (workspace dir, actions/cache-
|
||||
# persisted, visible via --volumes-from) replaces the fresh SDK's empty feeds/
|
||||
# dir, so `feeds update` git-fetches deltas instead of re-cloning the
|
||||
# ImmortalWrt feeds every run. Update always checks out feeds.conf's pinned
|
||||
# revisions; on any failure with cached checkouts the cache is wiped and the
|
||||
# update retried with fresh clones — a stale cache can never wedge the build.
|
||||
if [ -n "${FEEDS_CACHE:-}" ] && mkdir -p "$FEEDS_CACHE" 2>/dev/null; then
|
||||
rm -rf feeds
|
||||
ln -s "$FEEDS_CACHE" feeds
|
||||
echo "[apk-sdk] feeds/ -> $FEEDS_CACHE (persistent cache)"
|
||||
fi
|
||||
echo "[apk-sdk] feeds update -a"
|
||||
./scripts/feeds update -a
|
||||
if ! ./scripts/feeds update -a; then
|
||||
[ -L feeds ] || { echo "[apk-sdk] ERROR: feeds update failed"; exit 8; }
|
||||
echo "[apk-sdk] WARNING: feeds update failed on cached checkouts — wiping cache, cloning fresh"
|
||||
find "$FEEDS_CACHE" -mindepth 1 -maxdepth 1 -exec rm -rf {} + 2>/dev/null || true
|
||||
./scripts/feeds update -a
|
||||
fi
|
||||
echo "[apk-sdk] feeds install (prefer shater feed)"
|
||||
./scripts/feeds install -p shater shaterd shater-core byedpi luci-app-shater
|
||||
|
||||
# --- strip the SDK's generated per-package `default m` blocks ----------------
|
||||
# Run 60 settled the question that runs 58 and 59 left open. Writing an explicit
|
||||
# `# CONFIG_PACKAGE_kmod-x is not set` for all 1126 of them and re-running
|
||||
# defconfig deselected exactly nothing: the count came back 1078, unchanged.
|
||||
# Meanwhile the very same explicit form DID stick for CONFIG_ALL/ALL_KMODS/
|
||||
# ALL_NONSHARED. The difference is prompts. kconfig only honours a user value for
|
||||
# a symbol that has one (sym_calc_value ignores S_DEF_USER for a promptless
|
||||
# symbol and falls back to its `default`), and the ALL* symbols carry prompts in
|
||||
# the SDK's own Config.in while these generated blocks are bare:
|
||||
#
|
||||
# config PACKAGE_kmod-mlx5-core
|
||||
# tristate
|
||||
# default m
|
||||
#
|
||||
# So no value we write into .config can ever turn them off — the fix has to
|
||||
# remove the `default m` itself. That is what this does: drop every generated
|
||||
# `config PACKAGE_*` block from the SDK's Config-build.in before the first
|
||||
# defconfig. Nothing is lost by it — these blocks only replay which packages the
|
||||
# BUILDBOT happened to build; the packages themselves are still declared, with
|
||||
# prompts, by the package tree (tmp/.config-package.in), which is what makes our
|
||||
# four selectable and what `select` acts on. KERNEL_*/LIBC/TOOLCHAIN blocks are
|
||||
# left untouched, so the SDK still reproduces its own toolchain settings.
|
||||
CB=$(find . -maxdepth 2 -name 'Config-build.in' -print -quit 2>/dev/null || true)
|
||||
if [ -n "$CB" ] && command -v perl >/dev/null 2>&1; then
|
||||
pkg_before=$(grep -c '^config PACKAGE_' "$CB" || true)
|
||||
# Paragraph-wise delete: a block is `config PACKAGE_x`, its indented body, and
|
||||
# the blank line that ends it. Anchored per-line (/m) so nothing else matches.
|
||||
perl -0777 -pi -e 's/^config PACKAGE_\S+\n(?:[ \t]+\S[^\n]*\n)+\n//gm' "$CB"
|
||||
pkg_after=$(grep -c '^config PACKAGE_' "$CB" || true)
|
||||
echo "[apk-sdk] $CB: stripped $((pkg_before - pkg_after)) generated PACKAGE default blocks ($pkg_before -> $pkg_after)"
|
||||
else
|
||||
echo "[apk-sdk] WARNING: no Config-build.in found (or no perl) — per-package"
|
||||
echo "[apk-sdk] 'default m' blocks stay; the kmod tripwire will catch it"
|
||||
fi
|
||||
|
||||
# --- .config: turn OFF the SDK's mass-select defaults ------------------------
|
||||
# Symptom (v0.2.2, and still v0.2.3 run 58): the SDK ran `apk mkpkg` on ~1100
|
||||
# kmod-* packages — mlx5, amdgpu, ata, isdn, none of which we ship — and died
|
||||
# with `Disk quota exceeded` on the runner's 64 GB ZFS quota. Our kmod deps pull
|
||||
# in `package/kernel/linux/compile`, which packs every module marked =m.
|
||||
#
|
||||
# Why they are =m has nothing to do with anything we write here. An OpenWrt SDK
|
||||
# carries its OWN top-level Config.in (target/sdk/files/Config.in), and it reads:
|
||||
#
|
||||
# config ALL_NONSHARED
|
||||
# bool "Select all target specific packages by default"
|
||||
# default ALL
|
||||
# config ALL_KMODS
|
||||
# bool "Select all kernel module packages by default"
|
||||
# default ALL
|
||||
# config ALL
|
||||
# bool "Select all userspace packages by default"
|
||||
# default y <-- y, not n, and ONLY inside the SDK
|
||||
#
|
||||
# In the main tree those three default to n; the SDK flips ALL to y so that
|
||||
# `make world` in a bare SDK builds something useful. So `make defconfig` on ANY
|
||||
# .config — empty or not — selects the entire kernel. This is stock OpenWrt, not
|
||||
# an ImmortalWrt quirk: openwrt/openwrt's target/sdk/files/Config.in is identical.
|
||||
# (It also means the reference we copied, Slava-Shchipunov/awg-openwrt, builds
|
||||
# every kmod too — it just never hits a disk quota on GitHub's runners.)
|
||||
#
|
||||
# Fix: state all three explicitly. They carry prompts in the SDK's Config.in, so
|
||||
# they are user-settable and an explicit value beats the `default`. Note the FORM:
|
||||
# kconfig writes a false bool as `# CONFIG_X is not set` and `CONFIG_X=n` is not
|
||||
# reliably honoured, so `is not set` is the only form used here. All three are set
|
||||
# rather than just the root `ALL`, so this keeps working whichever symbol a future
|
||||
# SDK makes the root of the chain.
|
||||
# Stash anything the SDK shipped (see below — today there is nothing) and start
|
||||
# from a known-empty file, so what we build here is exactly what we intended.
|
||||
if [ -s .config ]; then mv -f .config .config.sdk; fi
|
||||
: > .config
|
||||
for s in ALL ALL_KMODS ALL_NONSHARED; do
|
||||
echo "# CONFIG_$s is not set" >> .config
|
||||
done
|
||||
|
||||
# About that stash: an SDK tarball ships NO top-level .config (run 58 logged
|
||||
# `grep: .config: No such file or directory` — the only `.config` inside the
|
||||
# tarball is the prebuilt KERNEL's, under the linux dir). This is also why the
|
||||
# first version of this fix was aimed at the wrong thing: there was never a
|
||||
# buildbot .config here to append to. Nothing needs carrying over from it either,
|
||||
# because
|
||||
# target/sdk/Makefile bakes the buildbot's non-package settings — every
|
||||
# CONFIG_KERNEL_* included — into the SDK's generated Config-build.in as kconfig
|
||||
# `default`s (target/sdk/convert-config.pl). defconfig therefore reproduces the
|
||||
# exact toolchain/kernel settings the SDK was built with, on its own; an earlier
|
||||
# attempt to copy those lines by hand was redundant and is gone.
|
||||
# Should a future SDK start shipping a .config, this keeps the two things that
|
||||
# would then be worth honouring — the target identity and the package format —
|
||||
# and still lets the lines above override the mass-select.
|
||||
if [ -s .config.sdk ]; then
|
||||
echo "[apk-sdk] SDK shipped a .config — carrying over target identity + format:"
|
||||
grep -E '^CONFIG_TARGET_[a-z0-9_]+=y$|^CONFIG_TARGET_(BOARD|SUBTARGET|ARCH_PACKAGES)=|^CONFIG_USE_APK=' \
|
||||
.config.sdk | tee -a .config | sed 's/^/[apk-sdk] /' || true
|
||||
fi
|
||||
|
||||
for p in shaterd shater-core byedpi luci-app-shater; do
|
||||
echo "CONFIG_PACKAGE_$p=m" >> .config
|
||||
done
|
||||
@@ -135,11 +251,135 @@ fi
|
||||
echo "[apk-sdk] defconfig"
|
||||
make defconfig >/dev/null
|
||||
|
||||
# --- second pass: deselect the kernel, keep only what our packages select -----
|
||||
# Turning ALL/ALL_KMODS/ALL_NONSHARED off (above) provably worked — run 59 shows
|
||||
# all three as `is not set` after defconfig — and changed the kmod count by
|
||||
# exactly zero, 1078 both times. The kmods are not selected through ALL_KMODS at
|
||||
# all. They are selected one by one, and here is where from:
|
||||
#
|
||||
# target/sdk/Makefile:
|
||||
# ./convert-config.pl $(TOPDIR)/.config > $(SDK_BUILD_DIR)/Config-build.in
|
||||
#
|
||||
# The SDK's Config-build.in is GENERATED from the buildbot's .config — a config
|
||||
# in which ALL_KMODS=y had already expanded into a `CONFIG_PACKAGE_kmod-*=m` line
|
||||
# per module. convert-config.pl turns every `CONFIG_X=<val>` line into a kconfig
|
||||
# symbol carrying an unconditional `default <val>`; its `next if
|
||||
# /^(# )?CONFIG_PACKAGE/` filter sits in the `else` branch, which a line with an
|
||||
# `=` in it never reaches. So the SDK ships, verbatim, 1078 blocks of:
|
||||
#
|
||||
# config PACKAGE_kmod-mlx5-core
|
||||
# tristate
|
||||
# default m
|
||||
#
|
||||
# Nothing there consults ALL_KMODS, which is why switching it off was inert.
|
||||
#
|
||||
# Fix: give those symbols an explicit user value. We cannot do it before the
|
||||
# first defconfig — the list of names only exists once kconfig has expanded the
|
||||
# tree — so this is a second pass: rewrite every selected kmod to `is not set`
|
||||
# and re-run defconfig. Two kconfig rules make the result exactly what we want,
|
||||
# and both are already demonstrated in our own logs:
|
||||
# * an explicit value in .config beats a `default` (this is precisely why the
|
||||
# `# CONFIG_ALL* is not set` lines survived defconfig in run 59), so the
|
||||
# ~1078 kmods we do not need stay off;
|
||||
# * `select` is a reverse dependency, OR-ed into the symbol's value AFTER the
|
||||
# user value in sym_calc_value(), so it cannot be overridden by an explicit
|
||||
# `n`. shater-core's `DEPENDS:=+kmod-nft-tproxy +kmod-nft-socket` becomes
|
||||
# `select PACKAGE_kmod-nft-tproxy` (scripts/package-metadata.pl: a `+` flag
|
||||
# sets `$m = "select"`, and it re-emits the dependency's own depends too, so
|
||||
# transitive kmods follow). Those come back on their own.
|
||||
# Net effect: we build the handful of kmods our packages actually pull in.
|
||||
#
|
||||
# Rejected alternatives:
|
||||
# * limiting what `package/kernel/linux/compile` packs — that target has no
|
||||
# such knob; it iterates the selected set, so the selection IS the knob;
|
||||
# * `package/kernel/linux/clean` + a targeted build — the kernel package would
|
||||
# simply be rebuilt in full as a dependency of shater-core, same cost;
|
||||
# * copying OpenWrt's own feed CI (openwrt/gh-action-sdk) — it does nothing
|
||||
# about this; it just runs `make defconfig` and builds. Its one disk-related
|
||||
# setting, CONFIG_AUTOREMOVE=y, is already the SDK's default;
|
||||
# * editing the SDK's generated Config-build.in to strip the offending blocks —
|
||||
# it would work, but it means parsing a generated kconfig file by hand and a
|
||||
# format change would corrupt it silently. The two-pass approach uses only
|
||||
# kconfig's documented semantics and leaves the evidence in .config.
|
||||
kmods_all=$(grep -c '^CONFIG_PACKAGE_kmod-[^=]*=[my]$' .config || true)
|
||||
if [ "$kmods_all" -gt 0 ]; then
|
||||
echo "[apk-sdk] deselecting $kmods_all kmod packages, then defconfig again"
|
||||
sed -i -E 's/^CONFIG_(PACKAGE_kmod-[^=]*)=[my]$/# CONFIG_\1 is not set/' .config
|
||||
make defconfig >/dev/null
|
||||
fi
|
||||
|
||||
# --- post-defconfig sanity + disk-cost readout -------------------------------
|
||||
# A failed run leaves a ~27 MB log; digging the cause out of it is miserable, so
|
||||
# print the handful of numbers that decide whether this run survives the
|
||||
# runner's disk quota BEFORE anything is compiled.
|
||||
kmods=$(grep -c '^CONFIG_PACKAGE_kmod.*=m' .config || true)
|
||||
echo "[apk-sdk] target: board=$(sed -n 's/^CONFIG_TARGET_BOARD=//p' .config)" \
|
||||
"subtarget=$(sed -n 's/^CONFIG_TARGET_SUBTARGET=//p' .config)" \
|
||||
"arch_packages=$(sed -n 's/^CONFIG_TARGET_ARCH_PACKAGES=//p' .config)"
|
||||
echo "[apk-sdk] kmod packages selected (=m): $kmods"
|
||||
# Proof the mass-select stayed off: these three must come back out of defconfig
|
||||
# as `is not set`. If any reads `=y`, the SDK's `default ALL`/`default y` won and
|
||||
# the kmod count above will be in the four digits.
|
||||
echo "[apk-sdk] mass-select symbols after defconfig:"
|
||||
grep -E '^(# )?CONFIG_ALL(_KMODS|_NONSHARED)?[ =]' .config | sed 's/^/[apk-sdk] /' || true
|
||||
# After the second pass the only kmods left are the ones shater-core's
|
||||
# `DEPENDS:=+kmod-nft-tproxy +kmod-nft-socket` turns into kconfig `select`s, plus
|
||||
# whatever those select in turn — a handful. Worth printing verbatim while the
|
||||
# list is short. A count of 0 is NOT fatal: those kmods ship in the router's own
|
||||
# base feed, so apk resolves them there; but it would mean the selects did not
|
||||
# fire, and that is something we want to see in the log rather than guess at.
|
||||
if [ "$kmods" -le 30 ]; then
|
||||
grep '^CONFIG_PACKAGE_kmod.*=m' .config | sed 's/^/[apk-sdk] /' || true
|
||||
fi
|
||||
# The two cache knobs are written before the first defconfig and have to survive
|
||||
# both of them — losing DOWNLOAD_FOLDER silently costs us the dl/ cache, and
|
||||
# losing LOCALMIRROR brings back the sourceware.org stalls. Cheap to just look.
|
||||
echo "[apk-sdk] cache settings after defconfig:"
|
||||
grep -E '^CONFIG_(LOCALMIRROR|DOWNLOAD_FOLDER)=' .config | sed 's/^/[apk-sdk] /' || true
|
||||
echo "[apk-sdk] our packages after defconfig:"
|
||||
grep -E '^CONFIG_PACKAGE_(shaterd|shater-core|byedpi|luci-app-shater)=' .config \
|
||||
| sed 's/^/[apk-sdk] /' || true
|
||||
|
||||
# Each of our 4 must have SURVIVED defconfig. If kconfig dropped one, it is
|
||||
# because a symbol it `select`s (a DEPENDS entry) does not exist in the installed
|
||||
# feeds — with the old append-everything .config that was masked by the SDK
|
||||
# pre-selecting half the distro. `make package/<p>/compile` would then die with a
|
||||
# cryptic "No rule to make target", far from the real cause.
|
||||
for p in shaterd shater-core byedpi luci-app-shater; do
|
||||
grep -q "^CONFIG_PACKAGE_$p=m" .config || {
|
||||
echo "[apk-sdk] ERROR: $p is NOT selected after defconfig."
|
||||
echo " kconfig dropped it -> one of its DEPENDS is missing from the"
|
||||
echo " installed feeds (check the 'feeds install' step above)."; exit 10; }
|
||||
done
|
||||
|
||||
# Only our two nft kmods (+ whatever they themselves depend on) have any business
|
||||
# being selected here — a dozen at the very most. A count in the hundreds means an
|
||||
# ALL_KMODS-style mass-select crept back in, and the run would spend ~40 min
|
||||
# packing the kernel before dying on `Disk quota exceeded`. Fail now instead.
|
||||
[ "$kmods" -le 200 ] || {
|
||||
echo "[apk-sdk] ERROR: $kmods kmod packages selected — that is the whole kernel."
|
||||
echo " Aborting before this fills the runner's disk. Two causes are"
|
||||
echo " possible, and the lines below tell them apart:"
|
||||
echo " (a) the mass-select is back on -> a CONFIG_ALL* line reads =y;"
|
||||
echo " (b) the second pass did not take -> ALL* are 'is not set' but the"
|
||||
echo " kmods returned anyway, i.e. the per-kmod 'default m' from the"
|
||||
echo " SDK's generated Config-build.in outlived our explicit 'n'."
|
||||
grep -E '^(# )?CONFIG_ALL(_KMODS|_NONSHARED)?[ =]' .config | sed 's/^/ /' || true
|
||||
echo " first few kmods still selected:"
|
||||
grep -m5 '^CONFIG_PACKAGE_kmod.*=m' .config | sed 's/^/ /' || true
|
||||
exit 11; }
|
||||
|
||||
for p in shaterd shater-core byedpi luci-app-shater; do
|
||||
echo "[apk-sdk] === build $p ==="
|
||||
make "package/$p/compile" V=s -j"$(nproc)"
|
||||
done
|
||||
|
||||
# What the build actually cost on disk. The runner's 64 GB ZFS quota is the
|
||||
# binding constraint on this lane, so record it while the tree still exists.
|
||||
echo "[apk-sdk] disk usage after compile:"
|
||||
du -sh build_dir staging_dir bin 2>/dev/null || true
|
||||
df -h /home/build || true
|
||||
|
||||
# A 25.12 apk-SDK must emit .apk — finding only .ipk means a wrong SDK was fed in.
|
||||
anyapk=$(find bin -type f -name '*.apk' | wc -l)
|
||||
[ "$anyapk" -gt 0 ] || {
|
||||
@@ -157,6 +397,25 @@ done
|
||||
[ "$found" -ge 4 ] || { echo "[apk-sdk] ERROR: expected >=4 of OUR .apk, collected $found"; echo "[apk-sdk] (all .apk under bin/:)"; find bin -type f -name '*.apk' | head -20; exit 6; }
|
||||
echo "[apk-sdk] collected $found of our .apk"
|
||||
|
||||
# --- assert the tag-derived version actually reached the packages -------------
|
||||
# B4's failure mode is a wrong-but-plausible version shipping silently, so the
|
||||
# env -> make hand-off is verified, not trusted: each of our three tag-versioned
|
||||
# packages must be named `<name>-<ver>-r<rel>.apk`. byedpi is excluded on purpose
|
||||
# (it carries upstream ByeDPI's own version). This runs BEFORE `apk mkndx`, so a
|
||||
# stale version can never even reach the index.
|
||||
if [ -n "${SHATER_PKG_VERSION:-}" ] && [ -n "${SHATER_PKG_RELEASE:-}" ]; then
|
||||
want="${SHATER_PKG_VERSION}-r${SHATER_PKG_RELEASE}"
|
||||
for p in shaterd shater-core luci-app-shater; do
|
||||
[ -f "$OUT/${p}-${want}.apk" ] || {
|
||||
echo "[apk-sdk] ERROR: $p was not built as version '$want'."
|
||||
echo " SHATER_PKG_VERSION/SHATER_PKG_RELEASE did not reach the package"
|
||||
echo " Makefile — the build would have shipped a stale version (bug B4)."
|
||||
echo "[apk-sdk] collected:"; ls -1 "$OUT" | sed 's/^/ /'
|
||||
exit 12; }
|
||||
done
|
||||
echo "[apk-sdk] version check OK — our 3 packages are $want"
|
||||
fi
|
||||
|
||||
# --- index + sign: exactly how the OpenWrt 25.12 buildsystem does it ---------
|
||||
# apk mkndx --root T --keys-dir T [--sign key] --allow-untrusted \
|
||||
# --output packages.adb *.apk
|
||||
@@ -188,7 +447,7 @@ INNER
|
||||
chmod 0644 /home/build/inner.sh
|
||||
|
||||
su build -s /bin/bash -c \
|
||||
"ARCH='$ARCH' REPO='$REPO' OUT='$OUT' SDKDIR='$SDKDIR' KEYFILE='${KEYFILE:-}' DL_DIR='${DL_DIR:-}' bash /home/build/inner.sh"
|
||||
"ARCH='$ARCH' REPO='$REPO' OUT='$OUT' SDKDIR='$SDKDIR' KEYFILE='${KEYFILE:-}' DL_DIR='${DL_DIR:-}' FEEDS_CACHE='${FEEDS_CACHE:-}' SHATER_PKG_VERSION='${SHATER_PKG_VERSION:-}' SHATER_PKG_RELEASE='${SHATER_PKG_RELEASE:-}' bash /home/build/inner.sh"
|
||||
|
||||
chmod -R a+rwX "$OUT" 2>/dev/null || true
|
||||
echo "[apk-sdk] OK arch=$ARCH — apk feed dir:"
|
||||
|
||||
+46
-1
@@ -27,6 +27,13 @@ OUT="${OUT:?OUT env required}"
|
||||
mkdir -p "$OUT"
|
||||
|
||||
echo "[sdk] arch=$ARCH repo=$REPO out=$OUT"
|
||||
# Package version, derived from the git tag by ci/version.sh and handed in by
|
||||
# ci/build-feed.sh. openwrt/{shaterd,shater-core,luci-app-shater}/Makefile read
|
||||
# these straight out of the environment ($(if $(SHATER_PKG_VERSION),...)); make
|
||||
# imports every environment variable as a variable, and it propagates through
|
||||
# `make package/<p>/compile`, the metadata dump and the sub-makes alike.
|
||||
# byedpi deliberately keeps its own upstream version (see its Makefile).
|
||||
echo "[sdk] package version: ${SHATER_PKG_VERSION:-<unset -> Makefile fallback>}-r${SHATER_PKG_RELEASE:-?}"
|
||||
test -f "$REPO/openwrt/shaterd/Makefile" || {
|
||||
echo "[sdk] ERROR: feed not mounted ($REPO/openwrt/shaterd/Makefile missing)"; ls -la "$REPO" || true; exit 9; }
|
||||
|
||||
@@ -50,8 +57,26 @@ grep -q '^src-link shater ' feeds.conf || echo "src-link shater $REPO/openwrt" >
|
||||
# luci, packages, routing, telephony). We need `luci` for feeds/luci/luci.mk and
|
||||
# `base`/`packages` for the runtime deps (kmod-nft-tproxy, kmod-nft-socket,
|
||||
# ip-full, rpcd, luci-base) to resolve.
|
||||
#
|
||||
# Persistent feeds checkouts: $FEEDS_CACHE (a workspace dir the runner restores
|
||||
# via actions/cache, shared into this container via --volumes-from) replaces
|
||||
# the SDK's ephemeral feeds/ dir, so `feeds update` git-fetches deltas instead
|
||||
# of re-cloning base+packages+luci every run (~7 min on the runner's slow
|
||||
# github.com link). Correctness-safe: update always checks out feeds.conf's
|
||||
# pinned revisions; if it ever fails on a cached checkout (e.g. a force-pushed
|
||||
# upstream), the cache is wiped and the update retried with fresh clones.
|
||||
if [ -n "${FEEDS_CACHE:-}" ] && mkdir -p "$FEEDS_CACHE" 2>/dev/null; then
|
||||
rm -rf feeds
|
||||
ln -s "$FEEDS_CACHE" feeds
|
||||
echo "[sdk] feeds/ -> $FEEDS_CACHE (persistent cache)"
|
||||
fi
|
||||
echo "[sdk] feeds update -a"
|
||||
./scripts/feeds update -a
|
||||
if ! ./scripts/feeds update -a; then
|
||||
[ -L feeds ] || { echo "[sdk] ERROR: feeds update failed"; exit 8; }
|
||||
echo "[sdk] WARNING: feeds update failed on cached checkouts — wiping cache, cloning fresh"
|
||||
find "$FEEDS_CACHE" -mindepth 1 -maxdepth 1 -exec rm -rf {} + 2>/dev/null || true
|
||||
./scripts/feeds update -a
|
||||
fi
|
||||
|
||||
echo "[sdk] feeds install (prefer shater feed)"
|
||||
./scripts/feeds install -p shater shaterd shater-core byedpi luci-app-shater
|
||||
@@ -93,6 +118,26 @@ for p in shaterd shater-core byedpi luci-app-shater; do
|
||||
done
|
||||
done
|
||||
[ "$found" -ge 4 ] || { echo "[sdk] ERROR: expected >=4 of OUR .ipk, collected $found"; echo "[sdk] (all .ipk under bin/:)"; find bin -type f -name '*.ipk' | head -20; exit 4; }
|
||||
|
||||
# --- assert the tag-derived version actually reached the packages -------------
|
||||
# The whole point of B4 is that a WRONG-but-plausible version ships silently. The
|
||||
# env -> make hand-off has several layers (docker -e, make's env import, the
|
||||
# metadata dump), so verify the result instead of trusting it: every one of our
|
||||
# three tag-versioned packages must be named `<name>_<ver>-r<rel>_<arch>.ipk`.
|
||||
# byedpi is excluded on purpose — it keeps upstream ByeDPI's own version.
|
||||
if [ -n "${SHATER_PKG_VERSION:-}" ] && [ -n "${SHATER_PKG_RELEASE:-}" ]; then
|
||||
want="${SHATER_PKG_VERSION}-r${SHATER_PKG_RELEASE}"
|
||||
for p in shaterd shater-core luci-app-shater; do
|
||||
ls "$OUT/${p}_${want}_"*.ipk >/dev/null 2>&1 || {
|
||||
echo "[sdk] ERROR: $p was not built as version '$want'."
|
||||
echo " SHATER_PKG_VERSION/SHATER_PKG_RELEASE did not reach the package"
|
||||
echo " Makefile — the build would have shipped a stale version (bug B4)."
|
||||
echo "[sdk] collected:"; ls -1 "$OUT" | sed 's/^/ /'
|
||||
exit 12; }
|
||||
done
|
||||
echo "[sdk] version check OK — our 3 packages are $want"
|
||||
fi
|
||||
|
||||
chmod -R a+rwX "$OUT" 2>/dev/null || true
|
||||
echo "[sdk] OK arch=$ARCH — collected $found of our .ipk:"
|
||||
ls -l "$OUT"
|
||||
|
||||
Executable
+133
@@ -0,0 +1,133 @@
|
||||
#!/bin/sh
|
||||
# ci/version.sh — the SINGLE source of truth for "what version is this build?".
|
||||
#
|
||||
# WHY THIS EXISTS (bug B4)
|
||||
# -----------------------
|
||||
# PKG_VERSION/PKG_RELEASE used to be hand-written literals in the four package
|
||||
# Makefiles, and nobody remembered to bump them: v0.2.2 … v0.2.6 all shipped as
|
||||
# `shaterd 0.2.0-r3` with DIFFERENT binaries inside (v0.2.6's ELF is 5 491 616 B
|
||||
# vs r2's 5 488 336 B). Since both opkg and apk offer an upgrade only when the
|
||||
# feed's version string differs from the installed one, `apk update` saw nothing
|
||||
# new and the routers could not be updated through the normal path at all.
|
||||
#
|
||||
# So the version is now DERIVED, in CI, from the git tag, and the package
|
||||
# Makefiles only carry a fallback for manual/offline builds.
|
||||
#
|
||||
# THE SCHEME
|
||||
# ----------
|
||||
# tag push `vX.Y.Z` -> PKG_VERSION=X.Y.Z PKG_RELEASE=1
|
||||
# any other build -> PKG_VERSION=X.Y.Z of the NEAREST reachable tag,
|
||||
# (workflow_dispatch, PKG_RELEASE=<commits since that tag> + 1
|
||||
# rolling `latest`)
|
||||
# no tag / no git at all -> PKG_VERSION=0.0.0 PKG_RELEASE=1 (+ warning)
|
||||
#
|
||||
# Both managers compare `<upstream>-r<rel>` the same way: the dotted upstream
|
||||
# part first (numerically, component by component), the `r<rel>` only as a
|
||||
# tie-break. Verified against the real tools, not from memory:
|
||||
# apk-tools 3.0.3 (`apk version -t`) and apk-tools 2.14.6:
|
||||
# 0.2.6-r1 > 0.2.0-r3 0.2.6-r12 > 0.2.6-r1
|
||||
# 0.2.7-r1 > 0.2.6-r12 0.0.0-r1 < 0.2.0-r3
|
||||
# opkg 38eccbb1 from openwrt/rootfs:x86-64-24.10.4 (`opkg compare-versions`):
|
||||
# identical results (opkg implements the Debian algorithm).
|
||||
# That is exactly the ordering this scheme needs:
|
||||
# * a release always outranks every rolling build that preceded it
|
||||
# (0.2.7-r1 > 0.2.6-rN for any N — the dotted part decides), and
|
||||
# * rolling builds between two releases grow monotonically (r2 < r10 < r11),
|
||||
# so a rolling build can never look newer than the next release, and the
|
||||
# `latest` feed still moves forward on every dispatch.
|
||||
#
|
||||
# +1 on the commit count (rather than the raw count) only avoids `-r0` and makes
|
||||
# a dispatch build of the tagged commit itself identical to the release build of
|
||||
# that same commit — which is the truth: same tree, same binary.
|
||||
#
|
||||
# `byedpi` is deliberately NOT versioned from our tag — see openwrt/byedpi/Makefile.
|
||||
#
|
||||
# USAGE
|
||||
# ci/version.sh # or --env: eval-able / $GITHUB_ENV-able lines
|
||||
# ci/version.sh --pkg-version # X.Y.Z
|
||||
# ci/version.sh --pkg-release # R
|
||||
# ci/version.sh --binary # vX.Y.Z-rR[-g<sha>] for constant.Version
|
||||
#
|
||||
# Env:
|
||||
# SHATER_REF / GITHUB_REF when it is `refs/tags/<tag>` that tag wins and no
|
||||
# git history is needed (the tag-push path is exact
|
||||
# even on a shallow checkout).
|
||||
set -eu
|
||||
|
||||
REPO="$(CDPATH='' cd -- "$(dirname -- "$0")/.." && pwd)"
|
||||
|
||||
TAG=""
|
||||
EXACT=0
|
||||
N=0
|
||||
SHA=""
|
||||
|
||||
# --- 1) an explicit tag ref is authoritative (and needs no git) --------------
|
||||
REF="${SHATER_REF:-${GITHUB_REF:-}}"
|
||||
case "$REF" in
|
||||
refs/tags/*) TAG="${REF#refs/tags/}"; EXACT=1 ;;
|
||||
esac
|
||||
|
||||
# --- 2) otherwise ask git for the nearest reachable release tag --------------
|
||||
# `--match 'v[0-9]*'` keeps non-release tags (latest, sdk-cache, apk-latest-*,
|
||||
# musl-toolchain-cache) out. This repo is a sing-box FORK and therefore also
|
||||
# carries upstream's v1.x tags — `git describe` picks the CLOSEST tag by commit
|
||||
# distance, so our own v0.2.x (a handful of commits back) always wins over
|
||||
# upstream's v1.x (thousands of commits back). The tag it picked is logged
|
||||
# below, so a surprise is visible in the CI log rather than silently shipped.
|
||||
if [ "$EXACT" -eq 0 ]; then
|
||||
if D="$(git -C "$REPO" describe --tags --long --match 'v[0-9]*' 2>/dev/null)"; then
|
||||
# `v0.2.6-1-g02c266188` -> TAG=v0.2.6 N=1 SHA=g02c266188.
|
||||
# `%` strips the SHORTEST matching suffix, so a tag that itself contains a
|
||||
# dash (`v0.2.0-healthplan`) survives intact.
|
||||
TAG="${D%-*-g*}"
|
||||
REST="${D#"$TAG"-}"
|
||||
N="${REST%%-*}"
|
||||
SHA="${REST#*-}"
|
||||
if [ "$N" -eq 0 ]; then EXACT=1; fi
|
||||
fi
|
||||
fi
|
||||
|
||||
# --- 3) tag -> numeric PKG_VERSION ------------------------------------------
|
||||
# Keep the leading dotted-numeric run only: `v0.2.0-healthplan` -> `0.2.0`.
|
||||
VER=""
|
||||
if [ -n "$TAG" ]; then
|
||||
VER="$(printf '%s' "${TAG#v}" | sed -n 's/^\([0-9][0-9.]*\).*/\1/p' | sed 's/\.*$//')"
|
||||
fi
|
||||
|
||||
if [ -z "$VER" ]; then
|
||||
# No release tag anywhere (shallow clone with no tags, a tarball export, a
|
||||
# fresh fork). 0.0.0 is BELOW every version we have ever published, so such a
|
||||
# build can never masquerade as an upgrade on a real router; the commit count
|
||||
# still makes successive dev builds distinguishable.
|
||||
VER="0.0.0"
|
||||
EXACT=0
|
||||
N="$(git -C "$REPO" rev-list --count HEAD 2>/dev/null || echo 0)"
|
||||
SHA="$(git -C "$REPO" rev-parse --short HEAD 2>/dev/null || echo '')"
|
||||
[ -z "$SHA" ] || SHA="g$SHA"
|
||||
echo "[version] WARNING: no reachable vX.Y.Z tag (and/or no git) -> $VER" >&2
|
||||
fi
|
||||
|
||||
# --- 4) PKG_RELEASE + the string stamped into the binary --------------------
|
||||
if [ "$EXACT" -eq 1 ]; then
|
||||
REL=1
|
||||
FULL="v${VER}-r${REL}"
|
||||
else
|
||||
REL=$((N + 1))
|
||||
FULL="v${VER}-r${REL}${SHA:+-$SHA}"
|
||||
fi
|
||||
|
||||
echo "[version] tag='${TAG:-none}' commits_since=$N exact=$EXACT -> ${VER}-r${REL} (binary: $FULL)" >&2
|
||||
|
||||
case "${1:---env}" in
|
||||
--env|"")
|
||||
printf 'SHATER_PKG_VERSION=%s\n' "$VER"
|
||||
printf 'SHATER_PKG_RELEASE=%s\n' "$REL"
|
||||
printf 'SHATER_VERSION=%s\n' "$FULL"
|
||||
;;
|
||||
--pkg-version) printf '%s\n' "$VER" ;;
|
||||
--pkg-release) printf '%s\n' "$REL" ;;
|
||||
--binary|--version) printf '%s\n' "$FULL" ;;
|
||||
*)
|
||||
echo "usage: $0 [--env|--pkg-version|--pkg-release|--binary]" >&2
|
||||
exit 2 ;;
|
||||
esac
|
||||
@@ -0,0 +1,84 @@
|
||||
// lx:begin health-board
|
||||
|
||||
// Health board (plan §5.A): failure tracking and verdict computation on top of
|
||||
// HistoryStorage. Successes keep flowing through StoreURLTestHistory; failures are
|
||||
// recorded with MarkFailed instead of deleting the entry (deletion stays reserved
|
||||
// for nodes removed from the configuration), and consumers classify a tag at read
|
||||
// time with Verdict. One store, one truth: whoever learns about a death — the
|
||||
// group's own checker, the observatory, or a failed user dial — marks it here.
|
||||
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
)
|
||||
|
||||
// HealthVerdict classifies a stored history entry at read time.
|
||||
type HealthVerdict int
|
||||
|
||||
const (
|
||||
// VerdictUntested means nothing fresh enough is known either way: no entry,
|
||||
// or every observation is older than the caller's TTL.
|
||||
VerdictUntested HealthVerdict = iota
|
||||
// VerdictAlive means the newest fresh observation is a success.
|
||||
VerdictAlive
|
||||
// VerdictDead means the newest fresh observation is a failure.
|
||||
VerdictDead
|
||||
)
|
||||
|
||||
func (v HealthVerdict) String() string {
|
||||
switch v {
|
||||
case VerdictAlive:
|
||||
return "alive"
|
||||
case VerdictDead:
|
||||
return "dead"
|
||||
default:
|
||||
return "untested"
|
||||
}
|
||||
}
|
||||
|
||||
// MarkFailed records a failed probe or dial for tag: LastFail is set to now while
|
||||
// LastOK/Delay of an existing entry are preserved, so a node that once worked keeps
|
||||
// its last known latency for display. The entry is never deleted here — a marked
|
||||
// tag stays distinguishable from "never measured" (plan §2 Д5).
|
||||
func (s *HistoryStorage) MarkFailed(tag string) {
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.access.Lock()
|
||||
updated := &adapter.URLTestHistory{LastFail: time.Now()}
|
||||
if previous := s.delayHistory[tag]; previous != nil {
|
||||
updated.LastOK = previous.LastOK
|
||||
updated.Delay = previous.Delay
|
||||
}
|
||||
s.delayHistory[tag] = updated
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
}
|
||||
|
||||
// Verdict classifies tag against the wall clock: alive when the last success is
|
||||
// newer than the last failure and younger than ttl, dead when the last failure is
|
||||
// newer than the last success and younger than ttl, untested otherwise.
|
||||
func (s *HistoryStorage) Verdict(tag string, ttl time.Duration) HealthVerdict {
|
||||
return s.VerdictAt(tag, ttl, time.Now())
|
||||
}
|
||||
|
||||
// VerdictAt is Verdict against an explicit clock, for deterministic tests.
|
||||
func (s *HistoryStorage) VerdictAt(tag string, ttl time.Duration, now time.Time) HealthVerdict {
|
||||
history := s.LoadURLTestHistory(tag)
|
||||
if history == nil {
|
||||
return VerdictUntested
|
||||
}
|
||||
switch {
|
||||
case history.LastOK.After(history.LastFail) && now.Sub(history.LastOK) < ttl:
|
||||
return VerdictAlive
|
||||
case history.LastFail.After(history.LastOK) && now.Sub(history.LastFail) < ttl:
|
||||
return VerdictDead
|
||||
default:
|
||||
return VerdictUntested
|
||||
}
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -59,6 +59,17 @@ func (s *HistoryStorage) DeleteURLTestHistory(tag string) {
|
||||
|
||||
func (s *HistoryStorage) StoreURLTestHistory(tag string, history *adapter.URLTestHistory) {
|
||||
s.access.Lock()
|
||||
// lx:begin health-board
|
||||
// Health board (plan §5.A): overwriting an entry with a fresh success must not
|
||||
// erase the recorded failure — Verdict compares LastOK against LastFail, so
|
||||
// dropping LastFail here would forge an eternal "alive". Callers only ever set
|
||||
// LastOK/Delay on success; a caller that deliberately sets LastFail wins.
|
||||
if history.LastFail.IsZero() {
|
||||
if previous := s.delayHistory[tag]; previous != nil {
|
||||
history.LastFail = previous.LastFail
|
||||
}
|
||||
}
|
||||
// lx:end health-board
|
||||
s.delayHistory[tag] = history
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
// lx:begin health-board
|
||||
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
)
|
||||
|
||||
// TestVerdictTable drives VerdictAt through every classification the health board
|
||||
// must produce (plan §5.A): the newest FRESH observation decides, and anything
|
||||
// older than the TTL decays to untested.
|
||||
func TestVerdictTable(t *testing.T) {
|
||||
const ttl = 10 * time.Minute
|
||||
now := time.Now()
|
||||
at := func(ago time.Duration) time.Time { return now.Add(-ago) }
|
||||
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
history *adapter.URLTestHistory // nil = no entry stored
|
||||
want HealthVerdict
|
||||
}{
|
||||
{
|
||||
name: "fresh success is alive",
|
||||
history: &adapter.URLTestHistory{LastOK: at(time.Minute), Delay: 42},
|
||||
want: VerdictAlive,
|
||||
},
|
||||
{
|
||||
name: "fresh failure is dead",
|
||||
history: &adapter.URLTestHistory{LastFail: at(time.Minute)},
|
||||
want: VerdictDead,
|
||||
},
|
||||
{
|
||||
name: "stale success (older than TTL) decays to untested",
|
||||
history: &adapter.URLTestHistory{LastOK: at(ttl + time.Minute), Delay: 42},
|
||||
want: VerdictUntested,
|
||||
},
|
||||
{
|
||||
name: "failure after success is dead",
|
||||
history: &adapter.URLTestHistory{
|
||||
LastOK: at(2 * time.Minute),
|
||||
Delay: 42,
|
||||
LastFail: at(time.Minute),
|
||||
},
|
||||
want: VerdictDead,
|
||||
},
|
||||
{
|
||||
name: "success after failure is alive",
|
||||
history: &adapter.URLTestHistory{
|
||||
LastOK: at(time.Minute),
|
||||
Delay: 42,
|
||||
LastFail: at(2 * time.Minute),
|
||||
},
|
||||
want: VerdictAlive,
|
||||
},
|
||||
{
|
||||
name: "no entry is untested",
|
||||
history: nil,
|
||||
want: VerdictUntested,
|
||||
},
|
||||
{
|
||||
name: "empty entry (both timestamps zero) is untested",
|
||||
history: &adapter.URLTestHistory{},
|
||||
want: VerdictUntested,
|
||||
},
|
||||
{
|
||||
name: "stale failure (older than TTL) decays to untested",
|
||||
history: &adapter.URLTestHistory{
|
||||
LastOK: at(2 * ttl),
|
||||
Delay: 42,
|
||||
LastFail: at(ttl + time.Minute),
|
||||
},
|
||||
want: VerdictUntested,
|
||||
},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
if tc.history != nil {
|
||||
storage.delayHistory[tag] = tc.history
|
||||
}
|
||||
if got := storage.VerdictAt(tag, ttl, now); got != tc.want {
|
||||
t.Fatalf("VerdictAt(%+v) = %v, want %v", tc.history, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestMarkFailedPreservesSuccess verifies MarkFailed records the failure without
|
||||
// deleting the entry or losing the last known success/latency (plan §2 Д5).
|
||||
func TestMarkFailedPreservesSuccess(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
lastOK := time.Now().Add(-time.Minute)
|
||||
storage.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: lastOK, Delay: 42})
|
||||
|
||||
storage.MarkFailed(tag)
|
||||
|
||||
history := storage.LoadURLTestHistory(tag)
|
||||
if history == nil {
|
||||
t.Fatal("MarkFailed deleted the entry; it must only set LastFail")
|
||||
}
|
||||
if !history.LastOK.Equal(lastOK) || history.Delay != 42 {
|
||||
t.Fatalf("MarkFailed lost the last success: got %+v", history)
|
||||
}
|
||||
if history.LastFail.IsZero() || !history.LastFail.After(lastOK) {
|
||||
t.Fatalf("MarkFailed did not record a fresh failure: got %+v", history)
|
||||
}
|
||||
if got := storage.Verdict(tag, 10*time.Minute); got != VerdictDead {
|
||||
t.Fatalf("Verdict after MarkFailed = %v, want %v", got, VerdictDead)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMarkFailedWithoutEntry verifies MarkFailed on a never-measured tag creates a
|
||||
// failure-only entry (dead, not untested) instead of doing nothing.
|
||||
func TestMarkFailedWithoutEntry(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
|
||||
storage.MarkFailed(tag)
|
||||
|
||||
history := storage.LoadURLTestHistory(tag)
|
||||
if history == nil {
|
||||
t.Fatal("MarkFailed on an unknown tag must create an entry")
|
||||
}
|
||||
if !history.LastOK.IsZero() || history.Delay != 0 {
|
||||
t.Fatalf("MarkFailed invented a success: got %+v", history)
|
||||
}
|
||||
if got := storage.Verdict(tag, 10*time.Minute); got != VerdictDead {
|
||||
t.Fatalf("Verdict = %v, want %v", got, VerdictDead)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStorePreservesLastFail verifies a success overwrite keeps the previously
|
||||
// recorded failure timestamp, so Verdict can still order the two observations.
|
||||
func TestStorePreservesLastFail(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
|
||||
storage.MarkFailed(tag)
|
||||
failedAt := storage.LoadURLTestHistory(tag).LastFail
|
||||
|
||||
// A strictly later success: with the wall clock, LastOK could land on the SAME
|
||||
// tick as LastFail (Windows clock granularity) and read as untested (ties are
|
||||
// deliberately not alive — the ordering is strict).
|
||||
storage.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: failedAt.Add(time.Second), Delay: 7})
|
||||
|
||||
history := storage.LoadURLTestHistory(tag)
|
||||
if !history.LastFail.Equal(failedAt) {
|
||||
t.Fatalf("StoreURLTestHistory dropped LastFail: got %+v, want LastFail=%v", history, failedAt)
|
||||
}
|
||||
if got := storage.Verdict(tag, 10*time.Minute); got != VerdictAlive {
|
||||
t.Fatalf("Verdict after success-over-failure = %v, want %v", got, VerdictAlive)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNilStorageVerdict guards the nil-receiver contract shared with
|
||||
// LoadURLTestHistory: reads on a nil store degrade to untested, never panic.
|
||||
func TestNilStorageVerdict(t *testing.T) {
|
||||
var storage *HistoryStorage
|
||||
if got := storage.Verdict("node", time.Minute); got != VerdictUntested {
|
||||
t.Fatalf("nil storage Verdict = %v, want %v", got, VerdictUntested)
|
||||
}
|
||||
storage.MarkFailed("node") // must be a no-op, not a panic
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -2,6 +2,9 @@ package daemon
|
||||
|
||||
import (
|
||||
"context"
|
||||
// lx:begin sec-oomgate
|
||||
"sync"
|
||||
// lx:end sec-oomgate
|
||||
"time"
|
||||
"unsafe"
|
||||
|
||||
@@ -19,6 +22,10 @@ type ManagedService struct {
|
||||
handler ManagedHandler
|
||||
debug bool
|
||||
oomReporter oomkiller.OOMReporter
|
||||
// lx:begin sec-oomgate
|
||||
oomReportMu sync.Mutex
|
||||
oomReportLast time.Time
|
||||
// lx:end sec-oomgate
|
||||
}
|
||||
|
||||
type ManagedServiceOptions struct {
|
||||
@@ -90,6 +97,18 @@ func (s *ManagedService) TriggerOOMReport(ctx context.Context, _ *emptypb.Empty)
|
||||
if s.oomReporter == nil {
|
||||
return nil, status.Error(codes.Unavailable, "OOM reporter not available")
|
||||
}
|
||||
// lx:begin sec-oomgate
|
||||
// Rate-limit operator-triggered reports to at most one per minute: each write
|
||||
// dumps process state + the config snapshot (secrets) to disk, so an
|
||||
// authenticated client must not be able to spin it in a tight loop.
|
||||
s.oomReportMu.Lock()
|
||||
if !s.oomReportLast.IsZero() && time.Since(s.oomReportLast) < time.Minute {
|
||||
s.oomReportMu.Unlock()
|
||||
return nil, status.Error(codes.ResourceExhausted, "OOM report rate-limited (max 1/min)")
|
||||
}
|
||||
s.oomReportLast = time.Now()
|
||||
s.oomReportMu.Unlock()
|
||||
// lx:end sec-oomgate
|
||||
return &emptypb.Empty{}, s.oomReporter.WriteReport(memory.Total())
|
||||
}
|
||||
|
||||
|
||||
+7
-1
@@ -2,6 +2,9 @@ package daemon
|
||||
|
||||
import (
|
||||
"context"
|
||||
// lx:begin sec-consttime
|
||||
"crypto/subtle"
|
||||
// lx:end sec-consttime
|
||||
"strings"
|
||||
|
||||
"google.golang.org/grpc"
|
||||
@@ -59,8 +62,11 @@ func authenticate(ctx context.Context, secret string) error {
|
||||
return status.Error(codes.Unauthenticated, "missing authorization")
|
||||
}
|
||||
token, isBearer := strings.CutPrefix(values[0], "Bearer ")
|
||||
if !isBearer || token != secret {
|
||||
// lx:begin sec-consttime
|
||||
// Constant-time compare: a plain != leaks the secret via response timing.
|
||||
if !isBearer || subtle.ConstantTimeCompare([]byte(token), []byte(secret)) != 1 {
|
||||
return status.Error(codes.Unauthenticated, "invalid authorization")
|
||||
}
|
||||
// lx:end sec-consttime
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -489,7 +489,7 @@ func (s *StartedService) readGroups() *Groups {
|
||||
item.Tag = itemTag
|
||||
item.Type = itemOutbound.Type()
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(itemOutbound)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix() // lx: health board §5.A — Time renamed to LastOK
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
g.Items = append(g.Items, &item)
|
||||
@@ -620,8 +620,8 @@ func (s *StartedService) URLTest(ctx context.Context, request *URLTestRequest) (
|
||||
historyStorage.DeleteURLTestHistory(outboundTag)
|
||||
} else {
|
||||
historyStorage.StoreURLTestHistory(outboundTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: t,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: t,
|
||||
})
|
||||
}
|
||||
return nil, nil
|
||||
@@ -1063,7 +1063,7 @@ func (s *StartedService) SubscribeOutbounds(_ *emptypb.Empty, server grpc.Server
|
||||
Type: ob.Type(),
|
||||
}
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(ob)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix() // lx: health board §5.A — Time renamed to LastOK
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
list.Outbounds = append(list.Outbounds, item)
|
||||
@@ -1074,7 +1074,7 @@ func (s *StartedService) SubscribeOutbounds(_ *emptypb.Empty, server grpc.Server
|
||||
Type: ep.Type(),
|
||||
}
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(ep)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix() // lx: health board §5.A — Time renamed to LastOK
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
list.Outbounds = append(list.Outbounds, item)
|
||||
|
||||
@@ -86,8 +86,8 @@ func (s *StartedService) URLTestOutbound(ctx context.Context, request *URLTestOu
|
||||
return &URLTestOutboundResponse{Error: err.Error()}, nil
|
||||
}
|
||||
boxService.urlTestHistoryStorage.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: delay,
|
||||
LastOK: time.Now(),
|
||||
Delay: delay,
|
||||
})
|
||||
return &URLTestOutboundResponse{Delay: uint32(delay)}, nil
|
||||
}
|
||||
@@ -168,7 +168,7 @@ func (s *StartedService) GetOutbounds(ctx context.Context, empty *emptypb.Empty)
|
||||
appendItem := func(detour adapter.Outbound) {
|
||||
item := &GroupItem{Tag: detour.Tag(), Type: detour.Type()}
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(detour)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix()
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
list.Outbounds = append(list.Outbounds, item)
|
||||
|
||||
@@ -131,7 +131,9 @@ func (s *StartedService) StartTailscaleSSHSession(
|
||||
continue
|
||||
}
|
||||
go ssh.DiscardRequests(reqs)
|
||||
go s.forwardSSHAgentChannel(channel)
|
||||
// lx:begin sec-sshagent
|
||||
go s.forwardSSHAgentChannel(sessionCtx, channel)
|
||||
// lx:end sec-sshagent
|
||||
}
|
||||
}()
|
||||
}
|
||||
@@ -313,7 +315,8 @@ func (s *StartedService) StartTailscaleSSHSession(
|
||||
return nil
|
||||
}
|
||||
|
||||
func (s *StartedService) forwardSSHAgentChannel(channel ssh.Channel) {
|
||||
// lx:begin sec-sshagent
|
||||
func (s *StartedService) forwardSSHAgentChannel(ctx context.Context, channel ssh.Channel) {
|
||||
defer channel.Close()
|
||||
fd, err := s.handler.ConnectSSHAgent()
|
||||
if err != nil {
|
||||
@@ -326,15 +329,33 @@ func (s *StartedService) forwardSSHAgentChannel(channel ssh.Channel) {
|
||||
return
|
||||
}
|
||||
defer conn.Close()
|
||||
|
||||
// The ssh-agent conn stays blocked in Read while idle, so io.Copy(channel,
|
||||
// conn) never returns on its own — without this it leaks a goroutine + the
|
||||
// agent fd for every closed session. Cancelling on either copy finishing (or
|
||||
// on the session ctx) closes both ends, unblocking the peer copy. Both Close
|
||||
// calls are idempotent with the deferred ones above.
|
||||
ctx, cancel := context.WithCancel(ctx)
|
||||
defer cancel()
|
||||
go func() {
|
||||
<-ctx.Done()
|
||||
conn.Close()
|
||||
channel.Close()
|
||||
}()
|
||||
|
||||
var wg sync.WaitGroup
|
||||
wg.Add(2)
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
io.Copy(conn, channel)
|
||||
cancel()
|
||||
}()
|
||||
go func() {
|
||||
defer wg.Done()
|
||||
io.Copy(channel, conn)
|
||||
cancel()
|
||||
}()
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// lx:end sec-sshagent
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 419 KiB |
@@ -10,6 +10,81 @@ tracks only the fork. Versions are tagged `vX.Y.Z-lx.N`; releases are built by
|
||||
`lx-release.yml`. Tags carrying an `-rc.N` / `-alpha.N` / `-beta.N` suffix publish
|
||||
as GitHub **pre-releases** and never become "Latest".
|
||||
|
||||
#### Unreleased (shater)
|
||||
|
||||
**Fork-layer + control-plane rework of proxy health** — ships with `shaterd`
|
||||
(the shater router daemon), not as an lx release tag; recorded here because the
|
||||
load-bearing half lives in fork zones (`common/urltest`, `protocol/group`).
|
||||
Routing rules now always get a **live** node under every group strategy, and
|
||||
the background prober measures **only what the rules can reach**.
|
||||
|
||||
* **Health board with real verdicts (`common/urltest`).** A history entry used
|
||||
to record only success (`{Time, Delay}`) and a failed check *deleted* it, so a
|
||||
dead member was indistinguishable from a never-measured one. The entry is now
|
||||
`{LastOK, Delay, LastFail}`; a failure marks, never deletes (deletion is
|
||||
reserved for removing a node from the config). The verdict is computed on
|
||||
read — `alive` (success fresher than failure and younger than TTL), `dead`
|
||||
(failure fresher, while it is fresh), `untested` (nothing fresh) — with
|
||||
TTL = max(3 × global probe interval, 10 min), so a stale success decays to
|
||||
`untested` instead of reading as alive forever. Everyone who learns of a death
|
||||
(the group checker, the observatory, a failed user dial) writes here;
|
||||
selection, balancer slots and the panel read here. The private `engine.dead`
|
||||
overlay that only the panel could see is gone.
|
||||
|
||||
* **Selection is alive-only, and a failed dial retries (`protocol/group`).**
|
||||
least ping ranks by delay **among alive members only**, falls back to untested
|
||||
members in config order, and only then to first-by-config; round_robin /
|
||||
random / failover slot liveness is fed by board verdicts (TTL included), not
|
||||
by "a history entry exists". A failed user dial marks the member dead on the
|
||||
board and transparently re-picks — at most 3 candidates per connection, within
|
||||
the dial deadline (UDP retries only until the first send) — so the connection
|
||||
survives as long as any member is alive. A failed group check marks-fail
|
||||
instead of deleting; an immediate first check on PostStart shrinks the
|
||||
cold-start blind window to seconds. `selector` (manual pin) semantics and the
|
||||
SPEC 019 slot invariants are untouched. When a whole group dies, its rule
|
||||
blocks fail-closed — no silent fallback.
|
||||
|
||||
* **Observatory replaces the background sweep (`shater/engine`).** The probe
|
||||
plan is derived from the applied config by reachability — enabled-rule targets
|
||||
(plus `Final` and DNS detours) → groups → members / egress copies; a chain is
|
||||
probed end-to-end by dialing its exit tag through the whole hop path (the Xray
|
||||
observatory-through-`proxySettings` equivalent; chains were previously not
|
||||
probed at all), and chain group-hop members are probed through their path
|
||||
prefix. Schedule: 10 s tick, batch 24, concurrency 12, 5 s probe timeout; a
|
||||
tag refreshes on the global probe interval; a freshness gate skips tags the
|
||||
active group's own checker already measures; the cursor survives no-op
|
||||
reconciles, and the first pass after apply is immediate. Every alive↔dead
|
||||
flip is logged (info, tag + probe/dial reason) — the diagnostic trail for a
|
||||
dead chain hop and for flapping. The `GroupHealth` toggle now gates the
|
||||
observatory.
|
||||
|
||||
* **Probe URL / Interval are global-only.** The per-group `ProbeURL` /
|
||||
`ProbeInterval` overrides are deleted (model, UCI, generate, panel); probing
|
||||
is configured solely by the global Settings fields, which also drive the
|
||||
native group checker and the exit test (failover keeps its 30 s default
|
||||
interval). With one URL all delay measurements are comparable, and the "one
|
||||
node in two groups with different URLs" ambiguity disappears. Old UCI configs
|
||||
still carrying the options parse silently and drain on the next write.
|
||||
|
||||
* **Subscription nodes move out of UCI.** Each subscription's nodes are cached
|
||||
in `/etc/shater/subs/<name>.json` (atomic tmp+rename; tmpfs fallback
|
||||
`/tmp/shater-subs` when the overlay cannot take the write); UCI keeps only
|
||||
hand-added nodes and the `subscription` sections, so `sub update` rewrites its
|
||||
own cache file instead of the whole `/etc/config/shater`. Legacy `from_sub`
|
||||
nodes migrate into the cache on first read.
|
||||
|
||||
* **One manual test — the exit test, now for chains too.** The manual
|
||||
"Test all nodes" probe-all run is removed together with the sweep
|
||||
(`/api/nodes/test` returns 404). The remaining manual test is the per-target
|
||||
"Test" (delay + exit IP + country through the real path), extended from groups
|
||||
to chains via the chain exit tag.
|
||||
|
||||
* **"Unused" badge.** A group or chain reachable from no enabled routing rule is
|
||||
outside the observatory plan and reported `used:false`; the panel shows
|
||||
"unused" instead of health counters. Unreferenced nodes are deliberately never
|
||||
probed — they stay honestly `untested` until a rule references them, at which
|
||||
point the observatory covers them within seconds.
|
||||
|
||||
#### v1.14.0-lx.3
|
||||
|
||||
**Stable release** (published as "Latest", not a pre-release) — a promotion of
|
||||
|
||||
@@ -329,3 +329,163 @@ makes it affordable on the target hardware: measured on the testbed, StevenBlack
|
||||
2.4 MB of text became **80873 domains in a 491 KB `.srs`**, costing ~0.5 MB of a rootfs
|
||||
with ~33 MB free — so it ships in the production posture rather than being traded away
|
||||
for the 8 KB geosite ads list.
|
||||
|
||||
## D18 — Node health = a board with computed verdicts, NOT delete-on-failure + a dead-overlay
|
||||
Decided 2026-07-24 (proxy-health plan §5.A). Upstream `common/urltest` recorded
|
||||
only success (`{Time, Delay}`) and a failed check **deleted** the entry — a dead
|
||||
member was indistinguishable from a never-measured one. On top of that, deaths
|
||||
found by shater's own probing went to a private `engine.dead` overlay that only
|
||||
the panel read: selection never saw them, and an old success in history counted
|
||||
as "alive forever". Net effect, reproduced in the field: YouTube dead on a real
|
||||
device while the panel showed the group healthy.
|
||||
|
||||
**Decision: one health board, one source of truth.** The history entry becomes
|
||||
`{LastOK, Delay, LastFail}`; a failure *marks* (`MarkFailed`), never deletes —
|
||||
deletion is reserved for removing a node from the config. The verdict is a
|
||||
**method computed on read**, not a stored field: `alive` when the success is
|
||||
fresher than the failure and younger than TTL; `dead` while a fresher failure is
|
||||
itself fresh; `untested` otherwise, with TTL = max(3 × global probe interval,
|
||||
10 min). Everyone who learns of a death — the group's native checker, the
|
||||
observatory, a failed user dial — writes the same board; selection, balancer
|
||||
slots and the panel read the same board. The `engine.dead` overlay is deleted.
|
||||
- **Rejected: keep delete-on-failure and widen the overlay to selection.** Two
|
||||
stores of the same truth with undefined precedence; the overlay would need its
|
||||
own TTL/pruning and every reader would have to merge — exactly the
|
||||
divergence ("green panel, dead path") this decision exists to kill.
|
||||
- **Rejected: persist the board across reboots.** Measurements go stale faster
|
||||
than an overlay write is worth; after reboot everything is `untested` for
|
||||
seconds until the observatory's immediate first pass. In-memory, deliberately.
|
||||
- **Deferred, not rejected: Xray-style sliding-window stats / hysteresis.** The
|
||||
delay of the last success is enough to rank alive members in v1; the board
|
||||
keys and record shape leave room to widen without migration. Flapping is
|
||||
visible instead through the alive↔dead flip log (info) — the single
|
||||
diagnostic trail.
|
||||
Consequence: a stale success **decays** (alive → untested after TTL) instead of
|
||||
reading as alive forever, and a dead group blocks fail-closed — visible and
|
||||
alertable — rather than silently falling back.
|
||||
|
||||
## D19 — Background probing = an observatory driven by rule reachability, NOT a population sweep
|
||||
Decided 2026-07-24 (proxy-health plan §5.C, replaces `shater/engine/sweep.go`).
|
||||
The sweep probed the **whole population** round-robin — all nodes plus all
|
||||
per-group copies, used or not — yet chain copies were excluded entirely, so the
|
||||
one path users actually complained about ("rule → chain") was never measured.
|
||||
The manual "Test all nodes" run duplicated the same full-population walk on
|
||||
demand. On router budgets that is the wrong shape twice: work grows with the
|
||||
subscription size (~200 nodes), not with what the config uses.
|
||||
|
||||
**Decision: probe only what the rules can reach.** Following the Xray model
|
||||
(central observatory + balancers reading observations, §3 of the plan), a plan
|
||||
is built from the **applied** `option.Options` by reachability: enabled-rule
|
||||
targets (plus `Final` and DNS detours) → groups → members / egress copies; a
|
||||
chain is probed **end-to-end** by dialing its exit tag through the whole hop
|
||||
path (the Xray observatory-through-`proxySettings` equivalent); chain group-hop
|
||||
members are probed through their path prefix. Everything unreferenced stays
|
||||
`untested` and its group/chain is badged "unused" in the panel — so untested
|
||||
never looks like a health problem. A freshness gate skips tags an active
|
||||
group's own checker already measures; the cursor survives no-op reconciles
|
||||
(the sweep's release-blocker: cron reconciles must not restart the cycle);
|
||||
the manual probe-all is removed with the sweep.
|
||||
- **Rejected: keep the sweep.** Probes hundreds of unused nodes on a router
|
||||
budget and still misses chains; its "coverage" is what made per-node health
|
||||
look authoritative while the used path went unmeasured.
|
||||
- **Rejected: probe every chain hop individually.** Multiplied probe traffic
|
||||
for diagnostics that never affects selection — no choice depends on a middle
|
||||
hop's individual health. A dead middle hop makes the exit verdict honestly
|
||||
`dead`; localization is served by the verdict flip log.
|
||||
- **Rejected: auto-reroute rules when a group dies.** A dead group blocks
|
||||
fail-closed — visible and alertable. Silent rerouting would hide the outage
|
||||
and change routing semantics behind the user's back.
|
||||
Consequence: the probe budget is bounded by the config, not the subscription;
|
||||
cold start converges in seconds (immediate first pass after apply); wanting
|
||||
numbers for an unused group has one honest answer — reference it from a rule.
|
||||
|
||||
## D20 — Probe URL / Interval are global-only; per-group overrides deleted
|
||||
Decided 2026-07-24 (proxy-health plan §5.D). Groups carried optional
|
||||
`ProbeURL`/`ProbeInterval` overrides. That bred a documented ambiguity — one
|
||||
node shared by two groups with different URLs yields incomparable delays and
|
||||
needs "whose URL wins" dedup machinery (the `(dial, URL)` plan key) — and it
|
||||
breaks the D18 board: a verdict is a *(record, TTL)* pair with TTL derived from
|
||||
the probe interval, so per-group intervals would make the same record mean
|
||||
different things to different readers. Xray's observatory has exactly one
|
||||
global `probeURL`/`probeInterval`; that is the model we mapped onto (§3).
|
||||
|
||||
**Decision: only `Globals.ProbeURL`/`Globals.ProbeInterval`.** They drive the
|
||||
native group checker, the observatory and the manual exit test alike (failover
|
||||
keeps its 30 s default interval). Probe-plan dedup collapses to the dial path.
|
||||
Migration is the standard dead-option drainage: old UCI configs carrying
|
||||
`probe_url`/`probe_interval` on a group parse silently and the options vanish
|
||||
on the next render.
|
||||
- **Rejected: keep per-group overrides.** Incomparable measurements across
|
||||
groups, undefined semantics for shared members, and a per-group TTL that
|
||||
fractures the single-board verdict.
|
||||
- **Deferred: per-tier probe cadence** (rule-critical tags more often). If ever
|
||||
needed it is a field on the observatory's `ProbeJob` — a scheduling knob,
|
||||
not a return of per-group *configuration*.
|
||||
Consequence: all delay numbers are comparable (least ping ranks apples against
|
||||
apples), and group settings lose two footgun fields while Settings keeps the
|
||||
two that actually govern every check.
|
||||
|
||||
## D21 — A rule's destination is a rule-set, and nothing else
|
||||
Decided 2026-07-25 (product owner). `config rule` carried THREE ways to say
|
||||
where traffic is going: `dst_domain` (an inline domain list), `dst_ip` (an inline
|
||||
CIDR list) and `dst_ruleset` (a reference to a `config ruleset`). Three
|
||||
mechanisms meant three sets of semantics to learn and keep straight, and the
|
||||
inline ones were the worse half of the trade: they are re-parsed per rule instead
|
||||
of being compiled once into a `.srs`, they cannot be shared between rules, and
|
||||
their matcher vocabulary had drifted from the rule-set one in a way nobody could
|
||||
see (below).
|
||||
|
||||
**Decision: `dst_domain` and `dst_ip` are removed (schema v2). `dst_ruleset` is
|
||||
the only destination matcher.** `Src`, `dst_port` and `proto` are untouched —
|
||||
they are not lists of destinations and have no rule-set form.
|
||||
|
||||
- **Rejected: keep the inline lists as a shorthand.** "One obvious way" is the
|
||||
whole point; a shorthand that quietly means something different from the long
|
||||
form (see the bare-entry trap) is worse than no shorthand.
|
||||
- **Rejected: promote inline lists to rule-sets lazily at generate time.** The
|
||||
config on disk would then not say what the router does, and the panel would
|
||||
have to render a list the user cannot find or edit.
|
||||
|
||||
### The bare-entry trap, and how the migration handles it
|
||||
The two contexts already disagreed about exactly one spelling, silently:
|
||||
|
||||
| entry | in a rule (`dst_domain`) | in a rule-set (`entry`) | migrated to |
|
||||
|--------------------|--------------------------|-------------------------|--------------|
|
||||
| `example.com` | **exact host** | **host + subdomains** | `full:example.com` |
|
||||
| `full:example.com` | exact host | exact host | unchanged |
|
||||
| `suffix:example.com` / `.example.com` | host + subdomains | host + subdomains | unchanged |
|
||||
| `keyword:ads` | substring | substring | unchanged |
|
||||
| `regexp:^ads\.` | pattern | pattern *(added here)* | unchanged |
|
||||
| `geosite:x` / `geoip:x` | inert (engine field removed) | inert (unknown prefix) | unchanged |
|
||||
|
||||
`shaterd migrate` (schema v1→v2, `shater/model/migrate.go`) creates one inline
|
||||
`config ruleset` per rule that still carries a legacy list — `rule-<rule name>`
|
||||
for domains, `rule-<rule name>-ip` for addresses — moves the entries across with
|
||||
the conversion above, appends the new name to `dst_ruleset`, and deletes the old
|
||||
option. It is idempotent, it resumes an interrupted run, and it never overwrites
|
||||
a hand-written rule-set that already owns the generated name (it picks
|
||||
`rule-<name>-2`). `regexp:` support was added to inline rule-sets in the same
|
||||
change precisely so the move can be lossless.
|
||||
|
||||
`geosite:`/`geoip:` entries are copied VERBATIM rather than promoted to a
|
||||
`source=geosite` rule-set: those matchers have been inert since the engine
|
||||
dropped the route-rule geosite/geoip fields, and turning a dead matcher live
|
||||
during an upgrade would be a behaviour change, not a migration. The text is kept
|
||||
so the operator can see it and convert it deliberately.
|
||||
|
||||
**One deliberate semantic change, called out:** a rule that used BOTH lists
|
||||
matched them with AND (an engine route rule ANDs its matcher fields), which is
|
||||
almost never what "these sites and these networks" meant. The two generated
|
||||
rule-sets are ORed, because `rule_set: [a, b]` matches when either matches. Such
|
||||
a rule matches more after the migration than before; it affects only configs that
|
||||
used both fields at once.
|
||||
|
||||
Consequence: one destination mechanism, one vocabulary, one place a list is
|
||||
edited; every list is compiled once and reused. The panel's rule editor drops its
|
||||
Domain(s) and IP/CIDR(s) fields; its destination control is a checkbox list of
|
||||
the rulesets that already exist, and nothing more. Creating and filling a list
|
||||
stays in the Rulesets panel — **rejected: a "create a list from here" shortcut in
|
||||
the rule editor**, because a second place to author a list is a second place for
|
||||
its semantics and its duplicate-name rules to drift, and the whole point of this
|
||||
decision was to stop having two.
|
||||
|
||||
|
||||
@@ -13,8 +13,12 @@ usable release, **[T1]** next, **[T2]** later. Phases refer to `ROADMAP.md`.
|
||||
- **[MVP]** TPROXY transparent proxy for multiple LAN interfaces (TCP + UDP), SNI/
|
||||
Host/QUIC sniffing.
|
||||
- **[MVP]** First-match routing rules by source (IP/CIDR/MAC/interface/zone),
|
||||
destination (domain/suffix/keyword/geosite), reusable domain/IP lists, port,
|
||||
proto → target (outbound/selector/chain/direct/block) + egress.
|
||||
destination, port, proto → target (outbound/selector/chain/direct/block) + egress.
|
||||
A rule names its **destination through a rule-set only** — a reusable named list
|
||||
(inline domains/CIDRs, a local or remote file, or a geosite/geoip category) that is
|
||||
compiled once into a `.srs` and shared by every rule that references it. Domain
|
||||
entries take `full:` (exact), `suffix:` / a leading dot (host + subdomains),
|
||||
`keyword:` (substring) and `regexp:`; a bare entry means host + subdomains.
|
||||
- **[MVP]** Node groups with balancer/observatory (least-ping/failover/round-robin).
|
||||
- **[T1]** Multi-hop chains (L1→Ln); per-rule egress selection; egress via any
|
||||
interface/tunnel (e.g. an AmneziaWG tunnel).
|
||||
|
||||
+81
-15
@@ -26,7 +26,9 @@ What it does:
|
||||
Arg / env:
|
||||
|
||||
- `VERSION` — stamped into `constant.Version`. Resolution: positional arg →
|
||||
`$SHATER_VERSION` → `git describe --tags` → `v0.2.0-dev`.
|
||||
`$SHATER_VERSION` → `ci/version.sh --binary` → `v0.2.0-dev`. `ci/version.sh` is
|
||||
the **same** computation the package version comes from (§2.1), so the string
|
||||
the panel shows always matches what `apk info shaterd` / `opkg status` report.
|
||||
- `--fast` — skip `npm ci` when `panel/node_modules` already exists.
|
||||
- `UPX=/path/to/upx` — override the UPX binary (default `upx` on `PATH`). UPX is
|
||||
cross-arch, so one host packs both the amd64 and aarch64 ELFs. (Note: UPX also
|
||||
@@ -84,15 +86,52 @@ it. Because the binary is UPX-packed, the package disables the SDK's default str
|
||||
feed installed and run `make package/shaterd/compile` (and the others) per target.
|
||||
See `openwrt-package-build-ci` for SDK/feed mechanics.
|
||||
|
||||
### 2.1 Package versions come from the git tag
|
||||
|
||||
`PKG_VERSION`/`PKG_RELEASE` are **not** maintained by hand. They used to be, and
|
||||
nobody bumped them: **v0.2.2 … v0.2.6 all shipped as `shaterd 0.2.0-r3`** with
|
||||
different binaries inside (v0.2.6's ELF is 5 491 616 B against r2's 5 488 336 B).
|
||||
Both package managers offer an upgrade only when the feed's version string
|
||||
differs from the installed one, so `apk update` saw nothing new and the routers
|
||||
could not be updated through the normal path at all.
|
||||
|
||||
`ci/version.sh` now derives them from `git describe`, once per CI job:
|
||||
|
||||
| Build | `PKG_VERSION` | `PKG_RELEASE` | `constant.Version` |
|
||||
|---|---|---|---|
|
||||
| tag push `v0.2.7` | `0.2.7` | `1` | `v0.2.7-r1` |
|
||||
| dispatch, 3 commits past `v0.2.7` | `0.2.7` | `4` | `v0.2.7-r4-g<sha>` |
|
||||
| no reachable tag / no git | `0.0.0` | `1` | `v0.0.0-r1` |
|
||||
|
||||
Ordering is what makes this safe, and both managers agree on it (checked with
|
||||
`apk version -t` on apk-tools 3.0.3 and `opkg compare-versions` on opkg
|
||||
38eccbb1): the dotted part decides first, `-rN` only breaks ties — so
|
||||
`0.2.7-r1 > 0.2.6-r12 > 0.2.6-r1 > 0.2.0-r3`. A release therefore always
|
||||
outranks every rolling build before it, rolling builds between two releases grow
|
||||
monotonically, and an untagged build (`0.0.0`) can never masquerade as an
|
||||
upgrade.
|
||||
|
||||
The value travels as `SHATER_PKG_VERSION`/`SHATER_PKG_RELEASE` in the SDK build
|
||||
environment; the Makefiles read it with a literal fallback for manual/offline
|
||||
builds. Both lanes then **assert** the produced `.ipk`/`.apk` really carries it,
|
||||
so a lost variable fails the build instead of shipping a stale version.
|
||||
|
||||
`byedpi` is deliberately excluded — `PKG_VERSION:=0.17.3` is *upstream ByeDPI's*
|
||||
version, which is what `PKG_HASH` pins and what tells you which ByeDPI is
|
||||
installed. Stamping our tag on it would also be a downgrade: every comparator
|
||||
reads `0.2.7 < 0.17.3` (component-wise, `2 < 17`). Bump its `PKG_RELEASE` by hand
|
||||
when our packaging of it changes.
|
||||
|
||||
## 3. Install on a router
|
||||
|
||||
Install order follows the deps (`shaterd` → `shater-core` → `luci-app-shater`):
|
||||
|
||||
```sh
|
||||
opkg install shaterd_0.2.0-1_<arch>.ipk # or: apk add shaterd (25.12+)
|
||||
opkg install shater-core_0.2.0-1_all.ipk
|
||||
opkg install luci-app-shater_0.2.0-1_all.ipk
|
||||
opkg install byedpi_0.17.3-1_<arch>.ipk # optional: ByeDPI egress
|
||||
# <ver> = the release version, e.g. 0.2.7-r1 (§2.1 — it comes from the git tag)
|
||||
opkg install shaterd_<ver>_<arch>.ipk # or: apk add shaterd (25.12+)
|
||||
opkg install shater-core_<ver>_all.ipk
|
||||
opkg install luci-app-shater_<ver>_all.ipk
|
||||
opkg install byedpi_0.17.3-r1_<arch>.ipk # optional: ByeDPI egress
|
||||
```
|
||||
|
||||
Installing from a signed feed instead:
|
||||
@@ -158,16 +197,25 @@ every `opkg update`; no `--nocheck-signature` needed. A **tagged** release
|
||||
|
||||
### Updating
|
||||
|
||||
Name the packages. **Never run a bare `opkg upgrade`** — with no arguments it
|
||||
tries to upgrade *every* installed package from *every* configured feed, which on
|
||||
OpenWrt means base/system packages on the overlay and is a well-known way to
|
||||
brick a router.
|
||||
|
||||
```sh
|
||||
opkg update
|
||||
opkg upgrade shaterd shater-core luci-app-shater byedpi # only our own packages
|
||||
```
|
||||
|
||||
Updates are only offered when the feed's `Version` differs from the installed one,
|
||||
so **bump `PKG_RELEASE`** (or `PKG_VERSION`) in the package Makefile on every
|
||||
shipped change — otherwise `opkg upgrade` sees the same version and does nothing.
|
||||
Do **not** `opkg upgrade` base/system packages from this feed; upgrade only the
|
||||
four shater packages above.
|
||||
Drop `byedpi` from the list if you never installed it. An upgrade is offered only
|
||||
when the feed's `Version` differs from the installed one — that is exactly what
|
||||
bug B4 broke (v0.2.2…v0.2.6 all published as `0.2.0-r3`). Since then CI derives
|
||||
the version from the git tag on every build (§2.1), so there is nothing to bump
|
||||
by hand any more; check with:
|
||||
|
||||
```sh
|
||||
opkg list-installed | grep -E 'shaterd|shater-core|luci-app-shater|byedpi'
|
||||
```
|
||||
|
||||
## 6. apk feed (OpenWrt/ImmortalWrt 25.12+ — incl. BananaWRT 25.12-mtk-vendor)
|
||||
|
||||
@@ -214,15 +262,33 @@ apk add byedpi # optional: ByeDPI desync egress
|
||||
|
||||
### Updating
|
||||
|
||||
**Never run a bare `apk upgrade`.** With no arguments apk reconciles *every*
|
||||
installed package against *every* configured repository at once; on a router
|
||||
whose distfeeds point at a moving snapshot that can pull in — or roll back —
|
||||
unrelated system packages. Always name ours:
|
||||
|
||||
```sh
|
||||
apk update
|
||||
apk upgrade shaterd shater-core luci-app-shater byedpi # only our own packages
|
||||
apk upgrade shaterd shater-core luci-app-shater byedpi
|
||||
```
|
||||
|
||||
Same rule as opkg: an upgrade is only offered when the feed version differs, so
|
||||
bump `PKG_RELEASE`/`PKG_VERSION` on every shipped change (apk shows it as
|
||||
`0.2.0-r1`). Pin a version instead of tracking rolling by pointing the repo line
|
||||
at `.../download/apk-vX.Y.Z-$(cat /etc/apk/arch)/packages.adb`.
|
||||
apk-tools 3 documents exactly this behaviour for `apk upgrade`: *"When no
|
||||
packages are specified, all packages are upgraded if possible. If list of
|
||||
packages is provided, only those packages are upgraded along with needed
|
||||
dependencies."* The equivalent form, which additionally re-pins the packages in
|
||||
`world`, is:
|
||||
|
||||
```sh
|
||||
apk add -u shaterd shater-core luci-app-shater byedpi # -u = --upgrade
|
||||
```
|
||||
|
||||
Drop `byedpi` from either list if you never installed it. Check what you are on
|
||||
with `apk list -I shaterd shater-core luci-app-shater byedpi` — the version reads
|
||||
`0.2.7-r1` (§2.1: `PKG_VERSION-rPKG_RELEASE`, derived from the git tag by CI, so
|
||||
every build really is a new version; before that fix v0.2.2…v0.2.6 all published
|
||||
as `0.2.0-r3` and `apk update` offered nothing). Pin a version instead of tracking
|
||||
rolling by pointing the repo line at
|
||||
`.../download/apk-vX.Y.Z-$(cat /etc/apk/arch)/packages.adb`.
|
||||
|
||||
### BananaWRT `25.12-mtk-vendor` compatibility
|
||||
|
||||
|
||||
@@ -195,7 +195,7 @@ type Chain struct { Name string; Hops []string } // "group:<n>" | "node:<n>", L1
|
||||
type Egress struct { Name,Type,Interface,Target string } // interface|proxy|direct|block
|
||||
type Rule struct {
|
||||
Name string; Enabled bool; Order int
|
||||
Src []string; DstDomain,DstRuleset,DstIP []string; DstPort,Proto string
|
||||
Src []string; DstRuleset []string; DstPort,Proto string // dst = ruleset only (v0.2 schema v2)
|
||||
Target string // chain:|group:|node:|direct|block
|
||||
Egress,Kill string
|
||||
SchedEnabled bool; SchedDays []string; SchedStart,SchedEnd string; SchedUTCOffset int
|
||||
@@ -257,8 +257,12 @@ Apply/rollback: `apSnapshot` (run→last-good, nft→last-good.nft, route marks)
|
||||
- `config node`: name, enabled, uri, mux, mux_concurrency, xudp_concurrency, xudp_udp443, sockopt_mark, tcp_fast_open, tcp_keepalive_idle.
|
||||
- `config group`: name, source, subscription, list node, strategy, include/exclude/filter_proto/filter_country, dedup, probe_url, probe_interval.
|
||||
- `config chain`: name, list hop. `config egress`: name, type, interface, target.
|
||||
- `config ruleset`: name, type(domain|ipcidr), source(inline|file|url), url, path, format, update_interval, list entry.
|
||||
- `config rule`: name, enabled, order, list src/dst_domain/dst_ruleset/dst_ip, dst_port, proto, target, egress, kill, sched_enabled, list sched_day, sched_start/end/tz.
|
||||
- `config ruleset`: name, type(domain|ipcidr), source(inline|file|url|geosite|geoip), url, path, format, update_interval, list category, list entry.
|
||||
- `config rule`: name, enabled, order, list src, list dst_ruleset, dst_port, proto, target, egress, kill, sched_enabled, list sched_day, sched_start/end, sched_utc_offset.
|
||||
v0.1 carried `dst_domain`/`dst_ip` on the rule itself; **schema v2 removed both** — a
|
||||
destination is a `config ruleset` and nothing else. `shaterd migrate` folds each legacy
|
||||
list into a generated `rule-<name>` (and `rule-<name>-ip`) inline ruleset; see
|
||||
`DECISIONS.md` D21 for the entry-by-entry conversion table.
|
||||
- `config preset`: name, enabled, order, target. `config profile`: name, enabled, priority, list match_iface, probe_url, probe_mode, sched_*, list enable_rule/disable_rule, default_target, default_egress.
|
||||
- `config resolver`: name, type, address, detour, pool. `config dns_rule`: order, list match_domain/match_src, resolver.
|
||||
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
# Документация shater
|
||||
|
||||
Документация продукта **shater** (управляемый интернет-шлюз для роутеров на
|
||||
OpenWrt). Лицо репозитория и быстрый старт — в корневом [`../README.md`](../README.md).
|
||||
|
||||
| Документ | О чём |
|
||||
|----------|-------|
|
||||
| [CONTEXT.md](CONTEXT.md) | **Начните здесь** — контекст проекта, история v0.1→v0.2, решения в кратце, testbed/инфра |
|
||||
| [INSTALL.md](INSTALL.md) | Сборка ship-артефакта (`shaterd`) и установка обоих фидов — opkg (24.10) и apk (25.12+) |
|
||||
| [ARCHITECTURE.md](ARCHITECTURE.md) | One-binary дизайн, auth-handoff LuCI→панель, data/DNS/apply-потоки (диаграммы) |
|
||||
| [FEATURES.md](FEATURES.md) | Полный список фич с тегами MVP/T1/T2 |
|
||||
| [ROADMAP.md](ROADMAP.md) | Фазовый план |
|
||||
| [DECISIONS.md](DECISIONS.md) | Почему sing-box, почему форк, split панели, лицензия и т.д. |
|
||||
| [DESIGN.md](DESIGN.md) | Визуальная система панели — направление «Faceplate», токены, компоненты |
|
||||
| [PORTING.md](PORTING.md) | Порт проверенных кусков из v0.1 |
|
||||
|
||||
Документация движка-форка (sing-box-lx) — в его слое: [`../docs-lx/`](../docs-lx/)
|
||||
и [`../SPECS/`](../SPECS/).
|
||||
@@ -0,0 +1,170 @@
|
||||
# Живое тестирование shater v0.2.6 на mini_router
|
||||
|
||||
**Дата:** 2026-07-25
|
||||
**Устройство:** Bananapi BPi-R3 Mini · ImmortalWrt **25.12-linkup** · `aarch64_cortex-a53`
|
||||
**Установка:** из подписанного apk-фида `apk-v0.2.6-aarch64_cortex-a53`
|
||||
**Пакеты:** `shaterd 0.2.0-r3`, `shater-core 0.2.0-r3`, `luci-app-shater 0.2.0-r2`, `byedpi 0.17.3-r1`
|
||||
**Сборка:** CI run 61, коммит `024e9308c` (вершина `main`)
|
||||
|
||||
Сценарий: полное удаление предыдущей установки → чистая установка из фида →
|
||||
проверка дефолтного состояния → восстановление рабочего конфига с подписками
|
||||
(315 узлов) → функциональная проверка.
|
||||
|
||||
**Итог: 79 проверок, 74 PASS, 5 находок** (детали и разбор — в
|
||||
`shater-bugs-2026-07-25.md` на рабочем столе).
|
||||
|
||||
---
|
||||
|
||||
## 1. Релиз и фид
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T1 | Публикация `apk-v0.2.6-<arch>` для обеих архитектур | PASS |
|
||||
| T2 | Ассеты: 4 `.apk` + `packages.adb` + `shater-apk.pem` | PASS |
|
||||
| T3 | `apk update` принимает индекс (проверка EC-подписи) | PASS |
|
||||
| T4 | Пакеты видны в нужных версиях (r3/r3/r2) | PASS |
|
||||
| T5 | Диагностика сборки: `kmod packages selected (=m): 0` (было 1078) | PASS |
|
||||
| T6 | Собраны ровно наши 4 пакета | PASS |
|
||||
| T7 | opkg-лейн v0.2.6 (24.10) тоже зелёный | PASS |
|
||||
|
||||
## 2. Установка
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T8 | `apk add luci-app-shater byedpi` — 4 пакета | PASS |
|
||||
| T9 | Зависимости `kmod-nft-tproxy`/`kmod-nft-socket` из базового фида | PASS |
|
||||
| T10 | Целостность: `apk manifest` = sha256 файла на диске | PASS |
|
||||
| T11 | Установлен именно бинарь v0.2.6 (5 491 616 Б vs 5 488 336 Б в r2) | PASS |
|
||||
| T12 | init-скрипты `shater`, `shater-cron` | PASS |
|
||||
| T13 | `sysctl.d/99-shater.conf`, `hotplug.d/iface/99-shater` | PASS |
|
||||
| T14 | boot-линки `S99shater`, `K10shater`, `S96shater-cron` | PASS |
|
||||
|
||||
## 3. Дефолтное состояние (чистая установка)
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T15 | Дефолтный конфиг создан uci-defaults (27 строк) | PASS |
|
||||
| T16 | `enabled='0'` — плоскость не ставится без согласия | PASS |
|
||||
| T17 | Заготовлен tproxy-inbound на LAN, пресеты выключены | PASS |
|
||||
| T18 | Демон стартует, `plane=none`, `table=false` | PASS |
|
||||
| T19 | Права конфига `-rw-------` (0600) | PASS |
|
||||
|
||||
## 4. Восстановление рабочего конфига
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T20 | Восстановление из бэкапа (3213 UCI-строк) | PASS |
|
||||
| T21 | Кэш подписок цел: 315 узлов в 4 файлах | PASS |
|
||||
| T22 | `shaterd migrate` → `ok`, схема v1 | PASS |
|
||||
| T23 | Старт с реальным конфигом: `active`, `engine_running`, `plane=full` | PASS |
|
||||
|
||||
## 5. Data plane
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T24 | Таблица `inet shater` создана (9 цепочек/сетов) | PASS |
|
||||
| T25 | 16 tproxy-правил | PASS |
|
||||
| T26 | `ip rule from all fwmark 0x2000 lookup shater` | PASS |
|
||||
| T27 | `accept_local=1` на `br-lan` | PASS |
|
||||
| T28 | DNS-divert: `dport 53 → tproxy :12345` для LAN-интерфейсов | PASS |
|
||||
| T29 | DoT заблокирован: `dport 853 reject` | PASS |
|
||||
| T30 | `block_doh=1`, правила присутствуют | PASS |
|
||||
| T31 | **Kill-switch fail-closed**: цепочка `forward` завершается `drop` для LAN (v4+v6) | PASS |
|
||||
| T32 | fw4 и dnsmasq не тронуты (свои таблицы целы) | PASS |
|
||||
|
||||
## 6. Панель и API
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T33 | SPA отдаётся на `:8088` | PASS |
|
||||
| T34 | `shaterd mint-token` выдаёт одноразовый токен | PASS |
|
||||
| T35 | `/api/status` без сессии → **401** | PASS |
|
||||
| T36 | `/api/session` (POST, JSON) → 200 + cookie `HttpOnly; SameSite=Strict; Max-Age=28800` | PASS |
|
||||
| T37 | `/api/status` по cookie отдаёт данные, совпадающие с CLI | PASS |
|
||||
| T38 | `/api/config` — 340 записей узлов | PASS |
|
||||
| T39 | `/api/groups/health` — 103 протестировано, 13 живых, выбран `FR-vless-8` | PASS |
|
||||
| T40 | `/api/devices` — устройства с IPv4/IPv6/MAC | PASS |
|
||||
| T41 | `/api/interfaces` — `ewan/eth1 10.0.0.125/24 zone=wan` | PASS |
|
||||
| T42 | `/api/ruleset/status` — remote-ruleset обновлён сегодня | PASS |
|
||||
| T43 | `/api/stats` — memory backend, счётчики и top-domains | PASS |
|
||||
| T44 | `/api/stats/log` — query-log с доменом, qtype, rcode, сервером | PASS |
|
||||
| T45 | `/api/log?range=100` — пусто (следствие `log_file='0'`, не дефект) | OK |
|
||||
|
||||
## 7. Жизненный цикл конфигурации
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T46 | `shaterd apply` → `{"changed":false}`, `can_rollback=true` | PASS |
|
||||
| T47 | `shaterd confirm` снимает авто-откат (`can_rollback=false`) | PASS |
|
||||
| T48 | `shaterd rollback` после confirm корректно сообщает об отсутствии last-good | PASS |
|
||||
| T49 | `shaterd reconcile` (SIGHUP) не роняет движок | PASS |
|
||||
| T50 | `shaterd sub update all-qomar` — реально обновил 143 узла | PASS |
|
||||
| T51 | `shaterd blocklist update` → reconcile signalled | PASS |
|
||||
| T52 | `shaterd schedule due` → reconcile signalled | PASS |
|
||||
|
||||
## 8. Устойчивость
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T53 | `kill -9` демона → procd поднимает новый PID | PASS |
|
||||
| T54 | После respawn: `engine_running=true`, `plane=full` | PASS |
|
||||
| T55 | `stop` снимает таблицу `inet shater` полностью | PASS |
|
||||
| T56 | `stop` → пауза → `start`: плоскость восстанавливается | PASS |
|
||||
| T57 | Сеть при остановленном shater не деградирует | PASS |
|
||||
| T58 | Память: 253 МБ занято из 2 ГБ при работающем движке | PASS |
|
||||
|
||||
## 9. DNS
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T59 | Резолв через `127.0.0.1` | PASS |
|
||||
| T60 | LAN-клиенты резолвят через движок (query-log растёт) | PASS |
|
||||
| T61 | `.lan`-домены остаются за dnsmasq | PASS |
|
||||
| T62 | dnsmasq жив и слушает на всех адресах | PASS |
|
||||
| T63 | **Резолв через LAN-адрес `10.67.0.1` после `restart`** | **FAIL — B3** |
|
||||
| T64 | Тот же резолв после `stop` → пауза → `start` | PASS |
|
||||
|
||||
## 10. Конфигурация и логи
|
||||
|
||||
| # | Проверка | Результат |
|
||||
|---|---|---|
|
||||
| T65 | 5 правил маршрутизации, 2 профиля, активен `ethernet-uplink` | PASS |
|
||||
| T66 | **Два правила `default`, оба catch-all — нижнее живое, верхнее мертво** | **FAIL — B1** |
|
||||
| T67 | **`shaterd nodes` всегда возвращает `[]`** | **FAIL — B2** |
|
||||
| T68 | Логи уходят в syslog (`log_syslog=1`, 22 записи) | PASS |
|
||||
| T69 | **ANSI-escape коды в syslog** | **FAIL — B5** |
|
||||
| T70 | `loglevel=warning` соблюдается | PASS |
|
||||
| T71–T79 | Прочие проверки состояния (статус-поля, права, uptime, счётчики, целостность таблиц) | PASS |
|
||||
|
||||
---
|
||||
|
||||
## Находки
|
||||
|
||||
| ID | Суть | Важность |
|
||||
|---|---|---|
|
||||
| **B1** | Два catch-all правила `default`; одно из них не работает никогда. **Поправка к первоначальному диагнозу:** правило без условий задаёт `route.Final`, а не выпускается как match-all, поэтому выигрывает ПОСЛЕДНЕЕ (`order=100 → group:auto`) — трафик идёт через прокси, а мёртвая настройка это `order=20 → direct` | средняя |
|
||||
| **B2** | `shaterd nodes` — заглушка, всегда `[]`, хотя usage обещает список узлов (в кэше 315, в `/api/config` 340) | средняя |
|
||||
| **B3** | После `service shater restart` резолв к LAN-адресу роутера не работает и не восстанавливается; `stop`+пауза+`start` — работает (гонка) | средняя |
|
||||
| **B4** | `PKG_RELEASE` не менялся с v0.2.1 → v0.2.2…v0.2.6 выходят как `r3` при разном содержимом; `apk upgrade` не увидит обновления | средняя |
|
||||
| **B5** | ANSI-раскраска попадает в syslog | низкая |
|
||||
|
||||
Разбор с воспроизведением — в `shater-bugs-2026-07-25.md`.
|
||||
|
||||
## История CI по этому релизу
|
||||
|
||||
Путь до зелёной сборки apk-лейна занял четыре итерации, каждая вскрывала
|
||||
следующий слой одной причины:
|
||||
|
||||
| Тег | Что чинили | Итог |
|
||||
|---|---|---|
|
||||
| v0.2.2 | — (первый прогон с фиксами аудита) | `Disk quota exceeded`, 3593 `apk mkpkg kmod-*` |
|
||||
| v0.2.3 | `.config` строится с нуля, а не дописывается | 1078 kmod — SDK вообще не везёт `.config` |
|
||||
| v0.2.4 | Выключены `ALL`/`ALL_KMODS`/`ALL_NONSHARED` | 1078 kmod — они выбираются не через `ALL_KMODS` |
|
||||
| v0.2.5 | Второй проход: явное `is not set` для каждого kmod | 1078 kmod — kconfig игнорирует user-значение у беспромптовых символов |
|
||||
| **v0.2.6** | Удаление сгенерированных блоков `config PACKAGE_*` (`default m`) из `Config-build.in` | **0 kmod, сборка зелёная** |
|
||||
|
||||
Корень: `target/sdk/Makefile` генерирует `Config-build.in` прогоном
|
||||
`convert-config.pl` по конфигу бильдбота, где `ALL_KMODS=y` уже развернулся в
|
||||
`CONFIG_PACKAGE_kmod-*=m` на каждый модуль. Фильтр `next if /^(# )?CONFIG_PACKAGE/`
|
||||
в скрипте стоит в ветке `else`, куда строка со знаком `=` не попадает, поэтому
|
||||
каждый kmod приезжает в SDK как безусловный `default m`.
|
||||
@@ -112,8 +112,8 @@ func getGroupDelay(server *Server) func(w http.ResponseWriter, r *http.Request)
|
||||
} else {
|
||||
server.logger.Debug("outbound ", tag, " available: ", t, "ms")
|
||||
server.urlTestHistory.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: t,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: t,
|
||||
})
|
||||
resultAccess.Lock()
|
||||
result[tag] = t
|
||||
|
||||
@@ -209,8 +209,8 @@ func getProxyDelay(server *Server) func(w http.ResponseWriter, r *http.Request)
|
||||
server.urlTestHistory.DeleteURLTestHistory(realTag)
|
||||
} else {
|
||||
server.urlTestHistory.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: delay,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: delay,
|
||||
})
|
||||
}
|
||||
}()
|
||||
|
||||
@@ -2,6 +2,9 @@ package libbox
|
||||
|
||||
import (
|
||||
"context"
|
||||
// lx:begin sec-consttime
|
||||
"crypto/subtle"
|
||||
// lx:end sec-consttime
|
||||
"errors"
|
||||
"net"
|
||||
"os"
|
||||
@@ -97,9 +100,11 @@ func unaryAuthInterceptor(ctx context.Context, req any, info *grpc.UnaryServerIn
|
||||
if len(values) == 0 {
|
||||
return nil, status.Error(codes.Unauthenticated, "missing authentication secret")
|
||||
}
|
||||
if values[0] != sCommandServerSecret {
|
||||
// lx:begin sec-consttime
|
||||
if subtle.ConstantTimeCompare([]byte(values[0]), []byte(sCommandServerSecret)) != 1 {
|
||||
return nil, status.Error(codes.Unauthenticated, "invalid authentication secret")
|
||||
}
|
||||
// lx:end sec-consttime
|
||||
return handler(ctx, req)
|
||||
}
|
||||
|
||||
@@ -115,9 +120,11 @@ func streamAuthInterceptor(srv any, ss grpc.ServerStream, info *grpc.StreamServe
|
||||
if len(values) == 0 {
|
||||
return status.Error(codes.Unauthenticated, "missing authentication secret")
|
||||
}
|
||||
if values[0] != sCommandServerSecret {
|
||||
// lx:begin sec-consttime
|
||||
if subtle.ConstantTimeCompare([]byte(values[0]), []byte(sCommandServerSecret)) != 1 {
|
||||
return status.Error(codes.Unauthenticated, "invalid authentication secret")
|
||||
}
|
||||
// lx:end sec-consttime
|
||||
return handler(srv, ss)
|
||||
}
|
||||
|
||||
|
||||
@@ -74,7 +74,11 @@ func (r *oomReporter) WriteReport(memoryUsage uint64) error {
|
||||
draftInfo = nil
|
||||
}
|
||||
reportsDir := filepath.Join(sWorkingPath, "oom_reports")
|
||||
err = os.MkdirAll(reportsDir, 0o777)
|
||||
// lx:begin sec-perms
|
||||
// OOM reports embed the config snapshot (server secrets, keys) and logs;
|
||||
// keep the tree owner-only (0700 dirs / 0600 files) instead of 0777/0666.
|
||||
err = os.MkdirAll(reportsDir, 0o700)
|
||||
// lx:end sec-perms
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
@@ -121,7 +125,9 @@ func discardDraftIfCurrent(draftPath string, draftInfo os.FileInfo) error {
|
||||
|
||||
func (r *oomReporter) writeSnapshot(destPath string, memoryUsage uint64) error {
|
||||
now := time.Now().UTC()
|
||||
err := os.MkdirAll(destPath, 0o777)
|
||||
// lx:begin sec-perms
|
||||
err := os.MkdirAll(destPath, 0o700)
|
||||
// lx:end sec-perms
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
|
||||
@@ -44,7 +44,10 @@ func baseReportMetadata() reportMetadata {
|
||||
|
||||
func writeReportFile(destPath string, name string, content []byte) {
|
||||
filePath := filepath.Join(destPath, name)
|
||||
os.WriteFile(filePath, content, 0o666)
|
||||
// lx:begin sec-perms
|
||||
// Report files may carry the config snapshot (secrets) — owner-only.
|
||||
os.WriteFile(filePath, content, 0o600)
|
||||
// lx:end sec-perms
|
||||
chownReport(filePath)
|
||||
}
|
||||
|
||||
@@ -69,7 +72,9 @@ func copyConfigSnapshot(destPath string) {
|
||||
}
|
||||
|
||||
func initReportDir(path string) {
|
||||
os.MkdirAll(path, 0o777)
|
||||
// lx:begin sec-perms
|
||||
os.MkdirAll(path, 0o700)
|
||||
// lx:end sec-perms
|
||||
chownReport(path)
|
||||
}
|
||||
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 204 KiB |
@@ -15,6 +15,17 @@
|
||||
include $(TOPDIR)/rules.mk
|
||||
|
||||
PKG_NAME:=byedpi
|
||||
|
||||
# DELIBERATELY NOT auto-versioned from our git tag (unlike shaterd/shater-core/
|
||||
# luci-app-shater, which take SHATER_PKG_VERSION/SHATER_PKG_RELEASE from
|
||||
# ci/version.sh). PKG_VERSION here is THIRD-PARTY UPSTREAM's version — it is what
|
||||
# PKG_SOURCE_URL/PKG_HASH pin, and what tells an operator which ByeDPI is
|
||||
# actually installed. Stamping our tag on it would be both a lie and a
|
||||
# regression: our tags are 0.2.x, and every version comparator (apk-tools 3 and
|
||||
# opkg alike, verified) reads 0.2.7 < 0.17.3 — component-wise numerically, 2 < 17
|
||||
# — so the "new" package would be a DOWNGRADE and routers would refuse it.
|
||||
# Bump PKG_RELEASE BY HAND when *our packaging* of it changes (init script, uci
|
||||
# defaults, build flags); bump PKG_VERSION+PKG_HASH when upstream releases.
|
||||
PKG_VERSION:=0.17.3
|
||||
PKG_RELEASE:=1
|
||||
|
||||
|
||||
@@ -24,8 +24,13 @@ LUCI_TITLE:=LuCI thin launcher for Shater (mini dashboard + panel handoff)
|
||||
LUCI_DEPENDS:=+shater-core +rpcd
|
||||
LUCI_PKGARCH:=all
|
||||
|
||||
PKG_VERSION:=0.2.0
|
||||
PKG_RELEASE:=1
|
||||
# Version comes from the git tag via ci/version.sh -> SHATER_PKG_VERSION /
|
||||
# SHATER_PKG_RELEASE in the SDK build env (see openwrt/shaterd/Makefile for the
|
||||
# full rationale — bug B4). The literals are the manual/offline fallback only.
|
||||
# These MUST stay above the luci.mk include: luci.mk only defaults PKG_VERSION/
|
||||
# PKG_RELEASE when they are still unset, and the i18n subpackages inherit them.
|
||||
PKG_VERSION:=$(if $(SHATER_PKG_VERSION),$(SHATER_PKG_VERSION),0.2.0)
|
||||
PKG_RELEASE:=$(if $(SHATER_PKG_RELEASE),$(SHATER_PKG_RELEASE),1)
|
||||
|
||||
PKG_MAINTAINER:=Shater <maqrota@icloud.com>
|
||||
PKG_LICENSE:=GPL-3.0-or-later
|
||||
|
||||
@@ -13,8 +13,13 @@
|
||||
include $(TOPDIR)/rules.mk
|
||||
|
||||
PKG_NAME:=shater-core
|
||||
PKG_VERSION:=0.2.0
|
||||
PKG_RELEASE:=2
|
||||
|
||||
# Version comes from the git tag via ci/version.sh -> SHATER_PKG_VERSION /
|
||||
# SHATER_PKG_RELEASE in the SDK build env (see openwrt/shaterd/Makefile for the
|
||||
# full rationale — bug B4: v0.2.2…v0.2.6 all shipped as 0.2.0-r3). The literals
|
||||
# are the manual/offline fallback only.
|
||||
PKG_VERSION:=$(if $(SHATER_PKG_VERSION),$(SHATER_PKG_VERSION),0.2.0)
|
||||
PKG_RELEASE:=$(if $(SHATER_PKG_RELEASE),$(SHATER_PKG_RELEASE),1)
|
||||
|
||||
PKG_MAINTAINER:=Shater <maqrota@icloud.com>
|
||||
PKG_LICENSE:=GPL-2.0-or-later
|
||||
|
||||
@@ -62,7 +62,29 @@ config inbound
|
||||
# list node 'my-node'
|
||||
#
|
||||
# A routing rule. target: chain:<n>|group:<n>|node:<n>|egress:<n>|direct|block.
|
||||
# Match on src / dst_domain / dst_ruleset / dst_ip / dst_port / proto.
|
||||
# Match on src / dst_ruleset / dst_port / proto. A rule with NO matcher at all is
|
||||
# the default route for everything that reached it.
|
||||
#
|
||||
# WHERE the traffic is going is named ONLY by dst_ruleset — one or more
|
||||
# `config ruleset` names; the rule matches when ANY of them matches. There is no
|
||||
# inline domain or address list on a rule (`dst_domain`/`dst_ip` were removed in
|
||||
# schema v2): a destination list is written once as a ruleset, compiled into a
|
||||
# .srs and shared by every rule that references it. `shaterd migrate` converts
|
||||
# older configs automatically, creating a `rule-<name>` ruleset per rule.
|
||||
#config ruleset
|
||||
# option name 'blocked-video'
|
||||
# option type 'domain'
|
||||
# option source 'inline'
|
||||
# list entry 'youtube.com'
|
||||
# list entry 'suffix:googlevideo.com'
|
||||
#
|
||||
#config rule
|
||||
# option name 'video-via-main'
|
||||
# option enabled '1'
|
||||
# option order '50'
|
||||
# list dst_ruleset 'blocked-video'
|
||||
# option target 'group:main'
|
||||
#
|
||||
#config rule
|
||||
# option name 'all-via-main'
|
||||
# option enabled '1'
|
||||
@@ -83,11 +105,16 @@ config inbound
|
||||
# option type 'direct'
|
||||
# option dpi 'fragment'
|
||||
#
|
||||
#config ruleset
|
||||
# option name 'youtube'
|
||||
# option source 'geosite'
|
||||
# list category 'youtube'
|
||||
#
|
||||
#config rule
|
||||
# option name 'youtube-fragment'
|
||||
# option enabled '1'
|
||||
# option order '50'
|
||||
# list dst_domain 'geosite:youtube'
|
||||
# list dst_ruleset 'youtube'
|
||||
# option target 'egress:frag'
|
||||
#
|
||||
# A DNS resolver (type: doh|dot|plain|local|fakeip). `detour` routes its queries
|
||||
|
||||
@@ -47,6 +47,13 @@ PROG=/usr/bin/shaterd
|
||||
# hotplug/shater-cron touch the data plane. tmpfs => cleared by reboot, so
|
||||
# nothing reconciles before this init has run at boot.
|
||||
ACTIVE_FLAG=/var/run/shater.active
|
||||
# Written by `shaterd run`; the single-owner token this init waits on so a
|
||||
# restart never overlaps a new data plane with the previous one's teardown.
|
||||
PIDFILE=/var/run/shaterd.pid
|
||||
# Seconds `start` will wait for a predecessor to finish its teardown. Must be
|
||||
# >= term_timeout below (procd's hard cap on a predecessor's life after SIGTERM)
|
||||
# so we never give up while procd is still letting it shut down cleanly.
|
||||
STOP_WAIT_SECS=40
|
||||
|
||||
# --- helpers ---------------------------------------------------------------
|
||||
|
||||
@@ -66,6 +73,54 @@ _slog() {
|
||||
[ "$(uci -q get shater.globals.log_syslog)" = "0" ] || logger -t shater "$@"
|
||||
}
|
||||
|
||||
# Echo the pid of a LIVE `shaterd run`, or fail. The pidfile is written by the
|
||||
# daemon itself and removed only by the daemon that owns it, AFTER its teardown
|
||||
# has completed — so "pidfile names a live process" is precisely "the previous
|
||||
# data plane has not been dismantled yet".
|
||||
shater_daemon_pid() {
|
||||
local pid
|
||||
pid=$(cat "$PIDFILE" 2>/dev/null) || return 1
|
||||
[ -n "$pid" ] || return 1
|
||||
kill -0 "$pid" 2>/dev/null || return 1
|
||||
echo "$pid"
|
||||
}
|
||||
|
||||
# Block until no predecessor daemon is left, bounded by STOP_WAIT_SECS.
|
||||
#
|
||||
# WHY THIS EXISTS. procd's `stop` is ASYNCHRONOUS: rc.common's `restart` is
|
||||
# literally `stop; start`, and the `service delete` ubus call returns the moment
|
||||
# procd has SENT SIGTERM — not when the instance is gone. `start` therefore
|
||||
# re-adds the instance while the outgoing `shaterd run` is still executing its
|
||||
# honest teardown (engine close, then `nft delete table`, `ip rule`/`ip route`
|
||||
# removal and the per-iface sysctl restore). The result is that `restart` is NOT
|
||||
# equivalent to `stop` + pause + `start`: the new plane is stood up on top of
|
||||
# kernel state the old one has not finished removing, which is what B3 (DNS to
|
||||
# the router's own LAN address dead after a restart, and never recovering) came
|
||||
# out of. Waiting here restores the equivalence, and costs literally nothing when
|
||||
# there is no predecessor — the check runs before the first sleep.
|
||||
#
|
||||
# Returning non-zero does NOT abort the start: the daemon carries its own
|
||||
# single-owner guard and will refuse (or wait) on its side. Better to hand the
|
||||
# decision to the process that can actually see the plane than to leave the box
|
||||
# with no service at all.
|
||||
shater_wait_stopped() {
|
||||
local i=0 pid
|
||||
pid=$(shater_daemon_pid) || return 0
|
||||
_slog -p daemon.info \
|
||||
"restart: waiting for the previous shaterd (pid $pid) to finish tearing the data plane down"
|
||||
while [ "$i" -lt "$STOP_WAIT_SECS" ]; do
|
||||
sleep 1
|
||||
i=$((i + 1))
|
||||
shater_daemon_pid >/dev/null || {
|
||||
_slog -p daemon.info "restart: previous shaterd exited after ${i}s; starting a fresh one"
|
||||
return 0
|
||||
}
|
||||
done
|
||||
_slog -p daemon.warn \
|
||||
"restart: previous shaterd (pid $pid) still alive after ${STOP_WAIT_SECS}s — starting anyway"
|
||||
return 1
|
||||
}
|
||||
|
||||
# --- procd lifecycle -------------------------------------------------------
|
||||
|
||||
start_service() {
|
||||
@@ -87,6 +142,14 @@ start_service() {
|
||||
return 0
|
||||
fi
|
||||
|
||||
# Do not stand a new data plane up on top of one that is still being taken
|
||||
# down. On `restart` procd has only just SIGTERMed the previous instance and
|
||||
# returned; this is the handshake that makes `restart` == `stop` + pause +
|
||||
# `start`. It also keeps `migrate` below from rewriting UCI underneath a
|
||||
# daemon that is still reading it. No-op (and no delay) when nothing is
|
||||
# running, which is the boot case.
|
||||
shater_wait_stopped
|
||||
|
||||
# Bring the UCI schema forward before the daemon reads it (idempotent;
|
||||
# refuses a newer schema) so an upgraded package never applies a stale config.
|
||||
"$PROG" migrate >/dev/null 2>&1
|
||||
@@ -111,7 +174,16 @@ start_service() {
|
||||
procd_set_param stderr 1
|
||||
# Give the daemon room to run its honest teardown (engine.Close + netplane
|
||||
# restore) before procd SIGKILLs it.
|
||||
procd_set_param term_timeout 10
|
||||
#
|
||||
# 30s, not 10s: an engine holding a few hundred outbounds closes its
|
||||
# urltest/observatory goroutines and flushes experimental.cache_file to FLASH
|
||||
# before the netplane teardown even starts, and on eMMC/NAND that alone can
|
||||
# outlast 10s. A SIGKILL there aborts the teardown at an arbitrary point and
|
||||
# leaves the plane HALF removed — the nft table gone but the policy routing
|
||||
# still installed, or vice versa — which is precisely the class of leftover
|
||||
# state the successor's idempotent fast-path cannot see and never repairs.
|
||||
# Shutdown is bounded by procd either way; we are only choosing where.
|
||||
procd_set_param term_timeout 30
|
||||
procd_close_instance
|
||||
|
||||
# Mark the stack live for hotplug/cron — but ONLY when interception is
|
||||
@@ -141,10 +213,12 @@ stop_service() {
|
||||
reload_service() {
|
||||
# Fired by the `shater` config.change reload-trigger (LuCI Save & Apply /
|
||||
# reload_config). Simplest correct behaviour: stop + start. `stop` clears the
|
||||
# flag and SIGTERMs the daemon (honest teardown); `start` re-guards on
|
||||
# enabled and, if still enabled, launches a fresh `shaterd run` that reads
|
||||
# the new UCI and applies it. When the stack is disabled, `start` is a no-op,
|
||||
# so a disable+apply cleanly tears everything down.
|
||||
# flag and SIGTERMs the daemon (honest teardown); `start` WAITS for that
|
||||
# teardown to actually finish (shater_wait_stopped) and then launches a fresh
|
||||
# `shaterd run` that reads the new UCI and applies it. When the stack is
|
||||
# disabled, `start` is a no-op, so a disable+apply cleanly tears everything
|
||||
# down. Because the wait lives in start_service, this path gets the same
|
||||
# stop-then-start ordering guarantee as `restart`.
|
||||
stop
|
||||
start
|
||||
}
|
||||
|
||||
@@ -34,8 +34,17 @@
|
||||
include $(TOPDIR)/rules.mk
|
||||
|
||||
PKG_NAME:=shaterd
|
||||
PKG_VERSION:=0.2.0
|
||||
PKG_RELEASE:=2
|
||||
|
||||
# VERSIONING — derived from the git tag, NOT hand-maintained here (bug B4).
|
||||
# ci/version.sh turns `git describe` into SHATER_PKG_VERSION/SHATER_PKG_RELEASE
|
||||
# (tag vX.Y.Z -> X.Y.Z + r1; off-tag -> last tag + r<commits+1>), and
|
||||
# ci/build-feed.sh / ci/build-feed-apk.sh export them into the SDK build env of
|
||||
# both lanes. Both lanes then ASSERT that the produced .ipk/.apk really carries
|
||||
# that version, so a lost env can never silently ship a stale one again.
|
||||
# The literals below are ONLY the manual/offline fallback (no CI, no git) — they
|
||||
# are not "the release version"; releases are named by the tag.
|
||||
PKG_VERSION:=$(if $(SHATER_PKG_VERSION),$(SHATER_PKG_VERSION),0.2.0)
|
||||
PKG_RELEASE:=$(if $(SHATER_PKG_RELEASE),$(SHATER_PKG_RELEASE),1)
|
||||
|
||||
PKG_MAINTAINER:=Shater <maqrota@icloud.com>
|
||||
PKG_LICENSE:=GPL-3.0-or-later
|
||||
|
||||
@@ -85,61 +85,6 @@
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
|
||||
/* ---- query log wrapper + honest empty state ---- */
|
||||
.log-wrap {
|
||||
margin-top: calc(var(--u, 8px) * 3);
|
||||
}
|
||||
/* DNS-filter readout that sits above the live query log. */
|
||||
.filter-readout {
|
||||
display: grid;
|
||||
grid-template-columns: minmax(0, 1.4fr) minmax(0, 1fr);
|
||||
gap: calc(var(--u, 8px) * 2);
|
||||
align-items: center;
|
||||
margin-bottom: calc(var(--u, 8px) * 1.5);
|
||||
}
|
||||
@media (max-width: 560px) {
|
||||
.filter-readout {
|
||||
grid-template-columns: 1fr;
|
||||
}
|
||||
}
|
||||
.topblocked {
|
||||
list-style: none;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 4px;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
}
|
||||
.topblocked li {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
gap: 8px;
|
||||
padding: 2px 6px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 3px;
|
||||
background: var(--recess, transparent);
|
||||
}
|
||||
.topblocked .tb-dom {
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
color: var(--ink, inherit);
|
||||
}
|
||||
.topblocked .tb-n {
|
||||
color: var(--accent, #e8823c);
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
.qempty {
|
||||
margin: 10px 2px 0;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.6;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--faint);
|
||||
}
|
||||
|
||||
/* ---- apply / confirm / rollback controls ---- */
|
||||
.controls {
|
||||
display: flex;
|
||||
|
||||
+1
-1
@@ -35,7 +35,7 @@ const SOON: Record<Exclude<Route, 'overview'>, string> = {
|
||||
dns: 'Pick resolvers and blocklists — DoH/DoT upstreams, fakeip pool, and per-source DNS rules.',
|
||||
devices: 'See LAN clients and their traffic — per-device policy and activity at a glance.',
|
||||
insights: 'Traffic & DNS insights — top domains, per-rule bytes, exits, and per-device volume.',
|
||||
profiles: 'WAN-mode / failover profiles — auto-switch routing by uplink, probe, or schedule.',
|
||||
profiles: 'WAN-mode / failover profiles — auto-switch routing by the active uplink.',
|
||||
settings: 'Global settings — kill-switch, IPv6, marks/tables, commit-confirm, health probe.',
|
||||
apply: 'Review, apply, and roll back config changes with the commit-confirm safety window.',
|
||||
}
|
||||
|
||||
+132
-139
@@ -308,11 +308,11 @@ export interface GroupMemberHealth {
|
||||
* auto 119 / 122 alive tested 122 of 298
|
||||
* stealth 2 / 2 alive
|
||||
*
|
||||
* Untested must NEVER be folded into dead. A group probes its members LAZILY —
|
||||
* only while it is being used — so a freshly booted router with a 376-node
|
||||
* subscription legitimately has almost no history, and "3 alive / 373 dead"
|
||||
* would be an alarm raised at the moment nothing is wrong. `dead` is only ever a
|
||||
* POSITIVE finding: a probe ran and failed.
|
||||
* Untested must NEVER be folded into dead. The daemon's observatory probes a
|
||||
* group only when an enabled routing rule can reach it (see `used`), and only a
|
||||
* probe that RAN and FAILED writes `dead` — so a freshly (re)started engine
|
||||
* legitimately reads mostly untested for a few seconds, and an unused group
|
||||
* reads untested forever. Neither is an alarm.
|
||||
*
|
||||
* `bound: true` means every member is a per-group egress COPY — this health was
|
||||
* measured through the group's own egress and is deliberately NOT comparable
|
||||
@@ -326,6 +326,12 @@ export interface GroupHealth {
|
||||
group: string
|
||||
type: string // "selector" | "urltest"
|
||||
bound: boolean
|
||||
/** An enabled routing rule (the Final target, a DNS-resolver detour, a device
|
||||
* target, …) reaches this group, so the observatory probes its members in
|
||||
* the background. false ⇒ nothing routes through the group: it is skipped by
|
||||
* the background probing and every member stays `untested`. That is an
|
||||
* "unused" note about the ROUTING CONFIG, never a health problem. */
|
||||
used: boolean
|
||||
/** Node name the group routes through right now; '' when it hasn't picked. */
|
||||
selected: string
|
||||
total: number
|
||||
@@ -338,61 +344,38 @@ export interface GroupHealth {
|
||||
members?: GroupMemberHealth[]
|
||||
}
|
||||
|
||||
/**
|
||||
* The daemon's background health sweep — the thing that keeps these numbers
|
||||
* filling in without anyone pressing a button. Absent on older daemons, in which
|
||||
* case the UI must not promise that untested members will resolve on their own.
|
||||
*/
|
||||
export interface HealthSweep {
|
||||
enabled: boolean
|
||||
cursor: number
|
||||
total: number
|
||||
cycles: number
|
||||
}
|
||||
|
||||
/**
|
||||
* GET /api/groups/health. `groups` is ALWAYS an array, never null.
|
||||
*
|
||||
* The `node_test_*` fields mirror GET /api/nodes/test: the probe-all run is what
|
||||
* turns a group's untested members into a real alive/dead verdict, so one poll of
|
||||
* this endpoint drives the numbers, the button and its progress readout.
|
||||
* The numbers fill in on their own: the daemon's observatory probes every group
|
||||
* and chain an enabled routing rule can reach (a ~10 s tick on the global Probe
|
||||
* URL / Interval), so a used group's untested members resolve to a real
|
||||
* alive/dead verdict within seconds. There is no manual run-everything button
|
||||
* any more and nothing to report here beyond the groups themselves.
|
||||
*/
|
||||
|
||||
/** Per-chain reachability, the chain analogue of {@link GroupHealth}.used (plan
|
||||
* §5.E): a chain no enabled routing rule routes through is outside the
|
||||
* observatory's plan, so its exit is never probed and the Targets card renders it
|
||||
* "unused" instead of an exit-test readout. A chain has no membership counters —
|
||||
* it is a fixed path, and its end-to-end health is the exit test's job. */
|
||||
export interface ChainHealth {
|
||||
name: string
|
||||
/** An enabled routing rule (the Final target, a DNS-resolver detour, a device
|
||||
* target, …) reaches this chain, so the observatory probes its exit in the
|
||||
* background. false ⇒ nothing routes through the chain: it is skipped by the
|
||||
* background probing and its end-to-end health stays untested. That is an
|
||||
* "unused" note about the ROUTING CONFIG, never a health problem. */
|
||||
used: boolean
|
||||
}
|
||||
|
||||
export interface GroupsHealth {
|
||||
groups: GroupHealth[]
|
||||
node_test_running: boolean
|
||||
node_test_done: number
|
||||
node_test_total: number
|
||||
sweep?: HealthSweep
|
||||
}
|
||||
|
||||
/**
|
||||
* The health run's scope, as the daemon publishes it (panel/api.go NodeTestScope).
|
||||
*
|
||||
* It is a CONSTANT, not a list, and that is the whole point: a health run measures
|
||||
* every node, every endpoint and every group's egress copies in one pass, so it can
|
||||
* never be attributed to one card. Render it as a single global progress indicator.
|
||||
* The scoped counterpart is {@link GroupTestStatus.scope}.
|
||||
*/
|
||||
export const NODE_TEST_SCOPE = 'all_nodes'
|
||||
export type NodeTestScope = typeof NODE_TEST_SCOPE
|
||||
|
||||
/**
|
||||
* GET /api/nodes/test — progress of a manual "Test all nodes" probe-all run.
|
||||
* `running` is true while a run is in flight; `done`/`total` count finished vs
|
||||
* targeted node probes. Idle (never run, or finished) reads `{running:false}`.
|
||||
*/
|
||||
export interface NodeTestStatus {
|
||||
running: boolean
|
||||
done: number
|
||||
total: number
|
||||
/** Always {@link NODE_TEST_SCOPE}; absent on daemons older than the split. */
|
||||
scope?: NodeTestScope
|
||||
}
|
||||
|
||||
/** POST /api/nodes/test reply: `started` when a fresh run began, else `running`. */
|
||||
export interface NodeTestStart {
|
||||
started?: boolean
|
||||
running?: boolean
|
||||
/** Per-chain reachability, same "unused" badge as groups (plan §5.E). Present on
|
||||
* the summary and `?members=` shapes; the single-group (`?group=`) shape is a
|
||||
* group detail request and omits it. Always an array when present; absent ⇒ not
|
||||
* reported by this daemon version (the page treats absence like "not known yet"). */
|
||||
chains?: ChainHealth[]
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -463,23 +446,6 @@ export interface Globals {
|
||||
EndpointResolver: string
|
||||
ProbeURL: string
|
||||
ProbeInterval: string
|
||||
/**
|
||||
* Tick period of the daemon's background health sweep — the walk that keeps
|
||||
* every node's health fresh so the group cards fill in without anyone pressing
|
||||
* a button (model.Globals.SweepInterval, UCI `sweep_interval`).
|
||||
*
|
||||
* `""` means ENABLED at the engine's default tick, NOT off. That default is
|
||||
* load-bearing: urltest groups probe only while they are being used and selector
|
||||
* groups never probe at all, so without the sweep a 376-node subscription reads
|
||||
* almost entirely "not measured".
|
||||
*
|
||||
* Accepted: a duration ("30s", "5m", "1h") or a bare integer of seconds; any of
|
||||
* `0` / `off` / `none` / `disabled` to switch the sweep off. Below 5 s the
|
||||
* daemon raises it to that floor. An UNRECOGNISED value does not disable the
|
||||
* sweep — the daemon warns and falls back to the default — so the panel refuses
|
||||
* it at the input instead of saving something that will be silently ignored.
|
||||
*/
|
||||
SweepInterval: string
|
||||
SchemaVersion: number
|
||||
ActiveProfile: string
|
||||
/**
|
||||
@@ -508,14 +474,17 @@ export interface Globals {
|
||||
DNSIntercept?: boolean // force ALL LAN plaintext DNS (:53) through the engine, incl. router-addressed queries
|
||||
BlockDoH?: boolean // block known public DoH resolvers (by host + IP:443 + Firefox canary) so clients fall back to plaintext :53
|
||||
/**
|
||||
* Our background health sweep of group members — the walk that fills the alive /
|
||||
* The daemon's observatory — the background probing that fills the alive /
|
||||
* dead / tested numbers the Targets page shows for each group (model.Globals
|
||||
* .GroupHealth, UCI `group_health`). Default ON.
|
||||
* .GroupHealth, UCI `group_health`). It probes only the groups and chains an
|
||||
* enabled routing rule can reach, on the global Probe URL / Interval, and
|
||||
* skips everything else. Default ON.
|
||||
*
|
||||
* This governs OUR sweep and the health/testing UI ONLY. It does NOT touch
|
||||
* sing-box's own internal urltest / least_test probes: a group always keeps
|
||||
* picking a live member under the hood regardless of this switch. Turning it off
|
||||
* just stops the extra sweep and hides the health statistics.
|
||||
* This gates the observatory and the health/testing UI ONLY. It does NOT
|
||||
* touch sing-box's own internal urltest / least_test probes: a group always
|
||||
* keeps picking a live member under the hood regardless of this switch.
|
||||
* Turning it off stops the background probing and hides the health
|
||||
* statistics.
|
||||
*
|
||||
* ABSENT ⇒ enabled (an older config never wrote the key), so the invariant the
|
||||
* panel reads by is `globals?.GroupHealth !== false` — never `=== true`.
|
||||
@@ -662,8 +631,6 @@ export interface Group {
|
||||
FilterProto?: string[] | null
|
||||
FilterCountry?: string[] | null
|
||||
Dedup?: boolean
|
||||
ProbeURL?: string
|
||||
ProbeInterval?: string
|
||||
/**
|
||||
* The egress EVERY node in this group dials its own server through — the same
|
||||
* binding `Node.Egress` gives one node, applied to the whole group. "" ⇒ the
|
||||
@@ -731,14 +698,6 @@ export interface RulesetStatus {
|
||||
rule_count: number
|
||||
}
|
||||
|
||||
/** A quick rule on/off bundle (config preset). */
|
||||
export interface Preset {
|
||||
Name: string
|
||||
Enabled: boolean
|
||||
Order?: number
|
||||
Target?: string
|
||||
}
|
||||
|
||||
/** A WAN-mode / failover conditional override (config profile). */
|
||||
export interface Profile {
|
||||
Name: string
|
||||
@@ -752,23 +711,8 @@ export interface Profile {
|
||||
// condition — so adding one SWITCHED OFF an otherwise working profile. Both were
|
||||
// deleted from the Go model; PUT decodes with DisallowUnknownFields, so sending
|
||||
// either key fails the whole write with 400.
|
||||
SchedDays?: string[] | null
|
||||
SchedStart?: string
|
||||
SchedEnd?: string
|
||||
/**
|
||||
* Minutes EAST of UTC that anchor the schedule's wall-clock times (Moscow =
|
||||
* 180, New York winter = −300). The router carries no IANA tzdata, so the
|
||||
* daemon evaluates the window at UTC+offset; the panel captures the editing
|
||||
* browser's offset whenever a schedule field is saved. 0/absent ⇒ UTC.
|
||||
* Known limitation: a fixed offset does not follow DST until re-saved.
|
||||
* (Replaces the deleted SchedTZ — an IANA name never resolved on the router,
|
||||
* so the promised local time was silently UTC; see model.go.)
|
||||
*/
|
||||
SchedUTCOffset?: number
|
||||
EnableRules?: string[] | null
|
||||
DisableRules?: string[] | null
|
||||
DefaultTarget?: string
|
||||
DefaultEgress?: string
|
||||
/** Per-profile override of the endpoint resolver (keyed by active WAN: SIM→yandex, WiFi→DoH). "" ⇒ no override (model.Profile.EndpointResolver, UCI `endpoint_resolver`). */
|
||||
EndpointResolver?: string
|
||||
}
|
||||
@@ -865,9 +809,17 @@ export interface Rule {
|
||||
Enabled: boolean
|
||||
Order: number
|
||||
Src?: string[] | null
|
||||
DstDomain?: string[] | null
|
||||
/**
|
||||
* WHERE the traffic is going — the rule's only destination matcher. Each entry
|
||||
* names a {@link Ruleset}; the rule matches when ANY of them matches.
|
||||
*
|
||||
* There is no inline domain or address list on a rule. `dst_domain`/`dst_ip`
|
||||
* were removed in schema v2, and `shaterd migrate` folds every existing one
|
||||
* into a generated `rule-<name>` ruleset, so a destination list is written and
|
||||
* edited in exactly one place and compiled once into a .srs that every rule
|
||||
* referencing it shares.
|
||||
*/
|
||||
DstRuleset?: string[] | null
|
||||
DstIP?: string[] | null
|
||||
DstPort?: string
|
||||
/**
|
||||
* Narrow the rule to one transport or one sniffed application protocol. A
|
||||
@@ -879,12 +831,19 @@ export interface Rule {
|
||||
Proto?: string // '' | tcp | udp | tls | http | quic | dns | stun | bittorrent | dtls | ssh | rdp | ntp
|
||||
Target?: string
|
||||
Egress?: string
|
||||
Kill?: string
|
||||
/**
|
||||
* Policy for when the target can't resolve at generate time (dead group,
|
||||
* broken chain, missing egress/node). The rule is always still emitted — its
|
||||
* traffic never falls through to the default route. ''/'default'/'closed' and
|
||||
* anything unrecognised block the traffic (fail-closed); 'open' is an explicit,
|
||||
* warned kill-switch bypass that sends it direct.
|
||||
*/
|
||||
Kill?: string // '' | default | closed | open
|
||||
SchedEnabled?: boolean
|
||||
SchedDays?: string[] | null
|
||||
SchedStart?: string
|
||||
SchedEnd?: string
|
||||
/** Minutes east of UTC anchoring SchedStart/SchedEnd/SchedDays — same contract as Profile.SchedUTCOffset. */
|
||||
/** Minutes east of UTC anchoring SchedStart/SchedEnd/SchedDays — the panel captures the editing browser's offset on save (the router has no tzdata). */
|
||||
SchedUTCOffset?: number
|
||||
}
|
||||
|
||||
@@ -958,7 +917,8 @@ export interface Interface {
|
||||
|
||||
/** GET /api/devices row — a discovered LAN client merged with its config, if any. */
|
||||
export interface DiscoveredDevice {
|
||||
ip: string
|
||||
ip: string // primary address (most recent lease)
|
||||
ips: string[] // every known address (v4+v6, multiple leases), primary first
|
||||
mac: string
|
||||
hostname: string
|
||||
online: boolean
|
||||
@@ -988,7 +948,6 @@ export interface Model {
|
||||
Devices?: Device[] | null
|
||||
Chains?: Chain[] | null
|
||||
Rulesets?: Ruleset[] | null
|
||||
Presets?: Preset[] | null
|
||||
Profiles?: Profile[] | null
|
||||
Inbounds?: Inbound[] | null
|
||||
DNSRules?: DNSRule[] | null
|
||||
@@ -1236,6 +1195,50 @@ export function getStatsConns(q: number | StatsLogQuery = {}): Promise<ConnLogEn
|
||||
return MOCK ? mock.getStatsConns(o) : req<ConnLogEntry[]>(`api/stats/conns${statsLogQS(o)}`)
|
||||
}
|
||||
|
||||
/**
|
||||
* One routing rule's reachability verdict — the rule analogue of
|
||||
* {@link ChainHealth}.used: a quiet note about the ROUTING CONFIG, never a health
|
||||
* signal.
|
||||
*
|
||||
* `unreachable` means the rule can NEVER take effect, whatever the traffic. Today
|
||||
* the daemon reports exactly one certain case, and it is a subtle one: a rule with
|
||||
* no conditions at all is not matched in sequence — it becomes the router's
|
||||
* default. Two such rules therefore retire each other, and the LAST one by Order
|
||||
* wins, so an earlier "default → direct" is dead even though it sorts first. A
|
||||
* condition-less rule never retires a rule that HAS conditions: those are matched
|
||||
* ahead of the default whatever their Order.
|
||||
*
|
||||
* `index` is the rule's position in GET /api/config's `Rules`, which is how a
|
||||
* verdict is matched to a row — rule names are not unique, and the config that
|
||||
* prompted this had two rules both called `default`. `name`/`order` are echoed so
|
||||
* a page holding a verdict fetched before an edit can check it still describes the
|
||||
* row it is about to badge, and drop it silently otherwise.
|
||||
*/
|
||||
export interface RuleReach {
|
||||
index: number
|
||||
name: string
|
||||
order: number
|
||||
unreachable: boolean
|
||||
/** The rule that supersedes this one; absent when `unreachable` is false. */
|
||||
shadowed_by?: string
|
||||
/** Its index in `Rules`, or -1 when there is none. */
|
||||
shadowed_by_index: number
|
||||
shadowed_by_order?: number
|
||||
/** Operator-facing sentence; absent when `unreachable` is false. */
|
||||
reason?: string
|
||||
}
|
||||
|
||||
/** GET /api/rules/reachability. `rules` is ALWAYS an array, one entry per rule in
|
||||
* the same order as GET /api/config's `Rules`. */
|
||||
export interface RulesReachability {
|
||||
rules: RuleReach[]
|
||||
}
|
||||
|
||||
/** GET /api/rules/reachability — which routing rules can never fire, and why. */
|
||||
export function getRulesReachability(): Promise<RulesReachability> {
|
||||
return MOCK ? mock.getRulesReachability() : req<RulesReachability>('api/rules/reachability')
|
||||
}
|
||||
|
||||
/** GET /api/ruleset/status — remote rule-set / blocklist freshness + rule counts. */
|
||||
export function getRulesetStatus(): Promise<RulesetStatus[]> {
|
||||
return MOCK ? mock.getRulesetStatus() : req<RulesetStatus[]>('api/ruleset/status')
|
||||
@@ -1357,21 +1360,6 @@ export function updateSubscription(name: string): Promise<{ added: number }> {
|
||||
: req('api/subscription/update', { method: 'POST', body: JSON.stringify({ name }) })
|
||||
}
|
||||
|
||||
/**
|
||||
* POST /api/nodes/test — force a health probe of every node. The daemon runs it in
|
||||
* the background (singleton: a second POST while one is running is a no-op that
|
||||
* reports `{running:true}`); the normal /api/stats poll then surfaces each fresh
|
||||
* result. Resolves to `{started}` or `{running}`.
|
||||
*/
|
||||
export function postNodesTest(): Promise<NodeTestStart> {
|
||||
return MOCK ? mock.postNodesTest() : req<NodeTestStart>('api/nodes/test', { method: 'POST' })
|
||||
}
|
||||
|
||||
/** GET /api/nodes/test — progress of the current/last probe-all run. */
|
||||
export function getNodesTest(): Promise<NodeTestStatus> {
|
||||
return MOCK ? mock.getNodesTest() : req<NodeTestStatus>('api/nodes/test')
|
||||
}
|
||||
|
||||
/**
|
||||
* GET /api/groups/health — per-group membership health (see {@link GroupHealth}).
|
||||
*
|
||||
@@ -1397,20 +1385,24 @@ export function getGroupsHealth(
|
||||
}
|
||||
|
||||
/**
|
||||
* One group's last test: which member the balancer picked, how fast it answered,
|
||||
* and what the internet saw as the source address.
|
||||
* One group's (or chain's) last test: which member the balancer picked, how fast
|
||||
* it answered, and what the internet saw as the source address.
|
||||
*
|
||||
* `ok:true` with an EMPTY `exit_ip`/`exit_country` is a valid, successful result,
|
||||
* not a partial failure: the delay was measured but the exit address could not be
|
||||
* determined (the lookup service was unreachable, or the answer wasn't parseable).
|
||||
* Render it as a success with an unknown address — never as an error.
|
||||
*
|
||||
* Chains ride the same endpoint. For a chain row, `group` carries the CHAIN's
|
||||
* name and `selected` the node its last group hop picked ('' when the exit hop
|
||||
* isn't a group). Everything else reads the same way.
|
||||
*
|
||||
* `ok:false` ⇒ the test failed and `error` carries the human reason; every other
|
||||
* field is meaningless. `tested_unix` is the router's clock, in seconds.
|
||||
*/
|
||||
export interface GroupTestResult {
|
||||
group: string
|
||||
selected: string // the member node the group chose for this test
|
||||
group: string // group name — or a chain name for a chain row
|
||||
selected: string // the member node the group (or the chain's exit group) chose
|
||||
delay_ms: number
|
||||
exit_ip: string // may be '' even when ok
|
||||
exit_country: string // ISO code; may be '' even when ok
|
||||
@@ -1421,25 +1413,26 @@ export interface GroupTestResult {
|
||||
|
||||
/**
|
||||
* GET /api/groups/test — progress plus every result so far. `results` is ALWAYS
|
||||
* an array (never null); `done`/`total` count finished vs targeted groups while
|
||||
* `running` is true. Idle reads `{running:false}` with the last run's results
|
||||
* still attached, so a reload after a test still shows what it found.
|
||||
* an array (never null); `done`/`total` count finished vs targeted groups and
|
||||
* chains while `running` is true. Idle reads `{running:false}` with the last
|
||||
* run's results still attached, so a reload after a test still shows what it found.
|
||||
*/
|
||||
export interface GroupTestStatus {
|
||||
running: boolean
|
||||
done: number
|
||||
total: number
|
||||
/**
|
||||
* The group names THIS run covers. Always an array (never JSON null); absent
|
||||
* only on daemons older than the split.
|
||||
* The group and chain names THIS run covers. Always an array (never JSON
|
||||
* null); absent only on daemons older than the split.
|
||||
*
|
||||
* It is what makes `running` usable. On its own that flag says only "a group
|
||||
* test is happening somewhere", which is why pressing Test on one group used to
|
||||
* put "measuring…" on every card. The rule: show the in-progress indicator on
|
||||
* card g iff `running && scope.includes(g)`. A run started with a name carries
|
||||
* exactly that name; a run started with no name carries every group, and then
|
||||
* the indicator on every card is correct. The scope PERSISTS after the run
|
||||
* ends, so displayed results stay attributable to the cards they came from.
|
||||
* exactly that name; a run started with no name carries every group and
|
||||
* every chain, and then the indicator on every card is correct. The scope
|
||||
* PERSISTS after the run ends, so displayed results stay attributable to the
|
||||
* cards they came from.
|
||||
*/
|
||||
scope?: string[]
|
||||
results: GroupTestResult[]
|
||||
@@ -1455,10 +1448,10 @@ export interface GroupTestStart {
|
||||
}
|
||||
|
||||
/**
|
||||
* POST /api/groups/test — measure a group's delay and exit address. Pass a group
|
||||
* name to test one; pass nothing (or '') to test every group. Singleton: a second
|
||||
* call while a run is in flight resolves to `{started:false, reason:'already
|
||||
* running'}` rather than failing.
|
||||
* POST /api/groups/test — measure a target's delay and exit address. Pass a
|
||||
* group or chain name to test one; pass nothing (or '') to test every group
|
||||
* and every chain. Singleton: a second call while a run is in flight resolves
|
||||
* to `{started:false, reason:'already running'}` rather than failing.
|
||||
*/
|
||||
export function postGroupsTest(name = ''): Promise<GroupTestStart> {
|
||||
return MOCK
|
||||
|
||||
+128
-140
@@ -6,7 +6,7 @@
|
||||
// state mutates in-memory so the Apply / Confirm / Rollback flow is exercisable.
|
||||
//
|
||||
// Type-only imports from api.ts (erased at build) keep this free of a runtime cycle.
|
||||
import type { ApplyResult, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, NodeTestStart, NodeTestStatus, QueryLogEntry, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning } from './api'
|
||||
import type { ApplyResult, ChainHealth, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, QueryLogEntry, RuleReach, RulesReachability, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning } from './api'
|
||||
|
||||
let armed = false // a pending commit-confirm auto-rollback
|
||||
let hasLastGood = false // a predecessor config exists to roll back to (post-apply)
|
||||
@@ -26,9 +26,6 @@ const CONFIG: Model = {
|
||||
EndpointResolver: '',
|
||||
ProbeURL: 'https://www.gstatic.com/generate_204',
|
||||
ProbeInterval: '60s',
|
||||
// '' = the engine's default tick (every 10 s), NOT off. Left blank on purpose
|
||||
// so `?mock` shows the recommended state and the placeholder that says so.
|
||||
SweepInterval: '',
|
||||
SchemaVersion: 2,
|
||||
// Points at the iface-driven profile below, so `?mock` lands in the WAN-watcher
|
||||
// AUTO-PIN state (not a manual override): the plate must say the router pins this
|
||||
@@ -39,7 +36,7 @@ const CONFIG: Model = {
|
||||
Untunnelable: new URLSearchParams(typeof location === 'undefined' ? '' : location.search).get('untun') ?? 'block',
|
||||
DNSIntercept: true, // force ALL LAN plaintext DNS (:53) through the engine
|
||||
BlockDoH: false, // block known public DoH resolvers so clients fall back to plaintext :53
|
||||
GroupHealth: true, // background group-member health sweep + Targets health stats (default on)
|
||||
GroupHealth: true, // observatory: background probing of used groups/chains + Targets health stats (default on)
|
||||
StatsRingSize: 500, // fixed cap — shows the "limit" rendering (500 rows)
|
||||
StatsTimelineMinutes: 0, // 0 ⇒ "Unlimited" rendering + memory warning
|
||||
StatsMaxDomains: 5000, // fixed cap — shows the "limit" rendering (5000)
|
||||
@@ -132,6 +129,9 @@ const CONFIG: Model = {
|
||||
{ Name: 'via-tunnel', Source: 'subscription', Subscription: 'primary', Strategy: 'leastping', Egress: 'awg' },
|
||||
{ Name: 'fallback', Source: 'subscription', Subscription: 'backup', Strategy: 'roundrobin', Egress: '' },
|
||||
],
|
||||
// One multi-hop chain so `?mock` exercises the chain card's Test button and
|
||||
// its result readout: enters through the awg tunnel, exits via the auto group.
|
||||
Chains: [{ Name: 'relay', Hops: ['egress:awg', 'group:auto'] }],
|
||||
Egresses: [
|
||||
{ Name: 'wan', Type: 'interface', Interface: 'wan' },
|
||||
// An AmneziaWG tunnel — the whole point of a group-level egress binding.
|
||||
@@ -146,6 +146,12 @@ const CONFIG: Model = {
|
||||
{ Name: 'block-ads', Enabled: true, Order: 10, DstRuleset: ['ad-hosts'], Target: 'block' },
|
||||
{ Name: 'ru-bypass', Enabled: true, Order: 20, DstRuleset: ['ru-inside'], Target: 'direct' },
|
||||
{ Name: 'private-direct', Enabled: true, Order: 30, DstRuleset: ['private-nets'], Target: 'direct' },
|
||||
// A SECOND condition-less rule, above the real default. It reads like a working
|
||||
// rule and does nothing: a rule with no conditions becomes the router's default,
|
||||
// and the last such rule by Order wins — so this one never applies. It is in the
|
||||
// fixture on purpose, to exercise the "never applies" badge; the field config
|
||||
// that prompted it had two rules BOTH named `default` (orders 20 and 100).
|
||||
{ Name: 'default-bypass', Enabled: true, Order: 40, Target: 'direct' },
|
||||
{ Name: 'default-tunnel', Enabled: true, Order: 900, Target: 'group:auto' },
|
||||
],
|
||||
// Named match-lists a rule points DstRuleset at. url + geosite + geoip are remote
|
||||
@@ -240,20 +246,14 @@ const CONFIG: Model = {
|
||||
Enabled: true,
|
||||
Priority: 20,
|
||||
MatchIface: ['wan1'],
|
||||
SchedDays: [],
|
||||
EnableRules: ['default-tunnel'],
|
||||
DisableRules: ['block-ads', 'ru-bypass', 'private-direct'],
|
||||
DefaultTarget: 'group:auto',
|
||||
},
|
||||
{
|
||||
Name: 'evening-direct',
|
||||
Enabled: true,
|
||||
Priority: 10,
|
||||
MatchIface: [],
|
||||
SchedDays: ['mon', 'tue', 'wed', 'thu', 'fri'],
|
||||
SchedStart: '19:00',
|
||||
SchedEnd: '23:00',
|
||||
DefaultTarget: 'direct',
|
||||
},
|
||||
],
|
||||
}
|
||||
@@ -302,6 +302,48 @@ const RULESET_STATUS: RulesetStatus[] = [
|
||||
{ tag: 'rs-ru-geoip-ru', name: 'ru-geoip', category: 'ru', kind: 'ruleset', remote: true, last_updated: '', interval_seconds: 86_400, rule_count: 0 },
|
||||
]
|
||||
|
||||
/** GET /api/rules/reachability. Mirrors the daemon's analysis over CONFIG.Rules:
|
||||
* a rule with no conditions is the router's default, and the LAST such rule by
|
||||
* Order wins — every earlier one can never apply. It reads the live CONFIG so
|
||||
* edits made in `?mock` keep the badge honest. */
|
||||
export async function getRulesReachability(): Promise<RulesReachability> {
|
||||
await wait(60)
|
||||
const rules = CONFIG.Rules ?? []
|
||||
const out: RuleReach[] = rules.map((r, index) => ({
|
||||
index,
|
||||
name: String(r.Name ?? ''),
|
||||
order: Number(r.Order ?? 0),
|
||||
unreachable: false,
|
||||
shadowed_by_index: -1,
|
||||
}))
|
||||
const conditionless = (r: (typeof rules)[number]): boolean =>
|
||||
!(r.Src ?? []).length &&
|
||||
!(r.DstRuleset ?? []).length &&
|
||||
!String(r.DstPort ?? '').trim() &&
|
||||
!String(r.Proto ?? '').trim()
|
||||
const target = (r: (typeof rules)[number]): string =>
|
||||
String(r.Target ?? '').trim() || (r.Egress ? `egress:${String(r.Egress).trim()}` : '')
|
||||
const defaults = rules
|
||||
.map((r, index) => ({ r, index }))
|
||||
.filter(({ r }) => r.Enabled && conditionless(r) && target(r))
|
||||
.sort((a, b) => Number(a.r.Order ?? 0) - Number(b.r.Order ?? 0) || a.index - b.index)
|
||||
const winner = defaults[defaults.length - 1]
|
||||
if (winner) {
|
||||
for (const { index } of defaults.slice(0, -1)) {
|
||||
out[index].unreachable = true
|
||||
out[index].shadowed_by = String(winner.r.Name ?? '')
|
||||
out[index].shadowed_by_index = winner.index
|
||||
out[index].shadowed_by_order = Number(winner.r.Order ?? 0)
|
||||
out[index].reason =
|
||||
`this rule has no conditions, so it sets the default for all traffic — but rule ` +
|
||||
`"${winner.r.Name}" (order ${winner.r.Order}) has none either and comes after it, so ` +
|
||||
`"${target(winner.r)}" is the default the router uses and this rule's target ` +
|
||||
`"${target(rules[index])}" is never applied`
|
||||
}
|
||||
}
|
||||
return { rules: out }
|
||||
}
|
||||
|
||||
export async function getRulesetStatus(): Promise<RulesetStatus[]> {
|
||||
await wait(90)
|
||||
return RULESET_STATUS.map((r) => ({ ...r }))
|
||||
@@ -861,11 +903,13 @@ export async function getStatsConnsPage(q: StatsLogQuery = {}): Promise<StatsLog
|
||||
// derives them the same way — otherwise managing a device in `?mock` would leave the
|
||||
// discovery row stubbornly claiming the opposite. Includes lan + guest networks.
|
||||
const MOCK_HOSTS: ReadonlyArray<Omit<DiscoveredDevice, 'configured' | 'name' | 'blockCount'>> = [
|
||||
{ ip: '192.168.1.20', mac: 'a4:83:e7:11:22:33', hostname: 'macbook-max', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.31', mac: 'f0:18:98:aa:bb:cc', hostname: 'iphone-lena', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.42', mac: '3c:22:fb:44:55:66', hostname: 'ipad-kids', online: true, state: 'idle', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.55', mac: 'dc:a6:32:77:88:99', hostname: 'tv-livingroom', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '10.20.0.14', mac: 'b8:27:eb:aa:00:11', hostname: '', online: false, state: 'offline', network: 'guest', iface: 'br-guest' },
|
||||
// Dual-stacked (v4 lease + link-local v6) — exercises the merged-row display:
|
||||
// one card, primary IP prominent, the extra address as a secondary chip.
|
||||
{ ip: '192.168.1.20', ips: ['192.168.1.20', 'fe80::a683:e7ff:fe11:2233'], mac: 'a4:83:e7:11:22:33', hostname: 'macbook-max', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.31', ips: ['192.168.1.31'], mac: 'f0:18:98:aa:bb:cc', hostname: 'iphone-lena', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.42', ips: ['192.168.1.42'], mac: '3c:22:fb:44:55:66', hostname: 'ipad-kids', online: true, state: 'idle', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.55', ips: ['192.168.1.55'], mac: 'dc:a6:32:77:88:99', hostname: 'tv-livingroom', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '10.20.0.14', ips: ['10.20.0.14'], mac: 'b8:27:eb:aa:00:11', hostname: '', online: false, state: 'offline', network: 'guest', iface: 'br-guest' },
|
||||
]
|
||||
|
||||
export async function getDevices(): Promise<DiscoveredDevice[]> {
|
||||
@@ -906,9 +950,8 @@ export async function getInterfaces(): Promise<Interface[]> {
|
||||
//
|
||||
// The fixture is a real member LIST per group, not four hand-written counters:
|
||||
// the counters are derived from it, so the daemon's invariants (tested ==
|
||||
// alive+dead, alive+dead+untested == total) hold by construction and cannot drift
|
||||
// as the mock probe-all run flips members. Between them the four groups cover
|
||||
// every state the panel has to render:
|
||||
// alive+dead, alive+dead+untested == total) hold by construction and cannot
|
||||
// drift. Between them the four groups cover every state the panel has to render:
|
||||
//
|
||||
// auto 298 members, 122 measured — 119 alive / 3 dead, 176 untested.
|
||||
// The ordinary case: healthy, with a real dead finding AND a large
|
||||
@@ -918,11 +961,9 @@ export async function getInterfaces(): Promise<Interface[]> {
|
||||
// via-tunnel the same subscription `auto` draws from, dialled through `awg` —
|
||||
// and 0 of 6 alive through it. Two groups, one node set, two
|
||||
// verdicts: the entire reason health is reported per group.
|
||||
// fallback 24 members, NOTHING measured. Neither healthy nor broken — the
|
||||
// state that invites a measurement instead of raising an alarm.
|
||||
//
|
||||
// `?mock&nodetest=1` starts the page with a probe-all already in flight, so the
|
||||
// running/progress rendering is reachable without racing a click.
|
||||
// fallback 24 members, NOTHING measured, and `used:false` — no enabled rule
|
||||
// routes through it, so the observatory never probes it. The card
|
||||
// renders the quiet "unused" note instead of health counters.
|
||||
|
||||
/** City pool for synthesised member names — the real feed looks like this. */
|
||||
const MEMBER_CITIES = [
|
||||
@@ -941,40 +982,42 @@ interface HealthShape {
|
||||
dead: number
|
||||
/** true ⇒ members are per-group egress copies (health measured through it). */
|
||||
bound: boolean
|
||||
/** An enabled rule reaches this group (GroupHealth.used). false ⇒ the
|
||||
* observatory skips it and its members stay untested. */
|
||||
used: boolean
|
||||
type: string
|
||||
/** Node name the group routes through right now; '' when it hasn't picked. */
|
||||
selectedIndex: number | null
|
||||
}
|
||||
|
||||
const HEALTH_SHAPE: Record<string, HealthShape> = {
|
||||
auto: { total: 298, alive: 119, dead: 3, bound: false, type: 'urltest', selectedIndex: 1 },
|
||||
stealth: { total: 2, alive: 2, dead: 0, bound: true, type: 'urltest', selectedIndex: 0 },
|
||||
'via-tunnel': { total: 6, alive: 0, dead: 6, bound: true, type: 'urltest', selectedIndex: null },
|
||||
fallback: { total: 24, alive: 0, dead: 0, bound: false, type: 'urltest', selectedIndex: null },
|
||||
auto: { total: 298, alive: 119, dead: 3, bound: false, used: true, type: 'urltest', selectedIndex: 1 },
|
||||
stealth: { total: 2, alive: 2, dead: 0, bound: true, used: true, type: 'urltest', selectedIndex: 0 },
|
||||
'via-tunnel': { total: 6, alive: 0, dead: 6, bound: true, used: true, type: 'urltest', selectedIndex: null },
|
||||
fallback: { total: 24, alive: 0, dead: 0, bound: false, used: false, type: 'urltest', selectedIndex: null },
|
||||
}
|
||||
|
||||
/**
|
||||
* `?mock&bias=1|2|3` reproduces the optimistic-ratio defect the owner hit on the
|
||||
* live router, on `auto` — 298 members, the same size he reported — and walks it
|
||||
* through the three states the rendering has to tell apart.
|
||||
* `?mock&bias=1|2` reproduces the optimistic-ratio defect the owner hit on the
|
||||
* live router, on `auto` — 298 members, the same size he reported.
|
||||
*
|
||||
* The mechanism: a group writes an entry when a member answers and DELETES it when
|
||||
* one doesn't, so until the background sweep has been over the group its history
|
||||
* holds nothing but successes. `11 / 11 alive` was that, not health.
|
||||
* The mechanism: a group's own probes write an entry when a member answers and
|
||||
* DELETE it when one doesn't, so until the daemon's board has a failure on
|
||||
* record the history holds nothing but successes. `11 / 11 alive` was that,
|
||||
* not health.
|
||||
*
|
||||
* bias=1 → 11 alive, 0 dead, 287 unchecked, sweep mid-first-pass → NO RATIO
|
||||
* bias=2 → 111 alive, 74 dead, 113 unchecked, sweep mid-first-pass → ratio (a
|
||||
* failure is on record, so something is writing both outcomes)
|
||||
* bias=3 → the same counts with a completed pass → ratio
|
||||
* bias=1 → 11 alive, 0 dead, 287 unchecked → NO RATIO (one-sided sample)
|
||||
* bias=2 → 111 alive, 74 dead, 113 unchecked → ratio (a failure is on record,
|
||||
* so something is writing both outcomes)
|
||||
*
|
||||
* Read 1 → 2 → 3 in order: the reading must get MORE PRECISE, never "good, then
|
||||
* Read 1 → 2 in order: the reading must get MORE PRECISE, never "good, then
|
||||
* suddenly bad".
|
||||
*/
|
||||
const BIAS_STAGE =
|
||||
typeof location === 'undefined' ? null : new URLSearchParams(location.search).get('bias')
|
||||
if (BIAS_STAGE === '1') {
|
||||
HEALTH_SHAPE.auto = { ...HEALTH_SHAPE.auto, alive: 11, dead: 0 }
|
||||
} else if (BIAS_STAGE === '2' || BIAS_STAGE === '3') {
|
||||
} else if (BIAS_STAGE === '2') {
|
||||
HEALTH_SHAPE.auto = { ...HEALTH_SHAPE.auto, alive: 111, dead: 74 }
|
||||
}
|
||||
|
||||
@@ -1000,13 +1043,13 @@ function buildMembers(group: string, shape: HealthShape): GroupMemberHealth[] {
|
||||
return out
|
||||
}
|
||||
|
||||
/** Live member state, keyed by group name. Mutated by the mock probe-all run. */
|
||||
/** Live member state, keyed by group name. */
|
||||
const GROUP_MEMBERS = new Map<string, GroupMemberHealth[]>(
|
||||
Object.entries(HEALTH_SHAPE).map(([g, s]) => [g, buildMembers(g, s)]),
|
||||
)
|
||||
|
||||
/** Derive one group's summary from its member list — never hand-written, so the
|
||||
* daemon's counter invariants hold no matter what the probe-all run did. */
|
||||
* daemon's counter invariants hold by construction. */
|
||||
function summarise(group: string, members: GroupMemberHealth[]): GroupHealth {
|
||||
const shape = HEALTH_SHAPE[group]
|
||||
let alive = 0
|
||||
@@ -1024,6 +1067,7 @@ function summarise(group: string, members: GroupMemberHealth[]): GroupHealth {
|
||||
group,
|
||||
type: shape?.type ?? 'urltest',
|
||||
bound: shape?.bound ?? false,
|
||||
used: shape?.used ?? true,
|
||||
selected: sel != null && members[sel] ? members[sel].node : '',
|
||||
total: members.length,
|
||||
tested: alive + dead,
|
||||
@@ -1050,88 +1094,31 @@ function healthList(): GroupHealth[] {
|
||||
return (CONFIG.Groups ?? []).map((g) => summarise(g.Name, GROUP_MEMBERS.get(g.Name) ?? []))
|
||||
}
|
||||
|
||||
// Mock "Test all nodes" probe-all: a run that advances a batch of members per GET
|
||||
// poll so the progress readout and — the point of the run — untested members
|
||||
// turning into a real alive/dead verdict are both exercisable offline.
|
||||
let nodeTest: NodeTestStatus = { running: false, done: 0, total: 0 }
|
||||
/** Members still to be measured by the in-flight run, as [group, index]. */
|
||||
let nodeTestQueue: Array<[string, number]> = []
|
||||
|
||||
function startNodeTest(): void {
|
||||
nodeTestQueue = []
|
||||
for (const [group, members] of GROUP_MEMBERS) {
|
||||
members.forEach((_, i) => nodeTestQueue.push([group, i]))
|
||||
}
|
||||
nodeTest = { running: true, done: 0, total: nodeTestQueue.length }
|
||||
/** Per-chain reachability for the Targets page's "unused" badge (plan §5.E) — the
|
||||
* chain analogue of healthList's `used` field. The mock's single chain `relay` is
|
||||
* NOT referenced by any rule in CONFIG.Rules (they target group:auto / block /
|
||||
* direct), so it reads used=false and its card renders "unused" — exactly the case
|
||||
* the badge exists to surface. A stopped engine reports no chains. */
|
||||
function chainHealthList(): ChainHealth[] {
|
||||
if (!mockPlane().engine) return []
|
||||
return (CONFIG.Chains ?? []).map((c) => ({ name: c.Name, used: chainUsed(c.Name) }))
|
||||
}
|
||||
|
||||
/** Advance the run by one poll's worth of probes, flipping untested members to a
|
||||
* verdict. Roughly 1 in 12 comes back dead, so the numbers move believably. */
|
||||
function advanceNodeTest(): void {
|
||||
if (!nodeTest.running) return
|
||||
const batch = Math.max(1, Math.round(nodeTest.total / 14))
|
||||
for (let n = 0; n < batch; n++) {
|
||||
const next = nodeTestQueue.shift()
|
||||
if (!next) break
|
||||
const [group, i] = next
|
||||
const m = GROUP_MEMBERS.get(group)?.[i]
|
||||
if (m) {
|
||||
// A bound group stays dead through its tunnel — measuring it again does not
|
||||
// make it work, and pretending otherwise would hide the case the feature
|
||||
// exists to show.
|
||||
const dead = HEALTH_SHAPE[group]?.bound && HEALTH_SHAPE[group]?.alive === 0 ? true : i % 12 === 7
|
||||
m.state = dead ? 'dead' : 'alive'
|
||||
m.delay_ms = dead ? 0 : 38 + ((i * 37) % 460)
|
||||
m.age_seconds = 1 + (i % 5)
|
||||
}
|
||||
nodeTest = { ...nodeTest, done: nodeTest.done + 1 }
|
||||
}
|
||||
if (nodeTestQueue.length === 0) nodeTest = { ...nodeTest, running: false }
|
||||
}
|
||||
|
||||
// ?mock&nodetest=1 lands straight in the running state (see the note above). This
|
||||
// is the HEALTH run — global by nature, so the panel must render it as ONE
|
||||
// indicator in the section header and never as a per-card badge.
|
||||
if (typeof location !== 'undefined' && new URLSearchParams(location.search).has('nodetest')) {
|
||||
startNodeTest()
|
||||
}
|
||||
|
||||
/**
|
||||
* The background sweep's progress, as GET /api/groups/health reports it.
|
||||
*
|
||||
* `cycles` is the field with teeth: 0 means the sweep has not been everywhere yet,
|
||||
* so "not measured" is simply "not reached". Once it is ≥ 1 the sweep HAS been
|
||||
* everywhere, and anything still unmeasured lost a reading it used to have.
|
||||
*
|
||||
* ?mock&sweep=first → mid first pass (cycles 0)
|
||||
* ?mock&sweep=off → sweep disabled; nothing fills in on its own
|
||||
*/
|
||||
function mockSweep(): { enabled: boolean; cursor: number; total: number; cycles: number } {
|
||||
const mode = typeof location === 'undefined' ? null : new URLSearchParams(location.search).get('sweep')
|
||||
if (mode === 'off') return { enabled: false, cursor: 0, total: 0, cycles: 0 }
|
||||
if (mode === 'first') return { enabled: true, cursor: 184, total: 707, cycles: 0 }
|
||||
// The bias walkthrough drives the sweep too: stages 1 and 2 are mid-first-pass
|
||||
// (the cursor advances between them, exactly as the live router's did), stage 3
|
||||
// has completed one. See BIAS_STAGE.
|
||||
if (BIAS_STAGE === '1') return { enabled: true, cursor: 312, total: 896, cycles: 0 }
|
||||
if (BIAS_STAGE === '2') return { enabled: true, cursor: 696, total: 896, cycles: 0 }
|
||||
if (BIAS_STAGE === '3') return { enabled: true, cursor: 148, total: 896, cycles: 1 }
|
||||
return { enabled: true, cursor: 184, total: 707, cycles: 3 }
|
||||
/** A chain is "used" when some enabled routing rule (or Final, or a DNS detour)
|
||||
* targets `chain:<name>` — the same reachability the daemon's observatory derives.
|
||||
* The mock's rules never target a chain, so every chain reads used=false; a real
|
||||
* config would mark the ones rules point at used=true. */
|
||||
function chainUsed(name: string): boolean {
|
||||
const target = `chain:${name}`
|
||||
return (CONFIG.Rules ?? []).some(
|
||||
(r) => r.Enabled !== false && (r.Target === target),
|
||||
)
|
||||
}
|
||||
|
||||
export async function getGroupsHealth(
|
||||
opts: { group?: string; members?: boolean } = {},
|
||||
): Promise<GroupsHealth> {
|
||||
await wait(70)
|
||||
advanceNodeTest()
|
||||
const base: Omit<GroupsHealth, 'groups'> = {
|
||||
node_test_running: nodeTest.running,
|
||||
node_test_done: nodeTest.done,
|
||||
node_test_total: nodeTest.total,
|
||||
// The daemon's background sweep, on by default — it is what lets the panel
|
||||
// promise that untested members resolve without anyone pressing anything.
|
||||
sweep: mockSweep(),
|
||||
}
|
||||
if (opts.group) {
|
||||
const members = GROUP_MEMBERS.get(opts.group)
|
||||
// A stopped engine has no groups at all, so every name is a 404 — the same
|
||||
@@ -1141,21 +1128,20 @@ export async function getGroupsHealth(
|
||||
throw new ApiErrorLike(404, 'no such group in the running engine')
|
||||
}
|
||||
return {
|
||||
...base,
|
||||
groups: [{ ...summarise(opts.group, members), members: members.map((m) => ({ ...m })) }],
|
||||
}
|
||||
}
|
||||
const groups = healthList()
|
||||
if (opts.members) {
|
||||
return {
|
||||
...base,
|
||||
groups: groups.map((g) => ({
|
||||
...g,
|
||||
members: (GROUP_MEMBERS.get(g.group) ?? []).map((m) => ({ ...m })),
|
||||
})),
|
||||
chains: chainHealthList(),
|
||||
}
|
||||
}
|
||||
return { ...base, groups }
|
||||
return { groups, chains: chainHealthList() }
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1173,27 +1159,13 @@ class ApiErrorLike extends Error {
|
||||
}
|
||||
}
|
||||
|
||||
export async function postNodesTest(): Promise<NodeTestStart> {
|
||||
await wait(60)
|
||||
if (nodeTest.running) return { running: true }
|
||||
startNodeTest()
|
||||
return { started: true }
|
||||
}
|
||||
|
||||
export async function getNodesTest(): Promise<NodeTestStatus> {
|
||||
await wait(60)
|
||||
advanceNodeTest()
|
||||
// The constant the daemon publishes (api.ts NODE_TEST_SCOPE, spelled inline
|
||||
// because this module may only import TYPES from api.ts): this run is GLOBAL and
|
||||
// can never be attributed to one group's card.
|
||||
return { ...nodeTest, scope: 'all_nodes' }
|
||||
}
|
||||
|
||||
// Mock group test. Deliberately covers every state the UI has to render, one per
|
||||
// group, so a single offline run exercises all of them:
|
||||
// Mock group/chain test. Deliberately covers every state the UI has to render,
|
||||
// one per target, so a single offline run exercises all of them:
|
||||
// auto → ok WITH an exit address
|
||||
// stealth → ok WITHOUT one (delay measured, address undeterminable) — a
|
||||
// SUCCESS, and the case the UI most easily gets wrong
|
||||
// relay → the chain: same wire shape, `group` carries the CHAIN's name and
|
||||
// `selected` the node its exit group picked
|
||||
// fallback → a failure carrying a human reason
|
||||
// Results land one per GET poll, so the running/progress state is visible too.
|
||||
const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_unix'>> = {
|
||||
@@ -1213,6 +1185,15 @@ const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_u
|
||||
ok: true,
|
||||
error: '',
|
||||
},
|
||||
// The chain — Selected is the node the chain's exit group (auto) picked.
|
||||
relay: {
|
||||
selected: 'nl-reality-2',
|
||||
delay_ms: 61,
|
||||
exit_ip: '185.12.34.56',
|
||||
exit_country: 'NL',
|
||||
ok: true,
|
||||
error: '',
|
||||
},
|
||||
// Dead through its tunnel, exactly as its membership health says — the exit
|
||||
// test and the member health tell the same story about the same group.
|
||||
'via-tunnel': {
|
||||
@@ -1251,9 +1232,13 @@ let groupTestPinned = false
|
||||
export async function postGroupsTest(name = ''): Promise<GroupTestStart> {
|
||||
await wait(60)
|
||||
if (groupTest.running) return { started: false, reason: 'already running' }
|
||||
const all = (CONFIG.Groups ?? []).map((g) => g.Name)
|
||||
// Empty name = every group AND every chain, exactly like the daemon.
|
||||
const all = [
|
||||
...(CONFIG.Groups ?? []).map((g) => g.Name),
|
||||
...(CONFIG.Chains ?? []).map((c) => c.Name),
|
||||
]
|
||||
const targets = name ? all.filter((g) => g === name) : all
|
||||
if (targets.length === 0) return { started: false, reason: `no group named “${name}”` }
|
||||
if (targets.length === 0) return { started: false, reason: `no group or chain named “${name}”` }
|
||||
groupTestQueue = [...targets]
|
||||
groupTestPinned = false
|
||||
groupTest = {
|
||||
@@ -1302,7 +1287,10 @@ export async function getGroupsTest(): Promise<GroupTestStatus> {
|
||||
if (typeof location !== 'undefined') {
|
||||
const want = new URLSearchParams(location.search).get('grouptest')
|
||||
if (want) {
|
||||
const all = (CONFIG.Groups ?? []).map((g) => g.Name)
|
||||
const all = [
|
||||
...(CONFIG.Groups ?? []).map((g) => g.Name),
|
||||
...(CONFIG.Chains ?? []).map((c) => c.Name),
|
||||
]
|
||||
const targets = want === 'all' || want === '1' ? all : all.filter((g) => g === want)
|
||||
if (targets.length > 0) {
|
||||
groupTestQueue = [...targets]
|
||||
|
||||
@@ -200,6 +200,16 @@
|
||||
.dev-ip {
|
||||
color: var(--dim);
|
||||
}
|
||||
/* secondary addresses of a merged (multi-IP) device — quiet chips beside the primary */
|
||||
.dev-ip-extra {
|
||||
padding: 1px 6px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 5px;
|
||||
background: color-mix(in srgb, var(--sink) 60%, transparent);
|
||||
color: var(--faint);
|
||||
font-size: 10.5px;
|
||||
cursor: help;
|
||||
}
|
||||
.dev-mac {
|
||||
color: var(--faint);
|
||||
cursor: help;
|
||||
|
||||
@@ -61,7 +61,8 @@ const STATE_LED: Record<string, LedVariant> = {
|
||||
/** A discovered client merged with its config policy (if any). */
|
||||
interface DeviceRow {
|
||||
key: string
|
||||
ip: string
|
||||
ip: string // primary address (most recent lease)
|
||||
ips: string[] // every known address, primary first; [] only for a config-only row without IP
|
||||
mac: string
|
||||
hostname: string
|
||||
state: string // online | idle | offline
|
||||
@@ -186,6 +187,7 @@ export default function Devices() {
|
||||
out.push({
|
||||
key: d.mac || d.ip || d.hostname,
|
||||
ip: d.ip,
|
||||
ips: d.ips?.length ? d.ips : d.ip ? [d.ip] : [],
|
||||
mac: d.mac,
|
||||
hostname: d.hostname,
|
||||
state: d.state || (d.online ? 'online' : 'offline'),
|
||||
@@ -202,6 +204,7 @@ export default function Devices() {
|
||||
out.push({
|
||||
key: d.MAC || d.IP || d.Name,
|
||||
ip: d.IP ?? '',
|
||||
ips: d.IP ? [d.IP] : [],
|
||||
mac: d.MAC ?? '',
|
||||
hostname: d.Name,
|
||||
state: 'offline',
|
||||
@@ -515,6 +518,11 @@ function DeviceCard({
|
||||
</div>
|
||||
<div className="dev-id-l2 mono">
|
||||
<span className="dev-ip">{row.ip || '—'}</span>
|
||||
{row.ips.slice(1).map((ip) => (
|
||||
<span key={ip} className="dev-ip-extra" title="Additional address of this device">
|
||||
{ip}
|
||||
</span>
|
||||
))}
|
||||
<span className="dev-mac" title="Hardware (MAC) address">
|
||||
{row.mac || 'no mac'}
|
||||
</span>
|
||||
|
||||
@@ -797,23 +797,6 @@
|
||||
.ins-conn--dns {
|
||||
grid-template-columns: 62px minmax(60px, 116px) 12px minmax(0, 1fr) minmax(72px, 160px) auto;
|
||||
}
|
||||
/* router-originated lookups (urltest probes / sub fetches): a dimmed pill so they
|
||||
* read distinctly from a real LAN client's IP/hostname. */
|
||||
.ins-conn-dev--router {
|
||||
justify-self: start;
|
||||
max-width: 100%;
|
||||
padding: 1px 8px;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.03em;
|
||||
color: var(--faint);
|
||||
background: color-mix(in srgb, var(--raised) 70%, transparent);
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 999px;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
}
|
||||
.ins-conn--dns .tag {
|
||||
justify-self: end;
|
||||
}
|
||||
|
||||
@@ -177,26 +177,19 @@ function ConnRow({ c, fresh }: { c: ConnLogEntry; fresh: boolean }) {
|
||||
* (block/proxy/pass). Mirrors ConnRow's `device → dest` reading (same `.ins-conn`
|
||||
* chrome + accent arrow) so the DNS log and the Connections log read identically:
|
||||
* the "who asked" is the LAN device, the resolver that answered is the trailing
|
||||
* secondary column. Router-originated lookups (urltest probes / sub fetches) carry
|
||||
* the literal device `router` and render as a dimmed chip so they're distinct from
|
||||
* real LAN clients; an empty device (older data) falls back to a dash.
|
||||
* secondary column. The backend drops the router's own resolutions (urltest probes /
|
||||
* sub fetches) at ingestion, so every row here is a real client; an empty device
|
||||
* (older data) falls back to a dash.
|
||||
*/
|
||||
function DnsRow({ r, fresh }: { r: QueryLogEntry; fresh: boolean }) {
|
||||
const tag = actionTag(r.action)
|
||||
const dev = r.device.trim()
|
||||
const isRouter = dev.toLowerCase() === 'router'
|
||||
return (
|
||||
<li className={fresh ? 'ins-conn ins-conn--dns new' : 'ins-conn ins-conn--dns'}>
|
||||
<span className="ins-conn-t mono">{fmtClock(r.unix)}</span>
|
||||
{isRouter ? (
|
||||
<span className="ins-conn-dev ins-conn-dev--router" title="router-originated lookup">
|
||||
router
|
||||
</span>
|
||||
) : (
|
||||
<span className="ins-conn-dev mono" title={dev || 'source device unknown'}>
|
||||
{dev || '—'}
|
||||
</span>
|
||||
)}
|
||||
<span className="ins-conn-dev mono" title={dev || 'source device unknown'}>
|
||||
{dev || '—'}
|
||||
</span>
|
||||
<span className="ins-conn-arrow" aria-hidden="true">
|
||||
→
|
||||
</span>
|
||||
@@ -1236,7 +1229,7 @@ export default function Insights() {
|
||||
{/* 8 · DNS log — the live DNS DECISIONS (time · device → domain · resolver ·
|
||||
block/proxy/pass). Same shell/scroll/actions as Connections above, and
|
||||
now the same device → target reading: the "who asked" is the LAN client
|
||||
(or `router` for the appliance's own lookups), the resolver trails. */}
|
||||
(router-originated lookups are dropped at ingestion), the resolver trails. */}
|
||||
<Section
|
||||
title="DNS log · decisions"
|
||||
led={dns.err ? 'crit' : dns.rows.length ? 'on' : 'off'}
|
||||
|
||||
+14
-44
@@ -389,55 +389,29 @@ select.fp-input {
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
/* protocol checkboxes */
|
||||
.opt-checks {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px 16px;
|
||||
}
|
||||
.opt-check {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-size: 12px;
|
||||
color: var(--dim);
|
||||
cursor: pointer;
|
||||
}
|
||||
.opt-check input {
|
||||
accent-color: var(--accent);
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
/* chip list */
|
||||
.chip-field {
|
||||
/* extra-headers key/value rows */
|
||||
.hdr-field {
|
||||
gap: 8px;
|
||||
}
|
||||
.chip-row {
|
||||
.hdr-rows {
|
||||
list-style: none;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
flex-direction: column;
|
||||
gap: 6px;
|
||||
}
|
||||
.chip {
|
||||
display: inline-flex;
|
||||
.hdr-row {
|
||||
display: grid;
|
||||
grid-template-columns: minmax(9rem, 1fr) 2fr auto;
|
||||
gap: 8px;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
padding: 3px 4px 3px 9px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 6px;
|
||||
background: color-mix(in srgb, var(--raised) 70%, transparent);
|
||||
}
|
||||
.chip-text {
|
||||
font-size: 11px;
|
||||
color: var(--ink);
|
||||
max-width: 32ch;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
.hdr-key {
|
||||
font-family: var(--font-mono);
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
.chip-x {
|
||||
.hdr-x {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
@@ -453,18 +427,14 @@ select.fp-input {
|
||||
cursor: pointer;
|
||||
transition: color 0.15s, background 0.15s;
|
||||
}
|
||||
.chip-x:hover:not(:disabled) {
|
||||
.hdr-x:hover:not(:disabled) {
|
||||
color: var(--crit);
|
||||
background: color-mix(in srgb, var(--crit) 14%, transparent);
|
||||
}
|
||||
.chip-x:disabled {
|
||||
.hdr-x:disabled {
|
||||
opacity: 0.5;
|
||||
cursor: default;
|
||||
}
|
||||
.chip-add {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
}
|
||||
|
||||
/* editor footer */
|
||||
.opt-actions {
|
||||
|
||||
+133
-218
@@ -576,15 +576,40 @@ export default function Nodes() {
|
||||
)
|
||||
|
||||
// Commit an options edit for one subscription. The editor hands back a fully
|
||||
// patched Subscription (Name/Enabled/URL preserved); we splice it in and reuse
|
||||
// the same save→apply machinery as every other mutation.
|
||||
// patched Subscription (Enabled and every off-form field preserved); we splice
|
||||
// it in and reuse the same save→apply machinery as every other mutation.
|
||||
//
|
||||
// A rename cascades: FromSub is the ONLY link between a sub and its cached
|
||||
// nodes, so every node pointing at the old name is rewritten in the same save
|
||||
// (or the whole group would fall out of the sub's bucket and turn orphan). A
|
||||
// collision with another sub's name blocks the save outright — names are the
|
||||
// Subscriptions' primary key.
|
||||
const editSub = useCallback(
|
||||
(idx: number, patch: Subscription): Promise<boolean> => {
|
||||
if (!config) return Promise.resolve(false)
|
||||
const next = subs.map((s, i) => (i === idx ? patch : s))
|
||||
return save({ ...config, Subscriptions: next }, `Updated ${patch.Name}`)
|
||||
async (idx: number, patch: Subscription): Promise<boolean> => {
|
||||
if (!config) return false
|
||||
const oldName = subs[idx].Name
|
||||
const renamed = patch.Name !== oldName
|
||||
if (renamed && subs.some((s, i) => i !== idx && s.Name === patch.Name)) {
|
||||
flash(`A subscription named “${patch.Name}” already exists.`)
|
||||
return false
|
||||
}
|
||||
const next: Model = { ...config, Subscriptions: subs.map((s, i) => (i === idx ? patch : s)) }
|
||||
if (renamed) {
|
||||
next.Nodes = nodes.map((n) => (n.FromSub === oldName ? { ...n, FromSub: patch.Name } : n))
|
||||
}
|
||||
const ok = await save(next, `Updated ${patch.Name}`)
|
||||
// Carry the group's open/closed override to the new key only once the
|
||||
// rename actually stuck (openMap is keyed by FromSub).
|
||||
if (ok && renamed) {
|
||||
setOpenMap((prev) => {
|
||||
if (!(oldName in prev)) return prev
|
||||
const { [oldName]: was, ...rest } = prev
|
||||
return { ...rest, [patch.Name]: was }
|
||||
})
|
||||
}
|
||||
return ok
|
||||
},
|
||||
[config, subs, save],
|
||||
[config, subs, nodes, save, flash],
|
||||
)
|
||||
|
||||
// Fetch one subscription now (feedback #8): pull it (through the tunnel when
|
||||
@@ -1296,79 +1321,71 @@ const FETCH_VIA: { value: string; label: string }[] = [
|
||||
{ value: 'direct', label: 'Direct' },
|
||||
{ value: 'proxy', label: 'Proxy (through the tunnel)' },
|
||||
]
|
||||
const FORMATS = ['auto', 'clash', 'xray', 'singbox', 'links'] as const
|
||||
const PROTO_FILTERS = ['vless', 'vmess', 'trojan', 'ss'] as const
|
||||
|
||||
/** Flat, all-strings-and-bools shape the form binds to (arrays stay arrays). */
|
||||
/** One editable Key/Value row of the extra-headers list. */
|
||||
interface HeaderRow {
|
||||
key: string
|
||||
value: string
|
||||
}
|
||||
|
||||
/** Split a stored raw "Key: value" header into an editable row on the FIRST
|
||||
* colon. A malformed entry (no colon at all) loads as a key-only row rather
|
||||
* than being silently dropped — the user can fix or delete it. */
|
||||
function toHeaderRow(raw: string): HeaderRow {
|
||||
const i = raw.indexOf(':')
|
||||
if (i < 0) return { key: raw.trim(), value: '' }
|
||||
return { key: raw.slice(0, i).trim(), value: raw.slice(i + 1).trim() }
|
||||
}
|
||||
|
||||
/** Flat, all-strings shape the form binds to (headers stay rows). */
|
||||
interface SubDraft {
|
||||
Name: string
|
||||
URL: string
|
||||
UpdateInterval: string
|
||||
FetchVia: string
|
||||
FetchDetour: string // canonical: direct | group:<n> | node:<n> | egress:<n>
|
||||
Format: string
|
||||
UA: string
|
||||
HWID: string
|
||||
DeviceOS: string
|
||||
VerOS: string
|
||||
DeviceModel: string
|
||||
Headers: string[]
|
||||
Include: string[]
|
||||
Exclude: string[]
|
||||
FilterProto: string[]
|
||||
FilterCountry: string // comma-separated ISO codes; leading "!" excludes
|
||||
Dedup: boolean
|
||||
ExpireAlertDays: string
|
||||
Headers: HeaderRow[]
|
||||
}
|
||||
|
||||
function toDraft(s: Subscription, cat: DetourCatalog): SubDraft {
|
||||
return {
|
||||
Name: s.Name,
|
||||
URL: s.URL,
|
||||
UpdateInterval: s.UpdateInterval ?? '',
|
||||
FetchVia: s.FetchVia === 'proxy' ? 'proxy' : 'direct',
|
||||
FetchDetour: canonDetour(s.FetchDetour, cat),
|
||||
Format: s.Format && s.Format.trim() ? s.Format : 'auto',
|
||||
UA: s.UA ?? '',
|
||||
HWID: s.HWID ?? '',
|
||||
DeviceOS: s.DeviceOS ?? '',
|
||||
VerOS: s.VerOS ?? '',
|
||||
DeviceModel: s.DeviceModel ?? '',
|
||||
Headers: asArray(s.Headers),
|
||||
Include: asArray(s.Include),
|
||||
Exclude: asArray(s.Exclude),
|
||||
FilterProto: asArray(s.FilterProto),
|
||||
FilterCountry: asArray(s.FilterCountry).join(', '),
|
||||
Dedup: s.Dedup ?? false,
|
||||
ExpireAlertDays: s.ExpireAlertDays != null ? String(s.ExpireAlertDays) : '',
|
||||
Headers: asArray(s.Headers).map(toHeaderRow),
|
||||
}
|
||||
}
|
||||
|
||||
/** Fold a draft back onto the base sub — empties collapse to undefined so the
|
||||
* JSON stays lean, and Name/Enabled/URL (incl. the secret token) ride untouched. */
|
||||
* JSON stays lean. Spreading `base` first keeps every field the form no longer
|
||||
* shows (Format, device identity, filters, dedup, expire alert …) exactly as
|
||||
* stored; a blanked Name or URL falls back to the old value instead of wiping
|
||||
* the sub. */
|
||||
function applyDraft(base: Subscription, d: SubDraft): Subscription {
|
||||
const s = (v: string): string | undefined => (v.trim() ? v.trim() : undefined)
|
||||
const list = (a: string[]): string[] | undefined => (a.length ? a : undefined)
|
||||
const countries = d.FilterCountry.split(',')
|
||||
.map((c) => c.trim())
|
||||
.filter(Boolean)
|
||||
const days = Number.parseInt(d.ExpireAlertDays, 10)
|
||||
const proxy = d.FetchVia === 'proxy'
|
||||
// Rows with an empty key are dropped; the rest serialize back to the model's
|
||||
// raw "Key: value" strings, single space after the colon.
|
||||
const headers = d.Headers.filter((h) => h.key.trim()).map(
|
||||
(h) => `${h.key.trim()}: ${h.value.trim()}`,
|
||||
)
|
||||
return {
|
||||
...base,
|
||||
Name: d.Name.trim() || base.Name,
|
||||
URL: d.URL.trim() || base.URL,
|
||||
UpdateInterval: s(d.UpdateInterval),
|
||||
FetchVia: proxy ? 'proxy' : undefined,
|
||||
// Detour only rides along when fetching via proxy and it isn't plain Direct.
|
||||
FetchDetour: proxy && d.FetchDetour !== 'direct' ? d.FetchDetour : undefined,
|
||||
Format: d.Format && d.Format !== 'auto' ? d.Format : undefined,
|
||||
UA: s(d.UA),
|
||||
HWID: s(d.HWID),
|
||||
DeviceOS: s(d.DeviceOS),
|
||||
VerOS: s(d.VerOS),
|
||||
DeviceModel: s(d.DeviceModel),
|
||||
Headers: list(d.Headers),
|
||||
Include: list(d.Include),
|
||||
Exclude: list(d.Exclude),
|
||||
FilterProto: list(d.FilterProto),
|
||||
FilterCountry: countries.length ? countries : undefined,
|
||||
Dedup: d.Dedup ? true : undefined,
|
||||
ExpireAlertDays: Number.isFinite(days) && days > 0 ? days : undefined,
|
||||
Headers: headers.length ? headers : undefined,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1419,19 +1436,33 @@ function SubOptions({
|
||||
|
||||
const proxy = draft.FetchVia === 'proxy'
|
||||
const disabled = busy || saving
|
||||
const toggleProto = (p: string) =>
|
||||
set(
|
||||
'FilterProto',
|
||||
draft.FilterProto.includes(p)
|
||||
? draft.FilterProto.filter((x) => x !== p)
|
||||
: [...draft.FilterProto, p],
|
||||
)
|
||||
|
||||
return (
|
||||
<div id={id} className="sub-options">
|
||||
<fieldset className="opt-group" disabled={disabled}>
|
||||
<legend className="opt-legend">Fetch</legend>
|
||||
<div className="opt-grid">
|
||||
<OptField label="Name" hint="Renaming keeps the sub’s cached nodes attached">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.Name}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('Name', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="URL" wide hint="The feed address — a token in it stays stored, only masked in the list">
|
||||
<input
|
||||
className="fp-input"
|
||||
type="text"
|
||||
inputMode="url"
|
||||
value={draft.URL}
|
||||
placeholder="https://provider.example/sub?token=…"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('URL', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Update interval" hint="e.g. 30m · 6h · 24h — blank means manual only">
|
||||
<input
|
||||
className="fp-input"
|
||||
@@ -1467,22 +1498,6 @@ function SubOptions({
|
||||
/>
|
||||
</OptField>
|
||||
)}
|
||||
<OptField
|
||||
label="Format"
|
||||
hint="Auto reads whatever the feed turns out to be. Pick one to force that parser when auto guesses wrong."
|
||||
>
|
||||
<select
|
||||
className="fp-input"
|
||||
value={draft.Format}
|
||||
onChange={(e) => set('Format', e.target.value)}
|
||||
>
|
||||
{FORMATS.map((f) => (
|
||||
<option key={f} value={f}>
|
||||
{f}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
</OptField>
|
||||
</div>
|
||||
</fieldset>
|
||||
|
||||
@@ -1513,110 +1528,8 @@ function SubOptions({
|
||||
onChange={(e) => set('HWID', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Device OS">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.DeviceOS}
|
||||
placeholder="android"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('DeviceOS', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Ver OS">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.VerOS}
|
||||
placeholder="14"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('VerOS', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Device model">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.DeviceModel}
|
||||
placeholder="Pixel 8"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('DeviceModel', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
</div>
|
||||
<ChipList
|
||||
label="Extra headers"
|
||||
hint="One Key: value per chip — sent verbatim with the fetch"
|
||||
placeholder="X-Api-Key: …"
|
||||
value={draft.Headers}
|
||||
disabled={disabled}
|
||||
onChange={(v) => set('Headers', v)}
|
||||
/>
|
||||
</fieldset>
|
||||
|
||||
<fieldset className="opt-group" disabled={disabled}>
|
||||
<legend className="opt-legend">Filters</legend>
|
||||
<ChipList
|
||||
label="Include"
|
||||
hint="Keep only nodes whose name matches one of these regexes"
|
||||
placeholder="^🇩🇪|premium"
|
||||
value={draft.Include}
|
||||
disabled={disabled}
|
||||
onChange={(v) => set('Include', v)}
|
||||
/>
|
||||
<ChipList
|
||||
label="Exclude"
|
||||
hint="Drop nodes whose name matches — applied after include"
|
||||
placeholder="expire|traffic"
|
||||
value={draft.Exclude}
|
||||
disabled={disabled}
|
||||
onChange={(v) => set('Exclude', v)}
|
||||
/>
|
||||
<div className="opt-grid">
|
||||
<OptField label="Protocols" wide hint="Keep only these link types (none = keep all)">
|
||||
<div className="opt-checks">
|
||||
{PROTO_FILTERS.map((p) => (
|
||||
<label key={p} className="opt-check">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={draft.FilterProto.includes(p)}
|
||||
onChange={() => toggleProto(p)}
|
||||
/>
|
||||
<span className="mono">{p}</span>
|
||||
</label>
|
||||
))}
|
||||
</div>
|
||||
</OptField>
|
||||
<OptField label="Countries" hint="ISO codes, comma-separated; prefix ! to exclude">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.FilterCountry}
|
||||
placeholder="de, nl, !ru"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('FilterCountry', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Expire-alert days" hint="Warn this many days before the sub expires">
|
||||
<input
|
||||
className="fp-input"
|
||||
type="number"
|
||||
inputMode="numeric"
|
||||
min={0}
|
||||
value={draft.ExpireAlertDays}
|
||||
placeholder="0"
|
||||
onChange={(e) => set('ExpireAlertDays', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Dedup" hint="Collapse identical endpoints pulled more than once">
|
||||
<Toggle
|
||||
pressed={draft.Dedup}
|
||||
onChange={(v) => set('Dedup', v)}
|
||||
label={`${draft.Dedup ? 'Disable' : 'Enable'} dedup for ${sub.Name}`}
|
||||
disabled={disabled}
|
||||
/>
|
||||
</OptField>
|
||||
</div>
|
||||
<HeaderRows value={draft.Headers} disabled={disabled} onChange={(v) => set('Headers', v)} />
|
||||
</fieldset>
|
||||
|
||||
<div className="opt-actions">
|
||||
@@ -1666,42 +1579,54 @@ function OptField({
|
||||
)
|
||||
}
|
||||
|
||||
function ChipList({
|
||||
label,
|
||||
hint,
|
||||
placeholder,
|
||||
/** Key/Value row editor for the extra request headers. Rows map 1:1 onto the
|
||||
* model's raw "Key: value" strings (toHeaderRow / applyDraft do the split and
|
||||
* join); a row left with an empty key is dropped on save rather than
|
||||
* serialising a nameless header. */
|
||||
function HeaderRows({
|
||||
value,
|
||||
disabled,
|
||||
onChange,
|
||||
}: {
|
||||
label: string
|
||||
hint?: string
|
||||
placeholder: string
|
||||
value: string[]
|
||||
value: HeaderRow[]
|
||||
disabled?: boolean
|
||||
onChange: (next: string[]) => void
|
||||
onChange: (next: HeaderRow[]) => void
|
||||
}) {
|
||||
const [text, setText] = useState('')
|
||||
const add = () => {
|
||||
const v = text.trim()
|
||||
if (!v) return
|
||||
if (!value.includes(v)) onChange([...value, v])
|
||||
setText('')
|
||||
}
|
||||
const setRow = (i: number, patch: Partial<HeaderRow>) =>
|
||||
onChange(value.map((h, k) => (k === i ? { ...h, ...patch } : h)))
|
||||
return (
|
||||
<div className="opt-field opt-field--wide chip-field">
|
||||
<span className="opt-label mono">{label}</span>
|
||||
<div className="opt-field opt-field--wide hdr-field">
|
||||
<span className="opt-label mono">Extra headers</span>
|
||||
{value.length > 0 && (
|
||||
<ul className="chip-row">
|
||||
{value.map((v, i) => (
|
||||
<li key={`${v}-${i}`} className="chip">
|
||||
<span className="chip-text mono">{v}</span>
|
||||
<ul className="hdr-rows">
|
||||
{value.map((h, i) => (
|
||||
<li key={i} className="hdr-row">
|
||||
<input
|
||||
className="fp-input hdr-key"
|
||||
value={h.key}
|
||||
placeholder="X-Api-Key"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
aria-label={`Header ${i + 1} name`}
|
||||
disabled={disabled}
|
||||
onChange={(e) => setRow(i, { key: e.target.value })}
|
||||
/>
|
||||
<input
|
||||
className="fp-input"
|
||||
value={h.value}
|
||||
placeholder="value"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
aria-label={`Header ${i + 1} value`}
|
||||
disabled={disabled}
|
||||
onChange={(e) => setRow(i, { value: e.target.value })}
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
className="chip-x"
|
||||
className="hdr-x"
|
||||
onClick={() => onChange(value.filter((_, k) => k !== i))}
|
||||
disabled={disabled}
|
||||
aria-label={`Remove ${v}`}
|
||||
aria-label={`Remove header ${h.key.trim() || i + 1}`}
|
||||
>
|
||||
×
|
||||
</button>
|
||||
@@ -1709,28 +1634,18 @@ function ChipList({
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
<div className="chip-add">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={text}
|
||||
placeholder={placeholder}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
aria-label={`Add ${label}`}
|
||||
<div>
|
||||
<Button
|
||||
className="row-del"
|
||||
onClick={() => onChange([...value, { key: '', value: '' }])}
|
||||
disabled={disabled}
|
||||
onChange={(e) => setText(e.target.value)}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === 'Enter') {
|
||||
e.preventDefault()
|
||||
add()
|
||||
}
|
||||
}}
|
||||
/>
|
||||
<Button className="row-del" onClick={add} disabled={disabled || !text.trim()}>
|
||||
Add
|
||||
>
|
||||
Add header
|
||||
</Button>
|
||||
</div>
|
||||
{hint && <span className="opt-hint">{hint}</span>}
|
||||
<span className="opt-hint">
|
||||
Sent verbatim with the fetch — a row without a name is dropped on save
|
||||
</span>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -1,14 +1,13 @@
|
||||
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||
import { Button, Led, Module, QueryLog, SegMeter } from '../components'
|
||||
import type { LedVariant, QueryEntry, QueryTag } from '../components'
|
||||
import { fmtClock, fmtDateTime, fmtDuration } from '../format'
|
||||
import { Button, Led, Module } from '../components'
|
||||
import type { LedVariant } from '../components'
|
||||
import { fmtDateTime, fmtDuration } from '../format'
|
||||
import {
|
||||
apply as apiApply,
|
||||
confirm as apiConfirm,
|
||||
rollback as apiRollback,
|
||||
getConfig,
|
||||
getStats,
|
||||
getStatsLog,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Model, Stats, Status, StatusWarning } from '../api'
|
||||
@@ -28,12 +27,6 @@ function short(hash: string): string {
|
||||
return h.length > 12 ? h.slice(0, 12) : h
|
||||
}
|
||||
|
||||
function actionTag(action?: string): QueryTag {
|
||||
if (action === 'block') return 'block'
|
||||
if (action === 'proxy') return 'proxy'
|
||||
return 'pass'
|
||||
}
|
||||
|
||||
type ControlKind = 'apply' | 'confirm' | 'rollback'
|
||||
|
||||
/**
|
||||
@@ -100,50 +93,19 @@ export function Overview({
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// ---- live query log + filter stats: poll both, degrade to honest empty states ----
|
||||
const [entries, setEntries] = useState<QueryEntry[]>([])
|
||||
// ---- live filter stats: poll the aggregate snapshot, degrade to honest empty states ----
|
||||
const [stats, setStats] = useState<Stats | null>(null)
|
||||
const [statsOk, setStatsOk] = useState(true)
|
||||
// `loggingOff` is read back out of the snapshot each tick so the poll can stop
|
||||
// asking for a log the daemon isn't keeping. A ref (not state) keeps the effect
|
||||
// stable — it must not re-subscribe every time the snapshot lands.
|
||||
const loggingOffRef = useRef(false)
|
||||
useEffect(() => {
|
||||
let alive = true
|
||||
const tick = async () => {
|
||||
// Nothing to poll for while the tab is in the background.
|
||||
if (document.hidden) return
|
||||
try {
|
||||
// Logging off ⇒ the log endpoint has nothing to give; skip it and keep
|
||||
// polling the snapshot alone so the page notices it being switched back on.
|
||||
const [log, s] = await Promise.all([
|
||||
loggingOffRef.current ? Promise.resolve([]) : getStatsLog(50),
|
||||
getStats(),
|
||||
])
|
||||
const s = await getStats()
|
||||
if (!alive) return
|
||||
loggingOffRef.current = s.backend === 'off'
|
||||
setStatsOk(true)
|
||||
setStats(s)
|
||||
const rows = Array.isArray(log) ? log : []
|
||||
setEntries(
|
||||
rows.slice(0, 10).map((r) => ({
|
||||
// The resolution unix-time + domain is a stable identity for the row,
|
||||
// so React only replays the slide-in for genuinely new queries.
|
||||
id: `${r.unix}-${r.domain}-${r.qtype}`,
|
||||
// Format from unix in the browser's local timezone (shared helper) so
|
||||
// this log matches the Insights logs — not the server's UTC `time`.
|
||||
time: fmtClock(r.unix) || (r.time ?? ''),
|
||||
domain: r.domain,
|
||||
// The REAL query source, same as the Insights DNS log: the LAN
|
||||
// client's hostname/IP, or 'router' for the appliance's own lookups.
|
||||
// (The resolver lives in the Insights log's own column.) '' only on
|
||||
// rows persisted before device attribution existed.
|
||||
device: r.device || '—',
|
||||
tag: actionTag(r.action),
|
||||
})),
|
||||
)
|
||||
} catch {
|
||||
if (alive) setStatsOk(false)
|
||||
// Transient — the interval retries, and the modules keep their last reading.
|
||||
}
|
||||
}
|
||||
void tick()
|
||||
@@ -463,64 +425,6 @@ export function Overview({
|
||||
/>
|
||||
</div>
|
||||
|
||||
<div className="log-wrap">
|
||||
{hasStats && (
|
||||
<div className="filter-readout">
|
||||
<SegMeter label="Filtered" value={blockedPct} max={100} unit="%" />
|
||||
{topBlocked.length > 0 ? (
|
||||
<ul className="topblocked" aria-label="Top blocked domains">
|
||||
{topBlocked.map((d) => (
|
||||
<li key={d.domain}>
|
||||
<span className="tb-dom">{d.domain}</span>
|
||||
<span className="tb-n">{d.blocked}</span>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
) : (
|
||||
<p className="qempty" style={{ margin: 0 }}>
|
||||
{blocked > 0 ? 'blocked queries logged' : 'nothing blocked yet'}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
<QueryLog
|
||||
title="Query log"
|
||||
led={!statsOk ? 'crit' : loggingOff ? 'off' : entries.length > 0 ? 'on' : 'amber'}
|
||||
hint={
|
||||
!statsOk
|
||||
? 'stats unavailable'
|
||||
: loggingOff
|
||||
? 'logging off'
|
||||
: hasStats
|
||||
? `${queries} queries · ${blocked} blocked`
|
||||
: // Aggregates can reset on a daemon restart while the persistent
|
||||
// log still streams rows — don't claim "waiting" when the log is
|
||||
// clearly live; only show it when there's genuinely nothing.
|
||||
entries.length > 0
|
||||
? 'recent queries'
|
||||
: 'waiting for engine stats'
|
||||
}
|
||||
entries={entries}
|
||||
/>
|
||||
{entries.length === 0 && (
|
||||
<p className="qempty">
|
||||
{!statsOk ? (
|
||||
'Stats endpoint unreachable — retrying every few seconds.'
|
||||
) : loggingOff ? (
|
||||
<>
|
||||
Logging is off — no DNS, connection, or per-device history is recorded.{' '}
|
||||
<a className="linkish" href="#/settings">
|
||||
Turn it on in Settings → Logging backend
|
||||
</a>
|
||||
.
|
||||
</>
|
||||
) : (
|
||||
'No query data yet — the live stream lights up once the engine resolves DNS (needs an active subscriber + traffic).'
|
||||
)}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Apply / Confirm / Rollback — active voice, honest results. */}
|
||||
<div className="controls" role="group" aria-label="Config actions">
|
||||
<div className="controls-btns">
|
||||
|
||||
@@ -142,9 +142,6 @@
|
||||
width: 7rem;
|
||||
flex: none;
|
||||
}
|
||||
.pf-field--time {
|
||||
width: auto;
|
||||
}
|
||||
.pf-flabel {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
@@ -315,9 +312,6 @@
|
||||
color: var(--faint);
|
||||
font-style: italic;
|
||||
}
|
||||
.pf-chip--target {
|
||||
border-color: color-mix(in srgb, var(--accent) 30%, var(--groove));
|
||||
}
|
||||
.pf-chip--on {
|
||||
border-color: color-mix(in srgb, var(--led-on) 45%, var(--groove));
|
||||
color: var(--led-on);
|
||||
@@ -432,9 +426,6 @@
|
||||
.pf-erow-ctl {
|
||||
min-width: 0;
|
||||
}
|
||||
/* (The .pf-erow--muted dimming that used to sit here is gone with its reason:
|
||||
the watcher DOES evaluate an iface-driven profile's schedule now, so the
|
||||
schedule fields are never inert.) */
|
||||
.pf-ehint {
|
||||
margin: 6px 0 0;
|
||||
font-family: var(--font-sans);
|
||||
@@ -477,63 +468,6 @@
|
||||
outline-offset: 1px;
|
||||
border-radius: 2px;
|
||||
}
|
||||
.pf-chipinput-in {
|
||||
flex: 1 1 8rem;
|
||||
border: 0;
|
||||
background: none;
|
||||
box-shadow: none;
|
||||
padding: 3px 4px;
|
||||
}
|
||||
.pf-chipinput-in:focus-visible {
|
||||
outline: none;
|
||||
}
|
||||
.pf-chipinput:focus-within {
|
||||
border-color: var(--accent);
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 1px;
|
||||
}
|
||||
|
||||
/* ---- schedule ---- */
|
||||
.pf-days {
|
||||
display: inline-flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 4px;
|
||||
}
|
||||
.pf-day {
|
||||
padding: 6px 9px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 6px;
|
||||
background: var(--sink);
|
||||
color: var(--dim);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.02em;
|
||||
cursor: pointer;
|
||||
transition: color 0.15s, border-color 0.15s, background 0.15s;
|
||||
}
|
||||
.pf-day:hover:not(:disabled) {
|
||||
color: var(--ink);
|
||||
}
|
||||
.pf-day.on {
|
||||
color: var(--accent);
|
||||
border-color: color-mix(in srgb, var(--accent) 55%, var(--groove));
|
||||
background: color-mix(in srgb, var(--accent) 12%, transparent);
|
||||
}
|
||||
.pf-day:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 2px;
|
||||
}
|
||||
.pf-day:disabled {
|
||||
opacity: 0.55;
|
||||
cursor: default;
|
||||
}
|
||||
.pf-times {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: flex-end;
|
||||
gap: 12px;
|
||||
margin-top: 10px;
|
||||
}
|
||||
|
||||
/* ---- rule on/off matrix ---- */
|
||||
.pf-rules {
|
||||
@@ -594,56 +528,6 @@
|
||||
cursor: default;
|
||||
}
|
||||
|
||||
/* ---- preset packs ---- */
|
||||
.pf-pack-desc {
|
||||
margin: 6px 0 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 12px;
|
||||
line-height: 1.5;
|
||||
color: var(--dim);
|
||||
max-width: 60ch;
|
||||
}
|
||||
.pf-pack-target {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
align-items: flex-end;
|
||||
gap: 6px;
|
||||
}
|
||||
.pf-adv {
|
||||
border-top: 1px solid var(--groove);
|
||||
}
|
||||
.pf-adv-summary {
|
||||
padding: 9px 14px;
|
||||
cursor: pointer;
|
||||
list-style: none;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: var(--track-label, 0.16em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.pf-adv-summary::-webkit-details-marker {
|
||||
display: none;
|
||||
}
|
||||
.pf-adv-summary::before {
|
||||
content: '▸ ';
|
||||
color: var(--faint);
|
||||
}
|
||||
.pf-adv[open] .pf-adv-summary::before {
|
||||
content: '▾ ';
|
||||
}
|
||||
.pf-adv-summary:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: -2px;
|
||||
border-radius: 4px;
|
||||
}
|
||||
.pf-adv-body {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 2px;
|
||||
padding: 0 14px 14px;
|
||||
}
|
||||
|
||||
/* ---- empty + skeleton ---- */
|
||||
.pf-empty {
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
@@ -706,10 +590,6 @@
|
||||
grid-column: 1 / -1;
|
||||
justify-content: flex-end;
|
||||
}
|
||||
.pf-pack-target {
|
||||
grid-column: 1 / -1;
|
||||
align-items: stretch;
|
||||
}
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
@@ -717,7 +597,6 @@
|
||||
animation: none;
|
||||
}
|
||||
.pf-input,
|
||||
.pf-day,
|
||||
.pf-edit,
|
||||
.pf-del {
|
||||
transition: none;
|
||||
|
||||
+101
-535
@@ -1,19 +1,16 @@
|
||||
import './Profiles.css'
|
||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
||||
import { Button, Led, Toggle } from '../components'
|
||||
import { apply as apiApply, getConfig, putConfig, ApiError } from '../api'
|
||||
import type { Model, Preset, Profile } from '../api'
|
||||
import { apply as apiApply, getConfig, getInterfaces, putConfig, ApiError } from '../api'
|
||||
import type { Interface, Model, Profile } from '../api'
|
||||
|
||||
// The Profiles page is a thin editor over the desired-state Model — the same
|
||||
// save→apply split as Settings / DNS / Routing. Every edit rewrites a slice in
|
||||
// place, PUTs the whole Model (save), marks the config dirty, and only Apply
|
||||
// pushes it onto the live data plane. Two sections:
|
||||
//
|
||||
// 1. PROFILES — named WAN-mode / failover overrides that activate by
|
||||
// condition (uplink iface, schedule). Manual override lives in
|
||||
// Globals.ActiveProfile ("" = auto, evaluated on the router).
|
||||
// 2. PRESET PACKS — three built-in curated rule bundles (block-ads / ru-bypass
|
||||
// / private). Toggling one upserts the matching entry in Model.Presets.
|
||||
// pushes it onto the live data plane. One section: PROFILES — named WAN-mode /
|
||||
// failover overrides that activate when the router's default-route uplink
|
||||
// matches. Manual override lives in Globals.ActiveProfile ("" = auto,
|
||||
// evaluated on the router).
|
||||
|
||||
// ---- helpers ---------------------------------------------------------------
|
||||
|
||||
@@ -27,59 +24,7 @@ const asArray = <T,>(a: T[] | null | undefined): T[] => (a ? a : [])
|
||||
const ruleSummary = (names: string[], max = 2): string =>
|
||||
names.length <= max ? names.join(', ') : `${names.slice(0, max).join(', ')} +${names.length - max} more`
|
||||
|
||||
// Weekday tokens as stored on the model (mon..sun), in display order.
|
||||
const DAYS = ['mon', 'tue', 'wed', 'thu', 'fri', 'sat', 'sun'] as const
|
||||
const cap = (s: string): string => (s ? s[0].toUpperCase() + s.slice(1) : s)
|
||||
|
||||
// ---- schedule timezone anchoring --------------------------------------------
|
||||
// The router carries no IANA tzdata, so the daemon cannot resolve a timezone
|
||||
// NAME — it evaluates schedule windows at a fixed UTC offset (minutes east,
|
||||
// model SchedUTCOffset). The panel's job is to capture the editing browser's
|
||||
// offset alongside every schedule edit, so "22:00" means 22:00 on the clock of
|
||||
// whoever wrote it. Known limitation (stated in the hint): a fixed offset does
|
||||
// not follow DST until the schedule is re-saved.
|
||||
|
||||
/** The browser's current UTC offset in minutes EAST (Moscow ⇒ 180). */
|
||||
const browserUTCOffsetMin = (): number => -new Date().getTimezoneOffset()
|
||||
|
||||
/** "UTC+03:00" / "UTC−05:00" for a minutes-east offset. */
|
||||
const utcOffsetLabel = (min: number): string => {
|
||||
const sign = min < 0 ? '−' : '+'
|
||||
const a = Math.abs(min)
|
||||
const hh = String(Math.floor(a / 60)).padStart(2, '0')
|
||||
const mm = String(a % 60).padStart(2, '0')
|
||||
return `UTC${sign}${hh}:${mm}`
|
||||
}
|
||||
|
||||
/** IANA-style name of the browser zone ("Europe/Moscow"), best-effort. */
|
||||
const browserTZName = (): string => {
|
||||
try {
|
||||
return Intl.DateTimeFormat().resolvedOptions().timeZone || 'local time'
|
||||
} catch {
|
||||
return 'local time'
|
||||
}
|
||||
}
|
||||
|
||||
// The three built-in packs. Fixed set — the operator toggles them, never adds or
|
||||
// removes. Name is the on-disk `config preset` name the engine expands.
|
||||
const PACKS: ReadonlyArray<{ name: string; title: string; desc: string }> = [
|
||||
{ name: 'block-ads', title: 'Block ads', desc: 'Blocks known ad & tracker domains.' },
|
||||
{ name: 'ru-bypass', title: 'Russia bypass', desc: 'Routes Russian services direct (no proxy).' },
|
||||
{ name: 'private', title: 'Private ranges', desc: 'Keeps LAN / private ranges off the proxy.' },
|
||||
]
|
||||
|
||||
// ---- target picker ---------------------------------------------------------
|
||||
// A routing target in the model's canonical form: `direct` | `block` |
|
||||
// `group:<n>` | `chain:<n>` | `node:<n>` | `egress:<n>`. Built entirely from the
|
||||
// live Model, exactly like the Routing / DNS pickers.
|
||||
|
||||
interface TargetCatalog {
|
||||
groups: string[]
|
||||
chains: string[]
|
||||
nodes: string[]
|
||||
egresses: { name: string; type: string }[]
|
||||
}
|
||||
|
||||
// Pull Name strings out of a model slice (Rules / Resolvers) for the pickers.
|
||||
type Named = { Name?: unknown }
|
||||
function namesOf(v: unknown): string[] {
|
||||
if (!Array.isArray(v)) return []
|
||||
@@ -88,16 +33,6 @@ function namesOf(v: unknown): string[] {
|
||||
.filter((n): n is string => typeof n === 'string' && n.length > 0)
|
||||
}
|
||||
|
||||
/** Every valid canonical target value for a catalog (excludes the empty option). */
|
||||
function targetValues(cat: TargetCatalog): Set<string> {
|
||||
const s = new Set<string>(['direct', 'block'])
|
||||
for (const g of cat.groups) s.add(`group:${g}`)
|
||||
for (const c of cat.chains) s.add(`chain:${c}`)
|
||||
for (const n of cat.nodes) s.add(`node:${n}`)
|
||||
for (const e of cat.egresses) s.add(`egress:${e.name}`)
|
||||
return s
|
||||
}
|
||||
|
||||
// ---- page ------------------------------------------------------------------
|
||||
|
||||
export default function Profiles() {
|
||||
@@ -117,6 +52,24 @@ export default function Profiles() {
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// The router's UCI interfaces feed the uplink-condition picker. Best-effort:
|
||||
// if the list can't be fetched the editor keeps existing chips removable and
|
||||
// simply offers nothing to add (same degradation as the Targets egress picker).
|
||||
const [interfaces, setInterfaces] = useState<Interface[]>([])
|
||||
useEffect(() => {
|
||||
let alive = true
|
||||
getInterfaces()
|
||||
.then((ifs) => {
|
||||
if (alive) setInterfaces(ifs)
|
||||
})
|
||||
.catch(() => {
|
||||
/* leave empty → chips stay removable, nothing to add */
|
||||
})
|
||||
return () => {
|
||||
alive = false
|
||||
}
|
||||
}, [])
|
||||
|
||||
// ---- toast + persistent apply banner (mirrors DNS / Settings) -------------
|
||||
const [toast, setToast] = useState<string | null>(null)
|
||||
const toastTimer = useRef<number | undefined>(undefined)
|
||||
@@ -174,22 +127,9 @@ export default function Profiles() {
|
||||
// ---- derived slices -------------------------------------------------------
|
||||
const globals = config?.Globals
|
||||
const profiles = useMemo<Profile[]>(() => asArray(config?.Profiles), [config])
|
||||
const presets = useMemo<Preset[]>(() => asArray(config?.Presets), [config])
|
||||
const ruleNames = useMemo(() => namesOf(config?.Rules), [config])
|
||||
const egresses = useMemo(() => asArray(config?.Egresses), [config])
|
||||
const resolverNames = useMemo(() => namesOf(config?.Resolvers), [config])
|
||||
|
||||
const catalog = useMemo<TargetCatalog>(
|
||||
() => ({
|
||||
groups: namesOf(config?.Groups),
|
||||
chains: namesOf(config?.['Chains']),
|
||||
nodes: namesOf(config?.Nodes),
|
||||
egresses: egresses.map((e) => ({ name: e.Name, type: e.Type })),
|
||||
}),
|
||||
[config, egresses],
|
||||
)
|
||||
const targetValid = useMemo(() => targetValues(catalog), [catalog])
|
||||
|
||||
const profileNames = useMemo(() => new Set(profiles.map((p) => p.Name)), [profiles])
|
||||
const active = globals?.ActiveProfile ?? ''
|
||||
const busy = saving || applying
|
||||
@@ -226,7 +166,6 @@ export default function Profiles() {
|
||||
Enabled: true,
|
||||
Priority: priority,
|
||||
MatchIface: [],
|
||||
SchedDays: [],
|
||||
EnableRules: [],
|
||||
DisableRules: [],
|
||||
}
|
||||
@@ -266,19 +205,6 @@ export default function Profiles() {
|
||||
[config, profiles, save],
|
||||
)
|
||||
|
||||
// ---- preset-pack upsert ---------------------------------------------------
|
||||
const setPreset = useCallback(
|
||||
(name: string, patch: Partial<Preset>, msg: string) => {
|
||||
if (!config) return
|
||||
const exists = presets.some((p) => p.Name === name)
|
||||
const next = exists
|
||||
? presets.map((p) => (p.Name === name ? { ...p, ...patch } : p))
|
||||
: [...presets, { Name: name, Enabled: false, ...patch }]
|
||||
void save({ ...config, Presets: next }, msg)
|
||||
},
|
||||
[config, presets, save],
|
||||
)
|
||||
|
||||
// ---- expansion (only one profile editor open at a time) -------------------
|
||||
const [openName, setOpenName] = useState<string | null>(null)
|
||||
// Rename remounts nothing (row keyed by index) but the open pointer must follow.
|
||||
@@ -319,7 +245,7 @@ export default function Profiles() {
|
||||
const loading = config === null && loadError === null
|
||||
|
||||
return (
|
||||
<section className="page pf-page" aria-label="Profiles and preset packs">
|
||||
<section className="page pf-page" aria-label="Profiles">
|
||||
{loadError && (
|
||||
<p className="page-error" role="alert">
|
||||
Couldn’t read config — {loadError}.{' '}
|
||||
@@ -351,7 +277,7 @@ export default function Profiles() {
|
||||
onChange={setActive}
|
||||
/>
|
||||
|
||||
{/* ---- 1. PROFILES ---- */}
|
||||
{/* ---- PROFILES ---- */}
|
||||
<div className="pf-section" aria-label="Profiles">
|
||||
<header className="pf-sec-hd">
|
||||
<h2 className="pf-sec-title">Profiles</h2>
|
||||
@@ -360,9 +286,9 @@ export default function Profiles() {
|
||||
</span>
|
||||
</header>
|
||||
<p className="pf-sec-note">
|
||||
A profile is a named override that switches on by condition — which uplink is carrying the
|
||||
router, or a time window. When active it can flip rules on or off and change the default
|
||||
route. Higher priority wins; a manual override above beats every condition.
|
||||
A profile is a named override that switches on by condition — which uplink is carrying
|
||||
the router. When active it can flip rules on or off. Higher priority wins; a manual
|
||||
override above beats every condition.
|
||||
</p>
|
||||
|
||||
<AddProfileForm busy={busy} disabled={!config} taken={profileNames} onAdd={addProfile} />
|
||||
@@ -376,7 +302,7 @@ export default function Profiles() {
|
||||
<div className="pf-empty">
|
||||
<span className="pf-empty-title mono">No profiles</span>
|
||||
<p className="pf-empty-body">
|
||||
No profiles — add one to auto-switch routing by uplink or schedule.
|
||||
No profiles — add one to auto-switch routing by the active uplink.
|
||||
</p>
|
||||
</div>
|
||||
) : (
|
||||
@@ -391,10 +317,8 @@ export default function Profiles() {
|
||||
onOpen={() => setOpenName(openName === p.Name ? null : p.Name)}
|
||||
busy={busy}
|
||||
ruleNames={ruleNames}
|
||||
egresses={egresses}
|
||||
interfaces={interfaces}
|
||||
resolverNames={resolverNames}
|
||||
catalog={catalog}
|
||||
valid={targetValid}
|
||||
taken={profileNames}
|
||||
onPatch={(patch, msg) => patchProfile(p.Name, patch, msg)}
|
||||
onRename={(nn) => onRename(p.Name, nn)}
|
||||
@@ -405,38 +329,6 @@ export default function Profiles() {
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* ---- 2. PRESET PACKS ---- */}
|
||||
<div className="pf-section" aria-label="Preset packs">
|
||||
<header className="pf-sec-hd">
|
||||
<h2 className="pf-sec-title">Preset packs</h2>
|
||||
<span className="pf-sec-count mono">
|
||||
{presets.filter((p) => p.Enabled).length} / {PACKS.length} on
|
||||
</span>
|
||||
</header>
|
||||
<p className="pf-sec-note">
|
||||
Curated rule bundles, built in. Turn one on to add its rules to the router; set a target to
|
||||
override where its traffic goes.
|
||||
</p>
|
||||
|
||||
<ul className="pf-rows">
|
||||
{PACKS.map((pack) => {
|
||||
const entry = presets.find((p) => p.Name === pack.name)
|
||||
return (
|
||||
<PackRow
|
||||
key={pack.name}
|
||||
pack={pack}
|
||||
preset={entry}
|
||||
busy={busy}
|
||||
disabled={!config}
|
||||
catalog={catalog}
|
||||
valid={targetValid}
|
||||
onSet={(patch, msg) => setPreset(pack.name, patch, msg)}
|
||||
/>
|
||||
)
|
||||
})}
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
{toast && (
|
||||
<div className="toast" role="status">
|
||||
{toast}
|
||||
@@ -660,10 +552,8 @@ function ProfileRow({
|
||||
onOpen,
|
||||
busy,
|
||||
ruleNames,
|
||||
egresses,
|
||||
interfaces,
|
||||
resolverNames,
|
||||
catalog,
|
||||
valid,
|
||||
taken,
|
||||
onPatch,
|
||||
onRename,
|
||||
@@ -676,10 +566,8 @@ function ProfileRow({
|
||||
onOpen: () => void
|
||||
busy: boolean
|
||||
ruleNames: string[]
|
||||
egresses: { Name: string }[]
|
||||
interfaces: Interface[]
|
||||
resolverNames: string[]
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
taken: Set<string>
|
||||
onPatch: (patch: Partial<Profile>, msg: string) => void
|
||||
onRename: (newName: string) => void
|
||||
@@ -688,8 +576,6 @@ function ProfileRow({
|
||||
const iface = asArray(profile.MatchIface)
|
||||
const enableRules = asArray(profile.EnableRules)
|
||||
const disableRules = asArray(profile.DisableRules)
|
||||
const days = asArray(profile.SchedDays)
|
||||
const hasSchedule = days.length > 0 || !!profile.SchedStart || !!profile.SchedEnd
|
||||
|
||||
return (
|
||||
<li className={profile.Enabled ? 'pf-row' : 'pf-row off'}>
|
||||
@@ -721,42 +607,16 @@ function ProfileRow({
|
||||
)}
|
||||
</div>
|
||||
<div className="pf-row-l2">
|
||||
{!hasSchedule && iface.length === 0 ? (
|
||||
{iface.length === 0 ? (
|
||||
<span className="pf-chip pf-chip--muted">always — no condition set</span>
|
||||
) : (
|
||||
<>
|
||||
{iface.length > 0 && (
|
||||
<span className="pf-chip">
|
||||
<b>iface</b>
|
||||
<span className="mono">{iface.join(', ')}</span>
|
||||
</span>
|
||||
)}
|
||||
{hasSchedule && (
|
||||
<span className="pf-chip">
|
||||
<b>time</b>
|
||||
<span className="mono">
|
||||
{days.length ? days.map(cap).join(',') + ' ' : ''}
|
||||
{(profile.SchedStart || '00:00') + '–' + (profile.SchedEnd || '24:00')}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
{(enableRules.length > 0 ||
|
||||
disableRules.length > 0 ||
|
||||
!!profile.DefaultTarget ||
|
||||
!!profile.DefaultEgress) && <span className="pf-chip-sep" aria-hidden="true">→</span>}
|
||||
{profile.DefaultTarget && (
|
||||
<span className="pf-chip pf-chip--target">
|
||||
<b>route</b>
|
||||
<span className="mono">{profile.DefaultTarget}</span>
|
||||
<span className="pf-chip">
|
||||
<b>iface</b>
|
||||
<span className="mono">{iface.join(', ')}</span>
|
||||
</span>
|
||||
)}
|
||||
{profile.DefaultEgress && (
|
||||
<span className="pf-chip pf-chip--target">
|
||||
<b>egress</b>
|
||||
<span className="mono">{profile.DefaultEgress}</span>
|
||||
</span>
|
||||
{(enableRules.length > 0 || disableRules.length > 0) && (
|
||||
<span className="pf-chip-sep" aria-hidden="true">→</span>
|
||||
)}
|
||||
{enableRules.length > 0 && (
|
||||
<span
|
||||
@@ -803,10 +663,8 @@ function ProfileRow({
|
||||
profile={profile}
|
||||
busy={busy}
|
||||
ruleNames={ruleNames}
|
||||
egresses={egresses}
|
||||
interfaces={interfaces}
|
||||
resolverNames={resolverNames}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
taken={taken}
|
||||
onPatch={onPatch}
|
||||
onRename={onRename}
|
||||
@@ -822,10 +680,8 @@ function ProfileEditor({
|
||||
profile,
|
||||
busy,
|
||||
ruleNames,
|
||||
egresses,
|
||||
interfaces,
|
||||
resolverNames,
|
||||
catalog,
|
||||
valid,
|
||||
taken,
|
||||
onPatch,
|
||||
onRename,
|
||||
@@ -833,10 +689,8 @@ function ProfileEditor({
|
||||
profile: Profile
|
||||
busy: boolean
|
||||
ruleNames: string[]
|
||||
egresses: { Name: string }[]
|
||||
interfaces: Interface[]
|
||||
resolverNames: string[]
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
taken: Set<string>
|
||||
onPatch: (patch: Partial<Profile>, msg: string) => void
|
||||
onRename: (newName: string) => void
|
||||
@@ -844,7 +698,6 @@ function ProfileEditor({
|
||||
const iface = asArray(profile.MatchIface)
|
||||
const enableRules = asArray(profile.EnableRules)
|
||||
const disableRules = asArray(profile.DisableRules)
|
||||
const days = asArray(profile.SchedDays)
|
||||
|
||||
// Name is the only field that can't be a bare instant-save (needs a uniqueness
|
||||
// guard + it moves the ActiveProfile pointer), so it commits on blur/Enter.
|
||||
@@ -862,18 +715,6 @@ function ProfileEditor({
|
||||
const removeIface = (dev: string) =>
|
||||
onPatch({ MatchIface: iface.filter((x) => x !== dev) }, `iface − ${dev}`)
|
||||
|
||||
// Every schedule edit re-anchors the window to the editing browser's UTC
|
||||
// offset — the daemon has no tzdata, so the offset IS the timezone (see the
|
||||
// schedule helpers at the top of the file).
|
||||
const patchSched = (patch: Partial<Profile>, msg: string) =>
|
||||
onPatch({ ...patch, SchedUTCOffset: browserUTCOffsetMin() }, msg)
|
||||
|
||||
const toggleDay = (d: string) =>
|
||||
patchSched(
|
||||
{ SchedDays: days.includes(d) ? days.filter((x) => x !== d) : [...days, d] },
|
||||
'Schedule updated',
|
||||
)
|
||||
|
||||
// A rule sits in at most one override list. Checking it in one clears the other.
|
||||
const toggleRule = (rule: string, list: 'enable' | 'disable') => {
|
||||
if (list === 'enable') {
|
||||
@@ -917,15 +758,14 @@ function ProfileEditor({
|
||||
|
||||
{/* -- conditions -- */}
|
||||
<fieldset className="pf-eblock">
|
||||
<legend className="pf-elegend">Conditions — all must hold</legend>
|
||||
<legend className="pf-elegend">Condition</legend>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Uplink interface</span>
|
||||
<div className="pf-erow-ctl">
|
||||
<ChipInput
|
||||
<IfaceChips
|
||||
chips={iface}
|
||||
placeholder="wwan0, usb0…"
|
||||
ariaLabel="Uplink interface names"
|
||||
interfaces={interfaces}
|
||||
busy={busy}
|
||||
onAdd={addIface}
|
||||
onRemove={removeIface}
|
||||
@@ -934,131 +774,17 @@ function ProfileEditor({
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* A "Probe" condition (URL + activate-when-up/down) used to sit here. It
|
||||
never probed anything, and it did worse than nothing: a profile that
|
||||
carried one was treated as having a condition that could never hold, so
|
||||
filling this in SWITCHED OFF an otherwise working profile. Both fields
|
||||
are gone from the daemon; the conditions that remain are the two below,
|
||||
which are read for real. */}
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Schedule</span>
|
||||
<div className="pf-erow-ctl">
|
||||
{iface.length > 0 && (
|
||||
<p className="pf-ehint">
|
||||
Applies together with the uplink match: the profile is active only while the uplink
|
||||
matches AND the window is open (the watcher re-checks both every ~25 s).
|
||||
</p>
|
||||
)}
|
||||
<div className="pf-days" role="group" aria-label="Active days (none = every day)">
|
||||
{DAYS.map((d) => {
|
||||
const on = days.includes(d)
|
||||
return (
|
||||
<button
|
||||
type="button"
|
||||
key={d}
|
||||
className={on ? 'pf-day on' : 'pf-day'}
|
||||
aria-pressed={on}
|
||||
onClick={() => toggleDay(d)}
|
||||
disabled={busy}
|
||||
>
|
||||
{cap(d)}
|
||||
</button>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
<div className="pf-times">
|
||||
<label className="pf-field pf-field--time">
|
||||
<span className="pf-flabel">From</span>
|
||||
<input
|
||||
className="pf-input mono"
|
||||
type="time"
|
||||
value={profile.SchedStart ?? ''}
|
||||
onChange={(e) => patchSched({ SchedStart: e.target.value }, 'Schedule updated')}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
<label className="pf-field pf-field--time">
|
||||
<span className="pf-flabel">To</span>
|
||||
<input
|
||||
className="pf-input mono"
|
||||
type="time"
|
||||
value={profile.SchedEnd ?? ''}
|
||||
onChange={(e) => patchSched({ SchedEnd: e.target.value }, 'Schedule updated')}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
</div>
|
||||
<p className="pf-ehint">
|
||||
No days = every day · empty times = all day · times are in your browser's timezone (
|
||||
{browserTZName()}, {utcOffsetLabel(browserUTCOffsetMin())}) — the offset is saved with the
|
||||
schedule; DST shifts apply after a re-save.
|
||||
</p>
|
||||
{(profile.SchedUTCOffset ?? 0) !== browserUTCOffsetMin() &&
|
||||
((profile.SchedStart ?? '') !== '' ||
|
||||
(profile.SchedEnd ?? '') !== '' ||
|
||||
days.length > 0) && (
|
||||
<p className="pf-ehint">
|
||||
This schedule was saved at {utcOffsetLabel(profile.SchedUTCOffset ?? 0)} — the times
|
||||
above are on that clock. Editing any schedule field re-anchors it to your timezone.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
{/* A "Probe" condition (URL + activate-when-up/down) used to sit here, and
|
||||
a schedule window below it. The probe never probed anything and treated
|
||||
a profile carrying one as unsatisfiable — filling it in SWITCHED OFF an
|
||||
otherwise working profile; the schedule editor left with the feature.
|
||||
The uplink match above is the one condition left, read for real. */}
|
||||
</fieldset>
|
||||
|
||||
{/* -- overrides -- */}
|
||||
<fieldset className="pf-eblock">
|
||||
<legend className="pf-elegend">Overrides while active</legend>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Default route</span>
|
||||
<div className="pf-erow-ctl">
|
||||
<TargetSelect
|
||||
value={profile.DefaultTarget ?? ''}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
includeNoOverride
|
||||
noOverrideLabel="No override — keep default"
|
||||
busy={busy}
|
||||
ariaLabel="Default route target"
|
||||
onChange={(v) =>
|
||||
onPatch({ DefaultTarget: v }, v ? `Default route → ${v}` : 'Default route override cleared')
|
||||
}
|
||||
/>
|
||||
<p className="pf-ehint">Where unmatched traffic goes while this profile is active.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Default egress</span>
|
||||
<div className="pf-erow-ctl">
|
||||
<select
|
||||
className="pf-select"
|
||||
value={profile.DefaultEgress ?? ''}
|
||||
onChange={(e) =>
|
||||
onPatch(
|
||||
{ DefaultEgress: e.target.value },
|
||||
e.target.value ? `Egress → ${e.target.value}` : 'Egress override cleared',
|
||||
)
|
||||
}
|
||||
disabled={busy}
|
||||
aria-label="Default egress"
|
||||
>
|
||||
<option value="">No override</option>
|
||||
{egresses.map((eg) => (
|
||||
<option key={eg.Name} value={eg.Name}>
|
||||
{eg.Name}
|
||||
</option>
|
||||
))}
|
||||
{profile.DefaultEgress && !egresses.some((eg) => eg.Name === profile.DefaultEgress) && (
|
||||
<option value={profile.DefaultEgress}>{profile.DefaultEgress} (missing)</option>
|
||||
)}
|
||||
</select>
|
||||
<p className="pf-ehint">Pins the outbound interface / egress for this profile.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Endpoint resolver</span>
|
||||
<div className="pf-erow-ctl">
|
||||
@@ -1141,172 +867,8 @@ function ProfileEditor({
|
||||
)
|
||||
}
|
||||
|
||||
// ---- a preset pack row -----------------------------------------------------
|
||||
|
||||
function PackRow({
|
||||
pack,
|
||||
preset,
|
||||
busy,
|
||||
disabled,
|
||||
catalog,
|
||||
valid,
|
||||
onSet,
|
||||
}: {
|
||||
pack: { name: string; title: string; desc: string }
|
||||
preset: Preset | undefined
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
onSet: (patch: Partial<Preset>, msg: string) => void
|
||||
}) {
|
||||
const enabled = preset?.Enabled ?? false
|
||||
const target = preset?.Target ?? ''
|
||||
const [advOpen, setAdvOpen] = useState(false)
|
||||
|
||||
return (
|
||||
<li className={enabled ? 'pf-row pf-pack' : 'pf-row pf-pack off'}>
|
||||
<div className="pf-row-head">
|
||||
<Toggle
|
||||
pressed={enabled}
|
||||
onChange={(on) => onSet({ Enabled: on }, `${pack.name} ${on ? 'enabled' : 'disabled'}`)}
|
||||
label={`${enabled ? 'Disable' : 'Enable'} ${pack.title}`}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<div className="pf-row-main">
|
||||
<div className="pf-row-l1">
|
||||
<span className="pf-row-name">{pack.title}</span>
|
||||
<span className="pf-badge mono">{pack.name}</span>
|
||||
</div>
|
||||
<p className="pf-pack-desc">{pack.desc}</p>
|
||||
</div>
|
||||
<label className="pf-pack-target">
|
||||
<span className="pf-active-ctl-label mono">Target</span>
|
||||
<TargetSelect
|
||||
value={target}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
includeNoOverride
|
||||
noOverrideLabel="Pack default"
|
||||
busy={busy}
|
||||
disabled={disabled}
|
||||
ariaLabel={`${pack.title} target`}
|
||||
onChange={(v) => onSet({ Target: v }, v ? `${pack.name} → ${v}` : `${pack.name} → pack default`)}
|
||||
/>
|
||||
</label>
|
||||
</div>
|
||||
|
||||
<details
|
||||
className="pf-adv"
|
||||
open={advOpen}
|
||||
onToggle={(e) => setAdvOpen((e.target as HTMLDetailsElement).open)}
|
||||
>
|
||||
<summary className="pf-adv-summary">Advanced</summary>
|
||||
<div className="pf-adv-body">
|
||||
<label className="pf-field pf-field--num">
|
||||
<span className="pf-flabel">Order</span>
|
||||
<BlurField
|
||||
value={preset?.Order != null ? String(preset.Order) : ''}
|
||||
onCommit={(v) => {
|
||||
const s = v.trim()
|
||||
if (s === '') {
|
||||
onSet({ Order: undefined }, `${pack.name} order cleared`)
|
||||
} else if (/^\d+$/.test(s)) {
|
||||
onSet({ Order: Number(s) }, `${pack.name} order → ${s}`)
|
||||
}
|
||||
}}
|
||||
placeholder="auto"
|
||||
ariaLabel={`${pack.title} order`}
|
||||
busy={busy}
|
||||
width="6rem"
|
||||
inputMode="numeric"
|
||||
/>
|
||||
</label>
|
||||
<p className="pf-ehint">Lower runs first among packs. Leave blank for the built-in order.</p>
|
||||
</div>
|
||||
</details>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- shared controls -------------------------------------------------------
|
||||
|
||||
/** Canonical target picker (direct/block + groups/chains/nodes/egresses). */
|
||||
function TargetSelect({
|
||||
value,
|
||||
catalog,
|
||||
valid,
|
||||
includeNoOverride,
|
||||
noOverrideLabel,
|
||||
busy,
|
||||
disabled,
|
||||
ariaLabel,
|
||||
onChange,
|
||||
}: {
|
||||
value: string
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
includeNoOverride?: boolean
|
||||
noOverrideLabel?: string
|
||||
busy: boolean
|
||||
disabled?: boolean
|
||||
ariaLabel: string
|
||||
onChange: (v: string) => void
|
||||
}) {
|
||||
const missing = value !== '' && !valid.has(value)
|
||||
return (
|
||||
<select
|
||||
className="pf-select"
|
||||
value={value}
|
||||
onChange={(e) => onChange(e.target.value)}
|
||||
disabled={busy || disabled}
|
||||
aria-label={ariaLabel}
|
||||
>
|
||||
{includeNoOverride && <option value="">{noOverrideLabel ?? 'No override'}</option>}
|
||||
<option value="direct">Direct (no proxy)</option>
|
||||
<option value="block">Block</option>
|
||||
{catalog.groups.length > 0 && (
|
||||
<optgroup label="Groups">
|
||||
{catalog.groups.map((g) => (
|
||||
<option key={g} value={`group:${g}`}>
|
||||
Group {g} (balancer)
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.chains.length > 0 && (
|
||||
<optgroup label="Chains">
|
||||
{catalog.chains.map((c) => (
|
||||
<option key={c} value={`chain:${c}`}>
|
||||
Chain {c}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.nodes.length > 0 && (
|
||||
<optgroup label="Nodes">
|
||||
{catalog.nodes.map((n) => (
|
||||
<option key={n} value={`node:${n}`}>
|
||||
Node {n}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.egresses.length > 0 && (
|
||||
<optgroup label="Interfaces / egresses">
|
||||
{catalog.egresses.map((e) => (
|
||||
<option key={e.name} value={`egress:${e.name}`}>
|
||||
Interface/egress {e.name}
|
||||
{e.type ? ` (${e.type})` : ''}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{missing && <option value={value}>{value} (missing)</option>}
|
||||
</select>
|
||||
)
|
||||
}
|
||||
|
||||
/** A text input that keeps a local draft and commits on blur / Enter, reverts on Escape. */
|
||||
function BlurField({
|
||||
value,
|
||||
@@ -1361,65 +923,69 @@ function BlurField({
|
||||
)
|
||||
}
|
||||
|
||||
/** A chips editor: type + Enter to add, ✕ to remove. */
|
||||
function ChipInput({
|
||||
/** Uplink-iface chips + a dropdown of the router's UCI interfaces to add from.
|
||||
* A stored iface missing from the live list still renders as a chip (marked
|
||||
* stale) so it stays removable; with no interfaces fetched there is simply
|
||||
* nothing to add. */
|
||||
function IfaceChips({
|
||||
chips,
|
||||
placeholder,
|
||||
ariaLabel,
|
||||
interfaces,
|
||||
busy,
|
||||
onAdd,
|
||||
onRemove,
|
||||
}: {
|
||||
chips: string[]
|
||||
placeholder?: string
|
||||
ariaLabel: string
|
||||
interfaces: Interface[]
|
||||
busy: boolean
|
||||
onAdd: (v: string) => void
|
||||
onRemove: (v: string) => void
|
||||
}) {
|
||||
const [draft, setDraft] = useState('')
|
||||
|
||||
const commit = () => {
|
||||
const v = draft.trim()
|
||||
if (!v) return
|
||||
onAdd(v)
|
||||
setDraft('')
|
||||
}
|
||||
|
||||
const known = new Set(interfaces.map((i) => i.name))
|
||||
const available = interfaces.filter((i) => !chips.includes(i.name))
|
||||
return (
|
||||
<div className="pf-chipinput">
|
||||
{chips.map((c) => (
|
||||
<span key={c} className="pf-chip pf-chip--edit">
|
||||
<span className="mono">{c}</span>
|
||||
<button
|
||||
type="button"
|
||||
className="pf-chip-x"
|
||||
aria-label={`Remove ${c}`}
|
||||
onClick={() => onRemove(c)}
|
||||
disabled={busy}
|
||||
{chips.map((c) => {
|
||||
const stale = interfaces.length > 0 && !known.has(c)
|
||||
return (
|
||||
<span
|
||||
key={c}
|
||||
className="pf-chip pf-chip--edit"
|
||||
title={stale ? 'Not among the router’s interfaces' : undefined}
|
||||
>
|
||||
✕
|
||||
</button>
|
||||
</span>
|
||||
))}
|
||||
<input
|
||||
className="pf-input pf-chipinput-in mono"
|
||||
type="text"
|
||||
value={draft}
|
||||
placeholder={placeholder}
|
||||
aria-label={ariaLabel}
|
||||
autoComplete="off"
|
||||
spellCheck={false}
|
||||
disabled={busy}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === 'Enter' || e.key === ',') {
|
||||
e.preventDefault()
|
||||
commit()
|
||||
}
|
||||
}}
|
||||
onBlur={commit}
|
||||
/>
|
||||
<span className="mono">
|
||||
{c}
|
||||
{stale ? ' (stale)' : ''}
|
||||
</span>
|
||||
<button
|
||||
type="button"
|
||||
className="pf-chip-x"
|
||||
aria-label={`Remove ${c}`}
|
||||
onClick={() => onRemove(c)}
|
||||
disabled={busy}
|
||||
>
|
||||
✕
|
||||
</button>
|
||||
</span>
|
||||
)
|
||||
})}
|
||||
{available.length > 0 && (
|
||||
<select
|
||||
className="pf-select"
|
||||
value=""
|
||||
onChange={(e) => {
|
||||
if (e.target.value) onAdd(e.target.value)
|
||||
}}
|
||||
disabled={busy}
|
||||
aria-label="Add uplink interface"
|
||||
>
|
||||
<option value="">Add interface…</option>
|
||||
{available.map((i) => (
|
||||
<option key={i.name} value={i.name}>
|
||||
{i.name} ({i.device || '?'}){i.up ? '' : ' — down'}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -252,6 +252,43 @@
|
||||
color: var(--faint);
|
||||
}
|
||||
|
||||
/* ---- a rule that can never fire (superseded by a later condition-less rule) ----
|
||||
*
|
||||
* Warn semantics only: --amber, never --accent. Orange is the ACTIVE state on this
|
||||
* faceplate, and a rule the router ignores is the opposite of active — painting it
|
||||
* orange is what made two `default` rows look equally live. It is a dashed amber
|
||||
* frame, an amber order chip, and a dimmed target, so the row reads as "wired but
|
||||
* not connected" without shouting: nothing is broken, one setting is just inert. */
|
||||
.rt-rule.dead {
|
||||
border-style: dashed;
|
||||
border-color: color-mix(in srgb, var(--amber) 55%, var(--groove));
|
||||
background: var(--panel);
|
||||
box-shadow: none;
|
||||
}
|
||||
.rt-ord.dead {
|
||||
color: var(--amber);
|
||||
border-color: color-mix(in srgb, var(--amber) 45%, var(--groove));
|
||||
}
|
||||
.rt-badge.dead {
|
||||
padding: 1px 7px;
|
||||
border: 1px solid color-mix(in srgb, var(--amber) 55%, var(--groove));
|
||||
border-radius: 999px;
|
||||
background: color-mix(in srgb, var(--amber) 12%, transparent);
|
||||
color: var(--amber);
|
||||
}
|
||||
.rt-dead-note {
|
||||
font-family: var(--font-sans);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.45;
|
||||
color: var(--dim);
|
||||
}
|
||||
/* The target is still what the operator asked for, so it stays readable — just
|
||||
* quiet, because the router is not using it. */
|
||||
.rt-rule.dead .rt-target {
|
||||
border-style: dashed;
|
||||
opacity: 0.62;
|
||||
}
|
||||
|
||||
/* ---- target chip (styled like the artifact's group:auto mono chips) ---- */
|
||||
.rt-target {
|
||||
display: inline-flex;
|
||||
@@ -359,9 +396,6 @@
|
||||
gap: 5px;
|
||||
min-width: 0;
|
||||
}
|
||||
.rt-field-wide {
|
||||
grid-column: span 2;
|
||||
}
|
||||
.rt-flabel {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 9px;
|
||||
@@ -529,6 +563,7 @@ select.rt-input {
|
||||
color: var(--faint);
|
||||
}
|
||||
|
||||
|
||||
/* textarea shares the input skin but grows vertically for a list of entries */
|
||||
.rt-textarea {
|
||||
min-height: 84px;
|
||||
@@ -800,9 +835,6 @@ select.rt-input {
|
||||
justify-content: flex-start;
|
||||
align-self: start;
|
||||
}
|
||||
.rt-field-wide {
|
||||
grid-column: auto;
|
||||
}
|
||||
.rt-rs-row {
|
||||
grid-template-columns: 1fr;
|
||||
row-gap: 10px;
|
||||
|
||||
+143
-119
@@ -6,11 +6,12 @@ import {
|
||||
apply as apiApply,
|
||||
getConfig,
|
||||
putConfig,
|
||||
getRulesReachability,
|
||||
getRulesetStatus,
|
||||
updateRuleset as apiUpdateRuleset,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Model, Rule, Ruleset, RulesetStatus } from '../api'
|
||||
import type { Model, Rule, RuleReach, Ruleset, RulesetStatus } from '../api'
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// The api.ts `Rule` is a deliberately thin subset (Name/Enabled/Order/Target/
|
||||
@@ -21,9 +22,7 @@ import type { Model, Rule, Ruleset, RulesetStatus } from '../api'
|
||||
// ---------------------------------------------------------------------------
|
||||
type RRule = Rule & {
|
||||
Src?: string[] | null
|
||||
DstDomain?: string[] | null
|
||||
DstRuleset?: string[] | null
|
||||
DstIP?: string[] | null
|
||||
DstPort?: string
|
||||
Proto?: string
|
||||
Kill?: string
|
||||
@@ -96,7 +95,6 @@ function ProtoOptions({ value }: { value: string }) {
|
||||
|
||||
const len = (a: unknown[] | null | undefined): number => (a ? a.length : 0)
|
||||
const byOrder = (a: RRule, b: RRule): number => a.Order - b.Order
|
||||
const csv = (s: string): string[] => s.split(',').map((x) => x.trim()).filter(Boolean)
|
||||
|
||||
// --- ruleset helpers --------------------------------------------------------
|
||||
// A `config ruleset` (api.ts Ruleset) is a named domain/ipcidr list a rule
|
||||
@@ -212,13 +210,13 @@ function everyLabel(sec: number): string {
|
||||
return `every ${sec}s`
|
||||
}
|
||||
|
||||
/** A rule with no matcher of any kind is the effective catch-all (route Final). */
|
||||
/** A rule with no matcher of any kind is the effective catch-all (route Final).
|
||||
* Mirrors model.IsCatchAll on the daemon side — the two must agree or the
|
||||
* "never applies" badge lands on a different row than the apply warning. */
|
||||
function isCatchAll(r: RRule): boolean {
|
||||
return (
|
||||
len(r.Src) === 0 &&
|
||||
len(r.DstDomain) === 0 &&
|
||||
len(r.DstRuleset) === 0 &&
|
||||
len(r.DstIP) === 0 &&
|
||||
!(r.DstPort && r.DstPort.trim()) &&
|
||||
!(r.Proto && r.Proto.trim())
|
||||
)
|
||||
@@ -262,16 +260,15 @@ interface TargetGroups {
|
||||
nodes: TargetOpt[] // node:<n> (huge — rendered last)
|
||||
}
|
||||
|
||||
// Free-text destination matchers offered by the ADD form. Domains are NOT one of
|
||||
// them (the user's call): domain matching goes through named rulesets — that's
|
||||
// what they exist for. 'none' = the rule matches by rulesets/source/proto alone.
|
||||
// (Legacy rules that already carry DstDomain stay editable in the edit form.)
|
||||
type MatchKind = 'none' | 'ip' | 'port'
|
||||
// The add form's fields. WHERE traffic is going is a ruleset choice and nothing
|
||||
// else — a rule has no inline domain or address list any more, so the old
|
||||
// Match-kind picker (rulesets / ip / port) collapsed into a plain Port field
|
||||
// beside the ruleset picker. Adding one domain is still one step: the picker can
|
||||
// build a list on the spot (RulesetPicker's "New list").
|
||||
interface AddForm {
|
||||
name: string
|
||||
src: string[]
|
||||
matchKind: MatchKind
|
||||
matchValue: string
|
||||
port: string
|
||||
rulesets: string[]
|
||||
proto: string
|
||||
target: string
|
||||
@@ -283,8 +280,7 @@ interface AddForm {
|
||||
const EMPTY_FORM: AddForm = {
|
||||
name: '',
|
||||
src: [],
|
||||
matchKind: 'none',
|
||||
matchValue: '',
|
||||
port: '',
|
||||
rulesets: [],
|
||||
proto: '',
|
||||
target: 'direct',
|
||||
@@ -334,6 +330,22 @@ export default function Routing() {
|
||||
toastTimer.current = window.setTimeout(() => setToast(null), 2600)
|
||||
}, [])
|
||||
|
||||
// Which rules can never fire, keyed by their position in the model's Rules array
|
||||
// — NOT by name. The config that made this necessary had two rules both called
|
||||
// `default`, which is exactly when a name-keyed verdict badges the wrong row.
|
||||
const [reach, setReach] = useState<Map<number, RuleReach>>(new Map())
|
||||
const loadReach = useCallback(async () => {
|
||||
try {
|
||||
const { rules } = await getRulesReachability()
|
||||
setReach(new Map(rules.map((r) => [r.index, r])))
|
||||
} catch {
|
||||
// An older daemon has no such endpoint, and a stopped one answers nothing.
|
||||
// Drop the verdicts rather than keep stale ones: no badge is honest, a badge
|
||||
// about the previous config is not.
|
||||
setReach(new Map())
|
||||
}
|
||||
}, [])
|
||||
|
||||
const load = useCallback(async () => {
|
||||
try {
|
||||
setLoadError(null)
|
||||
@@ -341,7 +353,8 @@ export default function Routing() {
|
||||
} catch (e) {
|
||||
setLoadError(errMsg(e))
|
||||
}
|
||||
}, [])
|
||||
void loadReach()
|
||||
}, [loadReach])
|
||||
useEffect(() => {
|
||||
void load()
|
||||
}, [load])
|
||||
@@ -407,6 +420,36 @@ export default function Routing() {
|
||||
// Every rule name, for the edit form's duplicate-name guard (it excludes self).
|
||||
const ruleNames = useMemo(() => new Set(rules.map((r) => r.Name)), [rules])
|
||||
|
||||
// Each rule's position in the model's Rules array — the key the daemon's
|
||||
// reachability verdicts use. `rules` above is a sorted COPY of the same object
|
||||
// references, so identity survives the sort and this map stays valid.
|
||||
const modelIndex = useMemo(() => {
|
||||
const m = new Map<RRule, number>()
|
||||
;((config?.Rules as RRule[] | null | undefined) ?? []).forEach((r, i) => m.set(r, i))
|
||||
return m
|
||||
}, [config])
|
||||
|
||||
/**
|
||||
* The verdict for one rule, or null when it can fire.
|
||||
*
|
||||
* Verdicts are fetched separately from the config, so between an optimistic edit
|
||||
* and the refetch they can describe the PREVIOUS rule list. Re-checking the
|
||||
* echoed name and order is what stops that window from putting a "never applies"
|
||||
* badge on a working rule: a mismatch means the verdict is not about this row,
|
||||
* and no badge is the honest answer.
|
||||
*/
|
||||
const shadowOf = useCallback(
|
||||
(r: RRule): { by: string; byOrder: number; reason: string } | null => {
|
||||
const i = modelIndex.get(r)
|
||||
if (i === undefined) return null
|
||||
const v = reach.get(i)
|
||||
if (!v || !v.unreachable || !v.shadowed_by) return null
|
||||
if (v.name !== r.Name || v.order !== r.Order) return null
|
||||
return { by: v.shadowed_by, byOrder: v.shadowed_by_order ?? 0, reason: v.reason ?? '' }
|
||||
},
|
||||
[modelIndex, reach],
|
||||
)
|
||||
|
||||
// Rulesets are named domain/IP lists rules match against (rule.DstRuleset).
|
||||
const rulesets = useMemo<Ruleset[]>(
|
||||
() => [...((config?.Rulesets as Ruleset[] | null | undefined) ?? [])],
|
||||
@@ -458,6 +501,10 @@ export default function Routing() {
|
||||
await putConfig(next)
|
||||
setSavedPending(true)
|
||||
flash(okMsg)
|
||||
// The verdicts describe the config on disk, which just changed — re-ask.
|
||||
// Adding or moving a rule is precisely what turns a working default into a
|
||||
// superseded one, and vice versa.
|
||||
void loadReach()
|
||||
} catch (e) {
|
||||
setConfig(prev)
|
||||
setActionError(errMsg(e))
|
||||
@@ -466,7 +513,7 @@ export default function Routing() {
|
||||
setSaving(false)
|
||||
}
|
||||
},
|
||||
[config, flash],
|
||||
[config, flash, loadReach],
|
||||
)
|
||||
|
||||
const commitRules = useCallback(
|
||||
@@ -622,18 +669,15 @@ export default function Routing() {
|
||||
return
|
||||
}
|
||||
setFormError(null)
|
||||
const mv = form.matchValue.trim()
|
||||
const rule: RRule = {
|
||||
Name: name,
|
||||
Enabled: true,
|
||||
Order: 0,
|
||||
Src: form.src,
|
||||
// Domains are matched via rulesets only — the add form has no free-text
|
||||
// domain matcher by design.
|
||||
DstDomain: [],
|
||||
// Destination = rulesets, always. Domains and addresses live in a
|
||||
// `config ruleset` so one list serves every rule that needs it.
|
||||
DstRuleset: form.rulesets,
|
||||
DstIP: form.matchKind === 'ip' ? csv(mv) : [],
|
||||
DstPort: form.matchKind === 'port' ? mv : '',
|
||||
DstPort: form.port.trim(),
|
||||
Proto: form.proto,
|
||||
Target: form.target,
|
||||
Egress: '',
|
||||
@@ -737,14 +781,18 @@ export default function Routing() {
|
||||
{rules.length === 0 ? (
|
||||
<div className="rt-empty">
|
||||
<p>No rules — all traffic follows the default route.</p>
|
||||
<p className="rt-empty-sub">Add a rule below to steer a domain, address, or port.</p>
|
||||
<p className="rt-empty-sub">Add a rule below to steer a destination list, source, or port.</p>
|
||||
</div>
|
||||
) : (
|
||||
<ol className="rt-list" aria-label="Routing rules in first-match order">
|
||||
{rules.map((r, i) =>
|
||||
editingRule === r.Name ? (
|
||||
<RuleEditForm
|
||||
key={r.Name}
|
||||
// Rule names are NOT unique in the wild — the config that prompted
|
||||
// the never-applies badge had two rules called `default`, and a
|
||||
// duplicate React key makes the second row shadow the first. The
|
||||
// model index disambiguates without changing row identity.
|
||||
key={`${modelIndex.get(r) ?? i}:${r.Name}`}
|
||||
initial={r}
|
||||
names={ruleNames}
|
||||
targets={targets}
|
||||
@@ -755,12 +803,13 @@ export default function Routing() {
|
||||
/>
|
||||
) : (
|
||||
<RuleRow
|
||||
key={r.Name}
|
||||
key={`${modelIndex.get(r) ?? i}:${r.Name}`}
|
||||
rule={r}
|
||||
index={i}
|
||||
total={rules.length}
|
||||
busy={saving}
|
||||
editingOther={editingRule !== null}
|
||||
shadow={shadowOf(r)}
|
||||
onEdit={onEditRule}
|
||||
onToggle={onToggle}
|
||||
onMove={onMove}
|
||||
@@ -810,6 +859,7 @@ function RuleRow({
|
||||
total,
|
||||
busy,
|
||||
editingOther,
|
||||
shadow,
|
||||
onEdit,
|
||||
onToggle,
|
||||
onMove,
|
||||
@@ -820,6 +870,8 @@ function RuleRow({
|
||||
total: number
|
||||
busy: boolean
|
||||
editingOther: boolean
|
||||
/** Set when the daemon reports this rule can never fire; null when it can. */
|
||||
shadow: { by: string; byOrder: number; reason: string } | null
|
||||
onEdit: (name: string) => void
|
||||
onToggle: (name: string) => void
|
||||
onMove: (name: string, dir: 'up' | 'down') => void
|
||||
@@ -829,13 +881,20 @@ function RuleRow({
|
||||
// the first-match order can't shift under the open form. Edit itself stays live —
|
||||
// clicking it just swaps which row is being edited.
|
||||
const frozen = busy || editingOther
|
||||
const isDefault = isCatchAll(rule)
|
||||
// A rule with no conditions is the router's default — but only ONE of them can
|
||||
// be, and the daemon says which. A superseded one must not wear the default's
|
||||
// marks (the dashed accent frame, the "· final" order chip, the "everything not
|
||||
// matched above" line): those are the claim that made two `default` rules
|
||||
// indistinguishable in the first place.
|
||||
const dead = shadow !== null
|
||||
const isDefault = isCatchAll(rule) && !dead
|
||||
const target = effectiveTarget(rule)
|
||||
const tone = targetTone(target)
|
||||
const cls = [
|
||||
'rt-rule',
|
||||
rule.Enabled ? '' : 'off',
|
||||
isDefault ? 'final' : '',
|
||||
dead ? 'dead' : '',
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join(' ')
|
||||
@@ -852,7 +911,7 @@ function RuleRow({
|
||||
>
|
||||
▲
|
||||
</button>
|
||||
<span className={isDefault ? 'rt-ord final' : 'rt-ord'}>
|
||||
<span className={isDefault ? 'rt-ord final' : dead ? 'rt-ord dead' : 'rt-ord'}>
|
||||
{isDefault ? '·' : rule.Order}
|
||||
</span>
|
||||
<button
|
||||
@@ -870,9 +929,19 @@ function RuleRow({
|
||||
<div className="rt-head">
|
||||
<span className="rt-name">{rule.Name}</span>
|
||||
{isDefault && <span className="rt-badge">default route · final</span>}
|
||||
{dead && <span className="rt-badge dead">never applies</span>}
|
||||
</div>
|
||||
<div className="rt-match">
|
||||
{isDefault ? (
|
||||
{dead ? (
|
||||
// The badge says it never fires; this line says what beat it and what to
|
||||
// do. Visible text, not a tooltip — the operator has to be able to find
|
||||
// the other rule, and two rows can carry the same name.
|
||||
<span className="rt-dead-note" title={shadow.reason}>
|
||||
“{shadow.by}” (order {shadow.byOrder}) has no conditions either and runs after this
|
||||
one, so it is the default the router uses. Give this rule a condition, or delete one
|
||||
of the two.
|
||||
</span>
|
||||
) : isDefault ? (
|
||||
<span className="rt-nomatch">everything not matched above</span>
|
||||
) : (
|
||||
<Matchers rule={rule} />
|
||||
@@ -935,9 +1004,7 @@ function Matchers({ rule }: { rule: RRule }): ReactNode {
|
||||
)
|
||||
}
|
||||
listChip('src', rule.Src, 'src')
|
||||
listChip('dns', rule.DstDomain, 'dom')
|
||||
listChip('ruleset', rule.DstRuleset, 'rs')
|
||||
listChip('ip', rule.DstIP, 'ip')
|
||||
if (rule.DstPort && rule.DstPort.trim()) {
|
||||
chips.push(
|
||||
<span className="rt-chip" key="port">
|
||||
@@ -1053,7 +1120,16 @@ function TargetOptions({ targets, current }: { targets: TargetGroups; current?:
|
||||
)
|
||||
}
|
||||
|
||||
/** The dst_ruleset checkbox group. Renders nothing when no rulesets exist. */
|
||||
/**
|
||||
* The destination picker: which rulesets this rule matches (dst_ruleset).
|
||||
*
|
||||
* Checkboxes and nothing else. This is the ONLY way a rule names a destination,
|
||||
* so it renders even when the config has no lists yet — an empty picker that says
|
||||
* where lists come from is the honest answer, and hiding it would leave the rule
|
||||
* form with no destination control at all. Building and filling a list is the
|
||||
* Rulesets panel's job, deliberately kept out of the rule editor so a list is
|
||||
* created in exactly one place.
|
||||
*/
|
||||
function RulesetPicker({
|
||||
options,
|
||||
selected,
|
||||
@@ -1065,24 +1141,26 @@ function RulesetPicker({
|
||||
busy: boolean
|
||||
onToggle: (name: string) => void
|
||||
}): ReactNode {
|
||||
if (options.length === 0) return null
|
||||
return (
|
||||
<div className="rt-rsel">
|
||||
<span className="rt-flabel">Match rulesets — dst_ruleset</span>
|
||||
<div className="rt-rsel-opts" role="group" aria-label="Match these rulesets">
|
||||
{options.map((n) => {
|
||||
const on = selected.includes(n)
|
||||
return (
|
||||
<label key={n} className={on ? 'rt-rsel-opt on' : 'rt-rsel-opt'}>
|
||||
<input type="checkbox" checked={on} onChange={() => onToggle(n)} disabled={busy} />
|
||||
<span className="mono">{n}</span>
|
||||
</label>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
<span className="rt-flabel">Destination — dst_ruleset</span>
|
||||
{options.length > 0 && (
|
||||
<div className="rt-rsel-opts" role="group" aria-label="Match these rulesets">
|
||||
{options.map((n) => {
|
||||
const on = selected.includes(n)
|
||||
return (
|
||||
<label key={n} className={on ? 'rt-rsel-opt on' : 'rt-rsel-opt'}>
|
||||
<input type="checkbox" checked={on} onChange={() => onToggle(n)} disabled={busy} />
|
||||
<span className="mono">{n}</span>
|
||||
</label>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
)}
|
||||
<p className="rt-rsel-hint">
|
||||
The rule also matches any traffic in the checked list(s). Combine with a domain, address, or
|
||||
port, or use a ruleset on its own.
|
||||
{options.length === 0
|
||||
? 'No rulesets yet. Add one under Rulesets below, then come back and check it here — a rule matches a destination through a ruleset only.'
|
||||
: 'The rule matches traffic in ANY checked list. Narrow it further with a source, port or protocol.'}
|
||||
</p>
|
||||
</div>
|
||||
)
|
||||
@@ -1208,7 +1286,6 @@ function AddRule({
|
||||
set('rulesets', form.rulesets.includes(n) ? form.rulesets.filter((x) => x !== n) : [...form.rulesets, n])
|
||||
const toggleDay = (d: string) =>
|
||||
set('schedDays', form.schedDays.includes(d) ? form.schedDays.filter((x) => x !== d) : [...form.schedDays, d])
|
||||
const matchPlaceholder = form.matchKind === 'ip' ? '10.0.0.0/8, 100.64.0.0/10' : '443, 8080-8090'
|
||||
|
||||
return (
|
||||
<form className="rt-add" onSubmit={onSubmit} aria-label="Add a routing rule">
|
||||
@@ -1241,32 +1318,17 @@ function AddRule({
|
||||
</label>
|
||||
|
||||
<label className="rt-field">
|
||||
<span className="rt-flabel">Match</span>
|
||||
<select
|
||||
<span className="rt-flabel">Port(s)</span>
|
||||
<input
|
||||
className="rt-input mono"
|
||||
value={form.matchKind}
|
||||
onChange={(e) => set('matchKind', e.target.value as MatchKind)}
|
||||
>
|
||||
<option value="none">rulesets only</option>
|
||||
<option value="ip">ip / cidr</option>
|
||||
<option value="port">port</option>
|
||||
</select>
|
||||
value={form.port}
|
||||
onChange={(e) => set('port', e.target.value)}
|
||||
placeholder="443, 8080-8090"
|
||||
autoComplete="off"
|
||||
spellCheck={false}
|
||||
/>
|
||||
</label>
|
||||
|
||||
{form.matchKind !== 'none' && (
|
||||
<label className="rt-field rt-field-wide">
|
||||
<span className="rt-flabel">{form.matchKind === 'port' ? 'Port(s)' : 'Address(es)'}</span>
|
||||
<input
|
||||
className="rt-input mono"
|
||||
value={form.matchValue}
|
||||
onChange={(e) => set('matchValue', e.target.value)}
|
||||
placeholder={matchPlaceholder}
|
||||
autoComplete="off"
|
||||
spellCheck={false}
|
||||
/>
|
||||
</label>
|
||||
)}
|
||||
|
||||
<label className="rt-field">
|
||||
<span className="rt-flabel">Proto</span>
|
||||
<select
|
||||
@@ -1324,11 +1386,11 @@ function AddRule({
|
||||
}
|
||||
|
||||
// --- edit-a-rule plate (inline, replaces the row it edits) ------------------
|
||||
// Unlike AddRule, editing exposes all three destination matchers at once
|
||||
// (Domain(s) / IP-CIDR(s) / Port) rather than a single Match picker — a real rule
|
||||
// can carry several matcher kinds simultaneously and none may be silently dropped.
|
||||
// The full original rule is spread into the result on save, so Order / Enabled /
|
||||
// Kill / Egress (and anything else off-form) survive untouched.
|
||||
// Same fields as AddRule, on purpose: a rule carries exactly one destination
|
||||
// mechanism (rulesets) plus port/proto/source, so there is nothing an edit can
|
||||
// reveal that the add form hides. The full original rule is spread into the
|
||||
// result on save, so Order / Enabled / Kill / Egress (and anything else off-form)
|
||||
// survive untouched.
|
||||
function RuleEditForm({
|
||||
initial,
|
||||
names,
|
||||
@@ -1348,8 +1410,6 @@ function RuleEditForm({
|
||||
}) {
|
||||
const [name, setName] = useState(initial.Name)
|
||||
const [src, setSrc] = useState<string[]>([...(initial.Src ?? [])])
|
||||
const [domain, setDomain] = useState((initial.DstDomain ?? []).join(', '))
|
||||
const [ip, setIp] = useState((initial.DstIP ?? []).join(', '))
|
||||
const [port, setPort] = useState(initial.DstPort ?? '')
|
||||
const [proto, setProto] = useState(initial.Proto ?? '')
|
||||
const [target, setTarget] = useState(effectiveTarget(initial))
|
||||
@@ -1367,12 +1427,7 @@ function RuleEditForm({
|
||||
|
||||
// A rule with no matcher of any kind is a catch-all — legal, but worth flagging.
|
||||
const noMatchers =
|
||||
src.length === 0 &&
|
||||
csv(domain).length === 0 &&
|
||||
csv(ip).length === 0 &&
|
||||
port.trim() === '' &&
|
||||
rulesets.length === 0 &&
|
||||
proto.trim() === ''
|
||||
src.length === 0 && port.trim() === '' && rulesets.length === 0 && proto.trim() === ''
|
||||
|
||||
const submit = (e: FormEvent) => {
|
||||
e.preventDefault()
|
||||
@@ -1394,8 +1449,6 @@ function RuleEditForm({
|
||||
...initial,
|
||||
Name: nm,
|
||||
Src: src,
|
||||
DstDomain: csv(domain),
|
||||
DstIP: csv(ip),
|
||||
DstPort: port.trim(),
|
||||
DstRuleset: rulesets,
|
||||
Proto: proto,
|
||||
@@ -1414,7 +1467,9 @@ function RuleEditForm({
|
||||
<form className="rt-add rt-edit-form" onSubmit={submit} aria-label={`Edit rule ${initial.Name}`}>
|
||||
<div className="rt-add-hd">
|
||||
<span className="rt-add-title">Edit {initial.Name}</span>
|
||||
<span className="rt-add-sub">Empty a field to drop that matcher. Save, then Apply.</span>
|
||||
<span className="rt-add-sub">
|
||||
Uncheck a list or clear a field to drop that matcher. Save, then Apply.
|
||||
</span>
|
||||
</div>
|
||||
|
||||
<div className="rt-fields">
|
||||
@@ -1444,37 +1499,6 @@ function RuleEditForm({
|
||||
/>
|
||||
</label>
|
||||
|
||||
{/* Domains are matched via rulesets by design — this legacy field only
|
||||
appears when the rule ALREADY carries free-text domains, so they
|
||||
stay visible and clearable rather than silently preserved. */}
|
||||
{(initial.DstDomain ?? []).length > 0 && (
|
||||
<label className="rt-field rt-field-wide">
|
||||
<span className="rt-flabel">Domain(s) — legacy</span>
|
||||
<input
|
||||
className="rt-input mono"
|
||||
value={domain}
|
||||
onChange={(e) => setDomain(e.target.value)}
|
||||
placeholder="youtube.com, *.googlevideo.com"
|
||||
autoComplete="off"
|
||||
spellCheck={false}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
)}
|
||||
|
||||
<label className="rt-field rt-field-wide">
|
||||
<span className="rt-flabel">IP / CIDR(s)</span>
|
||||
<input
|
||||
className="rt-input mono"
|
||||
value={ip}
|
||||
onChange={(e) => setIp(e.target.value)}
|
||||
placeholder="10.0.0.0/8, 100.64.0.0/10"
|
||||
autoComplete="off"
|
||||
spellCheck={false}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
|
||||
<label className="rt-field">
|
||||
<span className="rt-flabel">Port(s)</span>
|
||||
<input
|
||||
|
||||
@@ -103,64 +103,6 @@ function parseDuration(raw: string): ParseResult<string> {
|
||||
return { ok: true, value: s.toLowerCase() }
|
||||
}
|
||||
|
||||
// ---- health sweep interval -------------------------------------------------
|
||||
//
|
||||
// The daemon's own vocabulary (model.SweepDisabledValues / SweepIntervalMin /
|
||||
// SweepSchedule), mirrored here so the input accepts exactly what the router
|
||||
// accepts and nothing else.
|
||||
|
||||
/** The four spellings that switch the background sweep OFF. */
|
||||
const SWEEP_OFF = ['0', 'off', 'none', 'disabled']
|
||||
/** Anything shorter is raised to this by the daemon, so the field raises it here. */
|
||||
const SWEEP_MIN_S = 5
|
||||
const UNIT_S: Record<string, number> = { ms: 0.001, s: 1, m: 60, h: 3600 }
|
||||
|
||||
/** Seconds in a duration this field already validated, or null. */
|
||||
function durationSeconds(s: string): number | null {
|
||||
const m = /^(\d+(?:\.\d+)?)(ms|s|m|h)$/.exec(s)
|
||||
return m ? Number(m[1]) * UNIT_S[m[2]] : null
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the sweep tick.
|
||||
*
|
||||
* The rule worth stating: an unrecognised value does NOT mean "off". The daemon
|
||||
* warns about it and quietly falls back to its default, so a value typed as a way
|
||||
* to stop the sweep would leave the sweep running. The field refuses it here
|
||||
* instead, where the person can still see what they typed.
|
||||
*
|
||||
* '' → the engine default (every 10 s)
|
||||
* 0 / off / none / disabled → stopped
|
||||
* 30, 30s, 5m, 1h → that tick, raised to the 5 s floor
|
||||
*/
|
||||
function parseSweepInterval(raw: string): ParseResult<string> {
|
||||
const s = raw.trim().toLowerCase()
|
||||
if (s === '') return { ok: true, value: '' }
|
||||
if (SWEEP_OFF.includes(s)) return { ok: true, value: s }
|
||||
// A bare integer is seconds, as everywhere else in the model — normalised so the
|
||||
// stored value says which unit it meant.
|
||||
const normalised = /^\d+$/.test(s) ? `${s}s` : s
|
||||
const seconds = DURATION_RE.test(normalised) ? durationSeconds(normalised) : null
|
||||
if (seconds === null) {
|
||||
return {
|
||||
ok: false,
|
||||
error: 'Use a duration like 30s, 5m, or 1h — or “off” to stop the sweep.',
|
||||
}
|
||||
}
|
||||
// The floor, applied where it is visible. Left to the daemon it would be a
|
||||
// warning in a log nobody reads while the field went on showing 1s.
|
||||
if (seconds < SWEEP_MIN_S) return { ok: true, value: `${SWEEP_MIN_S}s` }
|
||||
return { ok: true, value: normalised }
|
||||
}
|
||||
|
||||
/** What a saved sweep value means, as the toast that confirms it. */
|
||||
function sweepSavedMsg(v: string): string {
|
||||
if (v === '') return 'Health sweep → the default, every 10 s'
|
||||
if (SWEEP_OFF.includes(v)) return 'Health sweep off — most nodes will read “not measured”'
|
||||
if (v === `${SWEEP_MIN_S}s`) return `Health sweep → ${v} (the shortest allowed)`
|
||||
return `Health sweep → every ${v}`
|
||||
}
|
||||
|
||||
// `none` really does silence the engine log — it is emitted as the engine's own
|
||||
// log-disable switch, not as a quieter level.
|
||||
const LOG_LEVELS: ReadonlyArray<{ value: string; label: string }> = [
|
||||
@@ -310,9 +252,8 @@ export default function Settings() {
|
||||
[flash],
|
||||
)
|
||||
|
||||
// Our group-member health sweep + all the health/testing stats on the Targets
|
||||
// page. Absent ⇒ enabled (older config), so read it as `!== false`. When off, the
|
||||
// sweep-interval field below is meaningless, so it is dimmed (value kept).
|
||||
// Our background group-member health probing + all the health/testing stats on
|
||||
// the Targets page. Absent ⇒ enabled (older config), so read it as `!== false`.
|
||||
const groupHealthOn = globals?.GroupHealth !== false
|
||||
|
||||
const killSwitch = globals?.KillSwitch === 'open' ? 'open' : 'closed'
|
||||
@@ -588,7 +529,7 @@ export default function Settings() {
|
||||
</p>
|
||||
<Field
|
||||
label="Group health checks"
|
||||
note="Runs a background sweep that checks each group’s member nodes and powers the alive / tested health stats on the Targets page. Turn it off to stop that sweep and hide those stats — your groups still keep picking a live node on their own (sing-box probes them under the hood). The probe URL and interval below still feed those built-in checks."
|
||||
note="Lets the daemon probe the groups and chains your routing rules actually use, in the background, and powers the alive / tested health stats on the Targets page. Anything no enabled rule routes through is skipped and simply reads as unused. Turn it off to stop that background probing and hide those stats — your groups still keep picking a live node on their own (sing-box probes them under the hood, using the probe URL and interval below)."
|
||||
>
|
||||
<Toggle
|
||||
pressed={groupHealthOn}
|
||||
@@ -596,7 +537,7 @@ export default function Settings() {
|
||||
setGlobal(
|
||||
'GroupHealth',
|
||||
on,
|
||||
on ? 'Group health checks on' : 'Group health checks off — sweep stopped',
|
||||
on ? 'Group health checks on' : 'Group health checks off — background probing stopped',
|
||||
)
|
||||
}
|
||||
label={groupHealthOn ? 'Turn off group health checks' : 'Turn on group health checks'}
|
||||
@@ -631,26 +572,6 @@ export default function Settings() {
|
||||
onCommit={(v) => setGlobal('ProbeInterval', v, `Probe interval → ${v}`)}
|
||||
/>
|
||||
</Field>
|
||||
<Field
|
||||
label="Health sweep interval"
|
||||
note={
|
||||
groupHealthOn
|
||||
? 'How often the router probes nodes in the background so the health numbers on the Targets page stay fresh. Blank = every 10 s (recommended). Each tick measures only the nodes nobody else is checking, a few at a time, so a full pass takes roughly 5–10 minutes. Set “off” to stop it — at the cost of most nodes reading “not measured”.'
|
||||
: 'Group health checks are off, so the sweep isn’t running and this interval has no effect. Turn group health checks back on to use it. (Your saved value is kept.)'
|
||||
}
|
||||
>
|
||||
<InlineEdit<string>
|
||||
value={globals?.SweepInterval ?? ''}
|
||||
format={(s) => s}
|
||||
parse={parseSweepInterval}
|
||||
width="8rem"
|
||||
placeholder="10s"
|
||||
ariaLabel="Health sweep interval"
|
||||
busy={busy}
|
||||
disabled={!ready || !groupHealthOn}
|
||||
onCommit={(v) => setGlobal('SweepInterval', v, sweepSavedMsg(v))}
|
||||
/>
|
||||
</Field>
|
||||
</Group>
|
||||
|
||||
{/* ---- STATISTICS & LOGGING ---- */}
|
||||
|
||||
+24
-13
@@ -627,13 +627,12 @@
|
||||
padding: 6px 14px;
|
||||
font-size: 11px;
|
||||
}
|
||||
/* ---- run indicators (header) ----
|
||||
* Two background runs report here, and they are different actions with different
|
||||
* reach — a health refresh covers every group at once, an exit test covers only
|
||||
* the groups it names. Each says WHAT it is and HOW FAR it has got, so neither can
|
||||
* be mistaken for the other, and neither is ever repeated as a badge on the cards.
|
||||
* The third is the sweep: not a run someone started, so no LED and no pulse — a
|
||||
* quiet gauge that explains why untested members fill in by themselves. */
|
||||
/* ---- run indicator (header) ----
|
||||
* The one manual run left — the exit test — reports here. It says WHAT it is
|
||||
* and HOW FAR it has got, and it is never repeated as a badge on cards outside
|
||||
* its scope. The observatory's background probing deliberately has no gauge:
|
||||
* it is not a run someone started, and its whole visible effect is that the
|
||||
* numbers below stay fresh on their own. */
|
||||
.tg-run {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
@@ -655,16 +654,11 @@
|
||||
font-size: 10.5px;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* An in-flight run is amber-edged; the sweep is not a run and stays neutral. */
|
||||
.tg-run--health,
|
||||
/* An in-flight run is amber-edged. */
|
||||
.tg-run--exit {
|
||||
border-color: color-mix(in srgb, var(--amber) 45%, var(--groove));
|
||||
color: var(--ink);
|
||||
}
|
||||
.tg-run--sweep {
|
||||
cursor: help;
|
||||
color: var(--faint);
|
||||
}
|
||||
.tg-test-err {
|
||||
margin: 10px 2px 0;
|
||||
font-family: var(--font-sans);
|
||||
@@ -853,6 +847,23 @@
|
||||
font-size: 12px;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* The unused note (GroupHealth.used === false). A routing fact, not a fault: no
|
||||
enabled rule reaches the group, so the observatory never probes it and there
|
||||
are no counters to show. Muted groove-badge styling on purpose — dim and
|
||||
bordered, never amber or red — because "nothing routes here" is not an
|
||||
alarm. */
|
||||
.gh-unused {
|
||||
display: inline-block;
|
||||
padding: 1px 7px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 999px;
|
||||
background: color-mix(in srgb, var(--sink) 45%, transparent);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.08em;
|
||||
text-transform: uppercase;
|
||||
color: var(--dim);
|
||||
cursor: help;
|
||||
}
|
||||
.gh-bound {
|
||||
padding: 1px 7px;
|
||||
border: 1px dashed color-mix(in srgb, var(--dim) 45%, var(--groove));
|
||||
|
||||
+158
-250
@@ -10,7 +10,6 @@ import {
|
||||
getInterfaces,
|
||||
getStatus,
|
||||
postGroupsTest,
|
||||
postNodesTest,
|
||||
putConfig,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
@@ -22,7 +21,6 @@ import type {
|
||||
GroupsHealth,
|
||||
GroupTestResult,
|
||||
GroupTestStatus,
|
||||
HealthSweep,
|
||||
Chain,
|
||||
Egress,
|
||||
Interface,
|
||||
@@ -77,7 +75,8 @@ const without = (set: Set<string>, self: string): Set<string> =>
|
||||
// spellings: a PREFIXED target-or-detour string ("group:auto", "egress:wan") and,
|
||||
// for egresses only, a BARE binding ("Egress":"wan" on a node/rule/profile).
|
||||
// Renaming or deleting one without touching those references silently orphans
|
||||
// them — the daemon then falls back to the default route with no warning. These
|
||||
// them — the daemon then warns and BLOCKS the traffic of every rule left pointing
|
||||
// at nothing (a rule's traffic never falls through to the default route). These
|
||||
// helpers find and rewrite every site, so a rename carries and a delete can say
|
||||
// exactly what it will break.
|
||||
|
||||
@@ -123,10 +122,6 @@ function findReferences(m: Model, kind: RefKind, name: string): RefSite[] {
|
||||
for (const s of asArray(m.Subscriptions)) {
|
||||
if (isPrefixed(s.FetchDetour, kind, name)) out.push({ label: `subscription “${s.Name}” fetch` })
|
||||
}
|
||||
for (const p of asArray(m.Profiles)) {
|
||||
if (isPrefixed(p.DefaultTarget, kind, name)) out.push({ label: `profile “${p.Name}” target` })
|
||||
if (bare && isBare(p.DefaultEgress, name)) out.push({ label: `profile “${p.Name}” egress` })
|
||||
}
|
||||
if (bare) {
|
||||
for (const n of asArray(m.Nodes)) {
|
||||
if (isBare(n.Egress, name)) out.push({ label: `node “${n.Name}” egress` })
|
||||
@@ -160,12 +155,6 @@ function renameReferences(m: Model, kind: RefKind, from: string, to: string): Mo
|
||||
if (m.Alerts) next.Alerts = m.Alerts.map((a) => ({ ...a, Via: pfx(a.Via) }))
|
||||
if (m.Subscriptions)
|
||||
next.Subscriptions = m.Subscriptions.map((s) => ({ ...s, FetchDetour: pfx(s.FetchDetour) }))
|
||||
if (m.Profiles)
|
||||
next.Profiles = m.Profiles.map((p) => ({
|
||||
...p,
|
||||
DefaultTarget: pfx(p.DefaultTarget),
|
||||
DefaultEgress: br(p.DefaultEgress),
|
||||
}))
|
||||
if (bare && m.Nodes) next.Nodes = m.Nodes.map((n) => ({ ...n, Egress: br(n.Egress) ?? n.Egress }))
|
||||
if (bare && m.Groups) next.Groups = m.Groups.map((g) => ({ ...g, Egress: br(g.Egress) }))
|
||||
return next
|
||||
@@ -182,8 +171,8 @@ function refWarning(refs: RefSite[]): string {
|
||||
const more = refs.length - shown.length
|
||||
const list = `${shown.join(', ')}${more > 0 ? `, and ${more} more` : ''}`
|
||||
return refs.length === 1
|
||||
? ` It is referenced by ${list}, which falls back to the default route.`
|
||||
: ` It is referenced by ${refs.length} places — ${list} — which fall back to the default route.`
|
||||
? ` It is referenced by ${list}, whose traffic will be blocked (an unresolved target never falls through to the default route).`
|
||||
: ` It is referenced by ${refs.length} places — ${list} — whose traffic will be blocked (an unresolved target never falls through to the default route).`
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -306,12 +295,12 @@ const DPI_TYPES = new Set(['interface', 'direct'])
|
||||
// The rendering contract, from the daemon (see engine/grouphealth.go):
|
||||
// - "alive out of TESTED", with the untested remainder as a quiet aside shown
|
||||
// ONLY when it is non-zero.
|
||||
// - untested is never folded into dead. Groups probe lazily, so a freshly
|
||||
// booted router with a 376-node subscription is legitimately almost all
|
||||
// untested; rendering that as dead raises an alarm at the exact moment
|
||||
// - untested is never folded into dead. The observatory needs a few seconds
|
||||
// after an engine (re)start to reach a used group, and it never probes an
|
||||
// unused one; rendering either as dead raises an alarm at the exact moment
|
||||
// nothing is wrong, which teaches the operator to distrust every reading.
|
||||
|
||||
// ---- the sample is not always symmetric -------------------------------------
|
||||
// ---- where dead verdicts come from -------------------------------------------
|
||||
//
|
||||
// A group's own history is written by TWO parties, and only one of them can
|
||||
// record a failure:
|
||||
@@ -320,38 +309,38 @@ const DPI_TYPES = new Set(['interface', 'direct'])
|
||||
// failure DELETES the entry (protocol/group/urltest.go:512). It is physically
|
||||
// incapable of recording "dead" — its failures are indistinguishable from
|
||||
// "never probed";
|
||||
// - the background sweep (our overlay) is the only writer that puts a
|
||||
// trustworthy "dead" on record.
|
||||
// - the daemon's health board is the only writer that puts a trustworthy
|
||||
// "dead" on record: the observatory probes every group and chain an enabled
|
||||
// rule can reach (~10 s tick) and records BOTH outcomes, and the group
|
||||
// checkers feed it too.
|
||||
//
|
||||
// So until the sweep has been over a group, its history holds ONLY successes.
|
||||
// "11 / 11 alive" then does not mean the group is healthy; it means the failures
|
||||
// erased themselves. The real reading, minutes later, was 111 alive / 74 dead.
|
||||
//
|
||||
// That is an OPTIMISTIC lie, which is the worst kind here: it invites someone to
|
||||
// route traffic through a group where 40% of the members are down. So while the
|
||||
// sample is knowably one-sided the panel does not present it as a ratio at all —
|
||||
// a ratio promises a denominator that was checked in both directions, and this
|
||||
// one wasn't.
|
||||
// A used group's untested members therefore resolve to a real verdict within
|
||||
// seconds of the engine coming up. Until then the history holds ONLY successes,
|
||||
// and "11 / 11 alive" does not mean the group is healthy — the failures erased
|
||||
// themselves. That is an OPTIMISTIC lie, the worst kind here: it invites someone
|
||||
// to route traffic through a group where 40% of the members are down. So while
|
||||
// the sample is knowably one-sided the panel does not present it as a ratio at
|
||||
// all — a ratio promises a denominator that was checked in both directions, and
|
||||
// this one wasn't. The window is seconds now, but honesty in it costs nothing.
|
||||
|
||||
/** What a group's numbers add up to, in the six states they actually have. */
|
||||
type Verdict = 'good' | 'partial' | 'degraded' | 'down' | 'unmeasured' | 'empty'
|
||||
|
||||
/**
|
||||
* Is this group's sample knowably one-sided — only successes on record, with
|
||||
* members still unaccounted for and nothing yet able to have recorded a failure?
|
||||
* members still unaccounted for and no failure yet on the board?
|
||||
*
|
||||
* All three conditions matter:
|
||||
* Both conditions matter:
|
||||
* - `dead === 0` — the moment ONE failure is on record, something has been
|
||||
* writing both outcomes and the ratio is honest;
|
||||
* - `untested > 0` — a fully measured group has no room for hidden failures;
|
||||
* - `!sweptOnce` — once the sweep has completed a pass it has had its say
|
||||
* about every member, so silence now means something.
|
||||
* - `untested > 0` — a fully measured group has no room for hidden failures.
|
||||
*
|
||||
* It self-clears on the first failure or the first completed pass, whichever
|
||||
* comes first — no timers, no flags, nothing to get stuck.
|
||||
* It self-clears on the first failure or the first full coverage — and with the
|
||||
* observatory over every used group, that is seconds away. An UNUSED group never
|
||||
* reaches this code: its card says "unused" instead of rendering counters.
|
||||
*/
|
||||
function isBiasedSample(h: GroupHealth, sweptOnce: boolean): boolean {
|
||||
return h.dead === 0 && h.untested > 0 && !sweptOnce
|
||||
function isBiasedSample(h: GroupHealth): boolean {
|
||||
return h.dead === 0 && h.untested > 0
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -362,12 +351,12 @@ function isBiasedSample(h: GroupHealth, sweptOnce: boolean): boolean {
|
||||
* no basis for a proportion. Both decline to make a health claim rather than
|
||||
* make a flattering one.
|
||||
*/
|
||||
function verdictOf(h: GroupHealth, sweptOnce: boolean): Verdict {
|
||||
function verdictOf(h: GroupHealth): Verdict {
|
||||
if (h.total === 0) return 'empty'
|
||||
if (h.tested === 0) return 'unmeasured'
|
||||
if (h.alive === 0) return 'down'
|
||||
if (h.dead > 0) return 'degraded'
|
||||
if (isBiasedSample(h, sweptOnce)) return 'partial'
|
||||
if (isBiasedSample(h)) return 'partial'
|
||||
return 'good'
|
||||
}
|
||||
|
||||
@@ -428,12 +417,7 @@ function sortMembers(members: GroupMemberHealth[]): GroupMemberHealth[] {
|
||||
}
|
||||
|
||||
/** Nothing read yet — an empty, honest starting state (groups is never null). */
|
||||
const IDLE_HEALTH: GroupsHealth = {
|
||||
groups: [],
|
||||
node_test_running: false,
|
||||
node_test_done: 0,
|
||||
node_test_total: 0,
|
||||
}
|
||||
const IDLE_HEALTH: GroupsHealth = { groups: [] }
|
||||
|
||||
/** No test has run (or the page hasn't read one yet). */
|
||||
const IDLE_TEST: GroupTestStatus = { running: false, done: 0, total: 0, scope: [], results: [] }
|
||||
@@ -452,14 +436,14 @@ const normalizeTest = (st: GroupTestStatus): GroupTestStatus => ({
|
||||
})
|
||||
|
||||
/**
|
||||
* How the header names the reach of a running group test. The name matters more
|
||||
* How the header names the reach of a running exit test. The name matters more
|
||||
* than the number when there is only one: "auto" tells the operator which button
|
||||
* they pressed; "1 group" tells them nothing they didn't already know.
|
||||
* they pressed; "1 target" tells them nothing they didn't already know.
|
||||
*/
|
||||
function scopeLabel(scope: string[], groupCount: number): string {
|
||||
function scopeLabel(scope: string[], targetCount: number): string {
|
||||
if (scope.length === 1) return scope[0]
|
||||
if (scope.length === 0) return 'group exits' // pre-scope daemon — say nothing false
|
||||
return scope.length >= groupCount ? 'every group' : `${scope.length} groups`
|
||||
if (scope.length === 0) return 'exits' // pre-scope daemon — say nothing false
|
||||
return scope.length >= targetCount ? 'every exit' : `${scope.length} exits`
|
||||
}
|
||||
|
||||
/** Which editor (add or edit-by-name) is open within a section. */
|
||||
@@ -542,10 +526,11 @@ export default function Targets() {
|
||||
toastTimer.current = window.setTimeout(() => setToast(null), 2600)
|
||||
}, [])
|
||||
|
||||
// Our group-member health sweep + every health/testing control on this page. A
|
||||
// The daemon's observatory + every health/testing control on this page. A
|
||||
// group still picks a live node without it (sing-box probes internally); this
|
||||
// switch (Settings → Health check) governs only OUR sweep and the stats it
|
||||
// feeds. Absent ⇒ on. When off we stop polling and hide all of the health UI.
|
||||
// switch (Settings → Health check) gates only the observatory's background
|
||||
// probing of used groups/chains and the stats it feeds. Absent ⇒ on. When off
|
||||
// we stop polling and hide all of the health UI.
|
||||
const groupHealthOn = config?.Globals?.GroupHealth !== false
|
||||
|
||||
// ---- per-group membership health -------------------------------------------
|
||||
@@ -555,7 +540,6 @@ export default function Targets() {
|
||||
// opened — see GroupMembers.
|
||||
const [health, setHealth] = useState<GroupsHealth>(IDLE_HEALTH)
|
||||
const [healthErr, setHealthErr] = useState<string | null>(null)
|
||||
const [measuring, setMeasuring] = useState(false)
|
||||
// Has the endpoint ever answered? Until it has, a group with no health entry
|
||||
// means "we don't know", not "the engine hasn't built it" — and a row must not
|
||||
// blame the operator's config for our failed read.
|
||||
@@ -596,32 +580,19 @@ export default function Targets() {
|
||||
[health],
|
||||
)
|
||||
|
||||
/**
|
||||
* Measure every group member now (POST /api/nodes/test). This is the only thing
|
||||
* that can turn an untested member into a real dead verdict, and it covers each
|
||||
* group's own egress copies — not just the base outbounds — which is why the
|
||||
* control lives here beside the numbers it moves rather than on the Nodes page.
|
||||
*/
|
||||
const measureAll = useCallback(async () => {
|
||||
setMeasuring(true)
|
||||
try {
|
||||
const r = await postNodesTest()
|
||||
flash(r.started ? 'Measuring every group member…' : 'A measurement is already running')
|
||||
const h = await getGroupsHealth()
|
||||
setHealth({ ...h, groups: asArray(h.groups) })
|
||||
setHealthErr(null)
|
||||
setHealthRead(true)
|
||||
} catch (e) {
|
||||
flash(`Couldn’t start the measurement — ${errText(e)}`)
|
||||
} finally {
|
||||
setMeasuring(false)
|
||||
}
|
||||
}, [flash])
|
||||
// Per-chain reachability (used/unused), indexed by chain name — the chain
|
||||
// analogue of healthByGroup, for the "unused" badge on a chain card (plan §5.E).
|
||||
// `chains` is absent from the single-group (?group=) health shape and from a
|
||||
// daemon version that doesn't report it yet; treat both as "not known".
|
||||
const healthByChain = useMemo(
|
||||
() => new Map((health.chains ?? []).map((c) => [c.name, c])),
|
||||
[health],
|
||||
)
|
||||
|
||||
// ---- group test: how fast, through which node, out which address ------------
|
||||
// Same shape as the "Test all nodes" probe-all on the Nodes page: the POST only
|
||||
// kicks a run off, and a 2 s poll of the GET carries progress plus every result
|
||||
// so far. Reused deliberately rather than invented a second time.
|
||||
// ---- group/chain exit test: how fast, through which node, out which address --
|
||||
// The POST only kicks a run off, and a 2 s poll of the GET carries progress
|
||||
// plus every result so far. One endpoint covers groups and chains alike:
|
||||
// POST with a group or chain name tests that one; an empty name tests them all.
|
||||
const [gtest, setGtest] = useState<GroupTestStatus>(IDLE_TEST)
|
||||
const [gtestErr, setGtestErr] = useState<string | null>(null)
|
||||
const [polling, setPolling] = useState(false)
|
||||
@@ -680,7 +651,7 @@ export default function Targets() {
|
||||
if (r.started) {
|
||||
setGtestErr(null)
|
||||
setPolling(true)
|
||||
flash(name ? `Testing ${name}…` : 'Testing every group…')
|
||||
flash(name ? `Testing ${name}…` : 'Testing every exit…')
|
||||
void readTest()
|
||||
} else if (r.reason === 'already running') {
|
||||
setPolling(true) // pick up the run someone else started
|
||||
@@ -701,7 +672,7 @@ export default function Targets() {
|
||||
)
|
||||
|
||||
/**
|
||||
* The groups the current run covers. This — not `running` — is what puts the
|
||||
* The groups and chains the current run covers. This — not `running` — is what puts the
|
||||
* in-progress badge on a card: `running` alone only says a test is happening
|
||||
* SOMEWHERE, which is why testing one group used to light up all four.
|
||||
*
|
||||
@@ -770,10 +741,6 @@ export default function Targets() {
|
||||
const chainNames = useMemo(() => new Set(chains.map((c) => c.Name)), [chains])
|
||||
const egressNames = useMemo(() => new Set(egresses.map((e) => e.Name)), [egresses])
|
||||
|
||||
const globals = config?.Globals
|
||||
const defaultProbeURL = globals?.ProbeURL ?? ''
|
||||
const defaultProbeInterval = globals?.ProbeInterval ?? ''
|
||||
|
||||
// Hop picker options for chains: every group and node in the config.
|
||||
const hopOptions = useMemo<Opt[]>(
|
||||
() => [
|
||||
@@ -805,7 +772,8 @@ export default function Targets() {
|
||||
return save({ ...config, Groups: [...groups, { ...draft, Name: nm }] }, `Added ${nm}`)
|
||||
}
|
||||
// A rename must carry every reference with it (rule targets, chain hops,
|
||||
// resolver detours…) or those silently fall back to the default route.
|
||||
// resolver detours…) or those are orphaned — rules left pointing at nothing
|
||||
// have their traffic blocked.
|
||||
const carried = renameReferences(config, 'group', editing, draft.Name)
|
||||
const moved = editing === draft.Name ? 0 : findReferences(config, 'group', editing).length
|
||||
return save(
|
||||
@@ -925,70 +893,33 @@ export default function Targets() {
|
||||
<header className="tg-sec-hd">
|
||||
<h2 className="tg-sec-title">Groups</h2>
|
||||
<span className="tg-sec-count mono">{groups.length} configured</span>
|
||||
{/* Two runs live here and they are NOT the same action. A health
|
||||
refresh measures every node in every group at once — it is global by
|
||||
construction, so it gets exactly one indicator, here, and never a
|
||||
badge on a card. A group exit test is scoped to the groups it names,
|
||||
so its progress says WHICH, and its badge lands only on those cards. */}
|
||||
{/* The observatory's background probing is invisible by design — it
|
||||
keeps every used group's and chain's numbers fresh on its own. The
|
||||
one manual run left is the exit test: it is scoped to the groups
|
||||
and chains it names, so its progress says WHICH, and its badge
|
||||
lands only on those cards. */}
|
||||
<div className="tg-sec-ctl">
|
||||
{groupHealthOn && (
|
||||
<>
|
||||
{health.node_test_running && (
|
||||
<span
|
||||
className="tg-run tg-run--health"
|
||||
role="status"
|
||||
title="One health run measures every member of every group, including each group’s own egress copies. It covers all groups at once, so it is reported here and not on any single card."
|
||||
>
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-run-what">health · all groups</span>
|
||||
<span className="tg-run-n mono">
|
||||
{health.node_test_done}/{health.node_test_total}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
{gtest.running && (
|
||||
<span
|
||||
className="tg-run tg-run--exit"
|
||||
role="status"
|
||||
title="An exit test sends one connection through each group it covers and reports the delay and the address the internet sees."
|
||||
title="An exit test sends one connection through each group or chain it covers and reports the delay and the address the internet sees."
|
||||
>
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-run-what">
|
||||
exit test · {scopeLabel(asArray(gtest.scope), groups.length)}
|
||||
exit test · {scopeLabel(asArray(gtest.scope), groups.length + chains.length)}
|
||||
</span>
|
||||
<span className="tg-run-n mono">
|
||||
{gtest.done}/{gtest.total}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
{!health.node_test_running && health.sweep?.enabled && health.sweep.total > 0 && (
|
||||
<span
|
||||
className="tg-run tg-run--sweep"
|
||||
title={
|
||||
health.sweep.cycles === 0
|
||||
? 'The router re-probes nodes in the background so these numbers stay fresh. It is still on its first pass, so members it hasn’t reached yet read “not measured”.'
|
||||
: `The router re-probes nodes in the background so these numbers stay fresh. It has completed ${health.sweep.cycles} full pass${health.sweep.cycles === 1 ? '' : 'es'}, so anything still unmeasured lost a reading it used to have.`
|
||||
}
|
||||
>
|
||||
<span className="tg-run-what">
|
||||
sweep{health.sweep.cycles === 0 ? ' · first pass' : ''}
|
||||
</span>
|
||||
<span className="tg-run-n mono">
|
||||
{health.sweep.cursor}/{health.sweep.total}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
<Button
|
||||
onClick={() => void measureAll()}
|
||||
disabled={measuring || health.node_test_running || groups.length === 0}
|
||||
title="Probe every member of every group now, including each group’s own egress copies. One run, all groups — health also refreshes on its own in the background."
|
||||
>
|
||||
{health.node_test_running ? 'Measuring…' : 'Refresh health'}
|
||||
</Button>
|
||||
<Button
|
||||
onClick={() => void runTest()}
|
||||
disabled={busy || !config || groups.length === 0 || gtest.running}
|
||||
title="Send one connection through each group and report the delay and the exit address the internet sees"
|
||||
disabled={busy || !config || (groups.length === 0 && chains.length === 0) || gtest.running}
|
||||
title="Send one connection through each group and chain and report the delay and the exit address the internet sees"
|
||||
>
|
||||
{gtest.running ? 'Testing…' : 'Test every exit'}
|
||||
</Button>
|
||||
@@ -1035,8 +966,6 @@ export default function Targets() {
|
||||
egresses={egresses}
|
||||
taken={groupNames}
|
||||
busy={busy}
|
||||
defaultProbeURL={defaultProbeURL}
|
||||
defaultProbeInterval={defaultProbeInterval}
|
||||
onCancel={() => setGroupEd(null)}
|
||||
onSave={async (g) => {
|
||||
const ok = await commitGroup(g, null)
|
||||
@@ -1066,8 +995,6 @@ export default function Targets() {
|
||||
egresses={egresses}
|
||||
taken={without(groupNames, g.Name)}
|
||||
busy={busy}
|
||||
defaultProbeURL={defaultProbeURL}
|
||||
defaultProbeInterval={defaultProbeInterval}
|
||||
onCancel={() => setGroupEd(null)}
|
||||
onSave={async (ng) => {
|
||||
const ok = await commitGroup(ng, g.Name)
|
||||
@@ -1084,8 +1011,6 @@ export default function Targets() {
|
||||
showHealth={groupHealthOn}
|
||||
health={healthByGroup.get(g.Name)}
|
||||
healthKnown={healthRead}
|
||||
sweep={health.sweep}
|
||||
onMeasure={() => void measureAll()}
|
||||
test={testByGroup.get(g.Name)}
|
||||
// The badge is this card's business only when the run names it.
|
||||
testing={gtest.running && testScope.has(g.Name)}
|
||||
@@ -1168,6 +1093,15 @@ export default function Targets() {
|
||||
key={c.Name}
|
||||
chain={c}
|
||||
busy={busy}
|
||||
showHealth={groupHealthOn}
|
||||
used={healthByChain.get(c.Name)?.used}
|
||||
test={testByGroup.get(c.Name)}
|
||||
// The badge is this card's business only when the run names it.
|
||||
testing={gtest.running && testScope.has(c.Name)}
|
||||
// …but the daemon runs one test at a time, so any run in flight
|
||||
// is what disables the button.
|
||||
testBusy={gtest.running}
|
||||
onTest={() => void runTest(c.Name)}
|
||||
onEdit={() => setChainEd({ mode: 'edit', name: c.Name })}
|
||||
onDelete={() => removeChain(c.Name)}
|
||||
/>
|
||||
@@ -1272,8 +1206,6 @@ function GroupRow({
|
||||
showHealth,
|
||||
health,
|
||||
healthKnown,
|
||||
sweep,
|
||||
onMeasure,
|
||||
test,
|
||||
testing,
|
||||
testBusy,
|
||||
@@ -1292,11 +1224,6 @@ function GroupRow({
|
||||
/** The health endpoint has answered at least once, so an absent entry really
|
||||
* does mean "the engine doesn't have this group". */
|
||||
healthKnown: boolean
|
||||
/** The daemon's background sweep, so the card can say whether untested members
|
||||
* resolve by themselves. Without it, "wait and it will fill in" is a false
|
||||
* promise. Absent on daemons that don't report it. */
|
||||
sweep?: HealthSweep
|
||||
onMeasure: () => void
|
||||
test?: GroupTestResult
|
||||
/**
|
||||
* A group exit test covering THIS group is in flight.
|
||||
@@ -1365,8 +1292,6 @@ function GroupRow({
|
||||
egress={group.Egress ?? ''}
|
||||
health={health}
|
||||
healthKnown={healthKnown}
|
||||
sweep={sweep}
|
||||
onMeasure={onMeasure}
|
||||
/>
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
</>
|
||||
@@ -1404,15 +1329,11 @@ function GroupHealthReadout({
|
||||
egress,
|
||||
health,
|
||||
healthKnown,
|
||||
sweep,
|
||||
onMeasure,
|
||||
}: {
|
||||
name: string
|
||||
egress: string
|
||||
health?: GroupHealth
|
||||
healthKnown: boolean
|
||||
sweep?: HealthSweep
|
||||
onMeasure: () => void
|
||||
}) {
|
||||
const [open, setOpen] = useState(false)
|
||||
const panelId = `gh-members-${name}`
|
||||
@@ -1438,10 +1359,27 @@ function GroupHealthReadout({
|
||||
)
|
||||
}
|
||||
|
||||
// A completed sweep pass is what makes silence meaningful; until then a group's
|
||||
// own history can only have recorded successes. See isBiasedSample.
|
||||
const sweptOnce = (sweep?.cycles ?? 0) >= 1
|
||||
const v = verdictOf(health, sweptOnce)
|
||||
// No enabled rule reaches this group, so the observatory never probes it and
|
||||
// its members would stay "untested" forever. That is a fact about the ROUTING
|
||||
// CONFIG, not about the members — so instead of counters that could only ever
|
||||
// read as a permanent unknown, the card says so, quietly: unused, not unwell.
|
||||
if (!health.used) {
|
||||
return (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this group, so its members are not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this group</span>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
const v = verdictOf(health)
|
||||
const { total, tested, alive, dead, untested } = health
|
||||
const boundTitle = `Every member of “${name}” is a private copy dialled through ${
|
||||
egress ? `egress “${egress}”` : 'this group’s egress'
|
||||
@@ -1487,7 +1425,7 @@ function GroupHealthReadout({
|
||||
/* No fraction here, on purpose. `alive / tested` would put a denominator
|
||||
on a sample that nothing has yet been able to fail a member into, and
|
||||
it always reads 100%. Two plain counts instead: what is confirmed, and
|
||||
what is still open. When the sweep fills the gap this becomes a real
|
||||
what is still open. When the observatory fills the gap this becomes a real
|
||||
ratio — which reads as the panel getting more precise, not as the
|
||||
group getting worse. */
|
||||
<>
|
||||
@@ -1542,31 +1480,14 @@ function GroupHealthReadout({
|
||||
|
||||
{/* The three states that need a sentence rather than a number. */}
|
||||
{/* Why there is no ratio. It names the mechanism, because "we're not sure"
|
||||
without a reason reads as hedging — and because the mechanism is also the
|
||||
answer to "when will I know": either the sweep gets there, or you ask. */}
|
||||
without a reason reads as hedging — and because the mechanism is also
|
||||
the answer to "when will I know": the observatory's next pass, seconds
|
||||
away. */}
|
||||
{v === 'partial' && (
|
||||
<p className="gh-say">
|
||||
Not a proportion yet — a group records only the members that answer, so any that failed
|
||||
are still counted as unchecked.{' '}
|
||||
{sweep?.enabled ? (
|
||||
<>
|
||||
The background sweep is the only thing that confirms a member is down, and it hasn’t
|
||||
finished its first pass over this group. It gets there on its own, or{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>{' '}
|
||||
to settle it.
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
The background sweep is what confirms a member is down, and it is off — so nothing
|
||||
will settle this until you{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
)}
|
||||
are still counted as unchecked. The daemon’s background probing confirms failures too and
|
||||
settles this within seconds.
|
||||
</p>
|
||||
)}
|
||||
{v === 'down' && (
|
||||
@@ -1576,41 +1497,14 @@ function GroupHealthReadout({
|
||||
nowhere to go.
|
||||
</p>
|
||||
)}
|
||||
{/* Why nothing has been measured, and what will change that. The sweep's
|
||||
`cycles` is what separates the two honest readings of the same word:
|
||||
cycles 0 means the sweep simply hasn't arrived; cycles ≥ 1 means it HAS
|
||||
been everywhere, so these members lost the readings they once had. */}
|
||||
{/* Why nothing has been measured, and what will change that. For a used
|
||||
group the observatory gets here on its own — within seconds, not
|
||||
minutes — so the sentence promises exactly that and nothing more. */}
|
||||
{v === 'unmeasured' && (
|
||||
<p className="gh-say">
|
||||
Nothing has been probed through this group yet, so there is nothing to report — not a
|
||||
fault.{' '}
|
||||
{!sweep?.enabled ? (
|
||||
<>
|
||||
The background sweep is off, so this fills in only when you ask:{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
) : sweep.cycles === 0 ? (
|
||||
<>
|
||||
The background sweep is still on its first pass and gets to these within a few
|
||||
minutes, or{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
The background sweep has already been everywhere, so these lost the readings they
|
||||
had — probe them to find out where they stand:{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
)}
|
||||
fault. The observatory probes every group your rules use in the background, so these
|
||||
numbers fill in by themselves within seconds.
|
||||
</p>
|
||||
)}
|
||||
|
||||
@@ -1637,7 +1531,8 @@ function GroupHealthReadout({
|
||||
id={panelId}
|
||||
group={name}
|
||||
// Refetch when the cheap summary poll shows this group's counters moved
|
||||
// (a probe-all landing, say) — the open panel keeps no poll of its own.
|
||||
// (the observatory landing fresh verdicts, say) — the open panel keeps
|
||||
// no poll of its own.
|
||||
version={`${total}-${tested}-${alive}-${dead}`}
|
||||
/>
|
||||
)}
|
||||
@@ -1815,8 +1710,6 @@ function GroupEditor({
|
||||
egresses,
|
||||
taken,
|
||||
busy,
|
||||
defaultProbeURL,
|
||||
defaultProbeInterval,
|
||||
onCancel,
|
||||
onSave,
|
||||
}: {
|
||||
@@ -1829,8 +1722,6 @@ function GroupEditor({
|
||||
egresses: Egress[]
|
||||
taken: Set<string>
|
||||
busy: boolean
|
||||
defaultProbeURL: string
|
||||
defaultProbeInterval: string
|
||||
onCancel: () => void
|
||||
onSave: (g: Group) => Promise<boolean>
|
||||
}) {
|
||||
@@ -1848,8 +1739,6 @@ function GroupEditor({
|
||||
const [protos, setProtos] = useState<string[]>(asArray(initial?.FilterProto))
|
||||
const [country, setCountry] = useState(joinList(initial?.FilterCountry))
|
||||
const [dedup, setDedup] = useState(initial?.Dedup ?? false)
|
||||
const [probeURL, setProbeURL] = useState(initial?.ProbeURL ?? '')
|
||||
const [probeInterval, setProbeInterval] = useState(initial?.ProbeInterval ?? '')
|
||||
const [egress, setEgress] = useState(initial?.Egress ?? '')
|
||||
const [search, setSearch] = useState('')
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
@@ -1892,8 +1781,6 @@ function GroupEditor({
|
||||
Name: nm,
|
||||
Source: source,
|
||||
Strategy: strategy,
|
||||
ProbeURL: probeURL.trim() || undefined,
|
||||
ProbeInterval: probeInterval.trim() || undefined,
|
||||
// Always sent, '' for "not set" — the field name is `Egress`, matching
|
||||
// Node.Egress. PUT /api/config rejects the WHOLE model on an unknown key.
|
||||
Egress: egress,
|
||||
@@ -2195,34 +2082,6 @@ function GroupEditor({
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="tg-ed-grid">
|
||||
<label className="tg-field">
|
||||
<span className="tg-flabel">Probe URL</span>
|
||||
<input
|
||||
className="tg-input"
|
||||
value={probeURL}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
inputMode="url"
|
||||
placeholder={defaultProbeURL || 'default'}
|
||||
onChange={(e) => setProbeURL(e.target.value)}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
<label className="tg-field">
|
||||
<span className="tg-flabel">Probe interval</span>
|
||||
<input
|
||||
className="tg-input"
|
||||
value={probeInterval}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder={defaultProbeInterval || 'default'}
|
||||
onChange={(e) => setProbeInterval(e.target.value)}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
</div>
|
||||
|
||||
<EditorFoot
|
||||
busy={busy}
|
||||
err={err}
|
||||
@@ -2238,11 +2097,33 @@ function GroupEditor({
|
||||
function ChainRow({
|
||||
chain,
|
||||
busy,
|
||||
showHealth,
|
||||
used,
|
||||
test,
|
||||
testing,
|
||||
testBusy,
|
||||
onTest,
|
||||
onEdit,
|
||||
onDelete,
|
||||
}: {
|
||||
chain: Chain
|
||||
busy: boolean
|
||||
/** Group health checks are on (Settings). When false, the card drops its
|
||||
* exit-test readout and Test button — it is config only. */
|
||||
showHealth: boolean
|
||||
/** This chain's reachability (GroupHealth.Used's chain analogue, plan §5.E).
|
||||
* undefined ⇒ the health endpoint hasn't reported this chain (not applied yet, or
|
||||
* a daemon version without chains): no badge. false ⇒ no enabled rule routes
|
||||
* through the chain, so the observatory never probes it and the card renders
|
||||
* "unused" instead of an exit-test readout. */
|
||||
used?: boolean
|
||||
test?: GroupTestResult
|
||||
/** An exit test covering THIS chain is in flight (the caller resolves it
|
||||
* against the run's scope, exactly as for a group card). */
|
||||
testing: boolean
|
||||
/** Any exit test is in flight; the daemon runs one at a time. */
|
||||
testBusy: boolean
|
||||
onTest: () => void
|
||||
onEdit: () => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
@@ -2281,6 +2162,30 @@ function ChainRow({
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
{showHealth && (
|
||||
<>
|
||||
{/* A chain no enabled rule routes through is never probed (the
|
||||
observatory walks only reachable paths), so instead of an exit-test
|
||||
readout the card says so, quietly — the same "unused" pattern the
|
||||
group card uses (GroupHealthReadout), not a new design. `used` is
|
||||
undefined until the health endpoint reports this chain (or from a
|
||||
daemon version without chains): no badge then. */}
|
||||
{used === false && (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this chain, so its exit is not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this chain</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
<RowActions
|
||||
onEdit={onEdit}
|
||||
@@ -2288,6 +2193,9 @@ function ChainRow({
|
||||
busy={busy}
|
||||
editLabel={`Edit chain ${chain.Name}`}
|
||||
deleteLabel={`Delete chain ${chain.Name}`}
|
||||
onTest={showHealth ? onTest : undefined}
|
||||
testLabel={showHealth ? `Test the exit of chain ${chain.Name}` : undefined}
|
||||
testDisabled={testBusy}
|
||||
/>
|
||||
</li>
|
||||
)
|
||||
@@ -2767,7 +2675,7 @@ function RowActions({
|
||||
busy: boolean
|
||||
editLabel: string
|
||||
deleteLabel: string
|
||||
// Only groups can be tested today, so the control is optional and absent
|
||||
// Only groups and chains can be tested, so the control is optional and absent
|
||||
// everywhere else rather than a disabled stub on every row.
|
||||
onTest?: () => void
|
||||
testLabel?: string
|
||||
|
||||
Binary file not shown.
|
Before Width: | Height: | Size: 288 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 280 KiB |
Binary file not shown.
|
Before Width: | Height: | Size: 279 KiB |
+121
-113
@@ -93,6 +93,10 @@ func (s *URLTest) Start() error {
|
||||
}
|
||||
group.balancer = s.balancer // lx: SPEC 019 v2 — health-check drives the pool through it
|
||||
if s.balancer != nil {
|
||||
// lx: health board §5.B — slot liveness reads through the board verdict, so a
|
||||
// death recorded by any prober or a failed dial takes effect on the next pick,
|
||||
// not the next health-check tick.
|
||||
s.balancer.verdict = group.slotVerdict
|
||||
// lx: SPEC 020 — a pool rebuild changes the active routing tree; invalidate
|
||||
// the router's reachable cache. ctx captured here has the invalidator.
|
||||
ctx := s.ctx
|
||||
@@ -148,9 +152,10 @@ type PoolSlot struct {
|
||||
}
|
||||
|
||||
// Pool returns the current rotation pool (one entry per slot) for round_robin groups. For
|
||||
// least_test (nil balancer) it returns nil — "this group has no pool". Delay is read from
|
||||
// history and clamped 0->1 for live nodes so 0 in the output unambiguously means dead/untested.
|
||||
// lx: SPEC 019 v2 (exposed to clients via the GetPool RPC).
|
||||
// least_test (nil balancer) it returns nil — "this group has no pool". Delay is reported only
|
||||
// for slots with a fresh-alive board verdict, clamped 0->1, so 0 in the output unambiguously
|
||||
// means dead/untested (failures now persist in history — an entry alone no longer means alive).
|
||||
// lx: SPEC 019 v2 (exposed to clients via the GetPool RPC); health board §5.B.
|
||||
func (s *URLTest) Pool() []PoolSlot {
|
||||
if s.balancer == nil || s.group == nil {
|
||||
return nil
|
||||
@@ -167,10 +172,12 @@ func (s *URLTest) Pool() []PoolSlot {
|
||||
if node, loaded := s.outbound.Outbound(tag); loaded {
|
||||
historyTag = RealTag(node)
|
||||
}
|
||||
if history := s.group.history.LoadURLTestHistory(historyTag); history != nil {
|
||||
delay = history.Delay
|
||||
if delay == 0 {
|
||||
delay = 1 // live sub-ms node: never report 0 (0 is reserved for dead/untested)
|
||||
if s.group.history.Verdict(historyTag, s.group.healthTTL()) == urltest.VerdictAlive {
|
||||
if history := s.group.history.LoadURLTestHistory(historyTag); history != nil {
|
||||
delay = history.Delay
|
||||
if delay == 0 {
|
||||
delay = 1 // live sub-ms node: never report 0 (0 is reserved for dead/untested)
|
||||
}
|
||||
}
|
||||
}
|
||||
slots[i] = PoolSlot{Slot: i, Tag: tag, Delay: delay}
|
||||
@@ -187,10 +194,12 @@ func (s *URLTest) CheckOutbounds() {
|
||||
}
|
||||
|
||||
// selectBalanced picks an outbound per-connection in round_robin mode from the balancer's
|
||||
// fixed-size pool. lx: SPEC 019 v2. fallback (Select's outbounds[0]) covers the cold-start
|
||||
// window before the first health-check fills the pool. Returns nil only when nothing usable.
|
||||
func (s *URLTest) selectBalanced(ctx context.Context, network string, destination M.Socksaddr) adapter.Outbound {
|
||||
fallback, _ := s.group.Select(network)
|
||||
// fixed-size pool. lx: SPEC 019 v2. fallback (selectExcluding) covers the cold-start window
|
||||
// before the first health-check fills the pool and the all-slots-dead state. exclude carries
|
||||
// the members already tried by this connection's dial-retry loop (health board §5.B).
|
||||
// Returns nil only when nothing usable.
|
||||
func (s *URLTest) selectBalanced(ctx context.Context, network string, destination M.Socksaddr, exclude map[string]bool) adapter.Outbound {
|
||||
fallback, _ := s.group.selectExcluding(network, exclude)
|
||||
selected := s.balancer.pick(ctx, destination, fallback, func(tag string) adapter.Outbound {
|
||||
node, _ := s.outbound.Outbound(tag)
|
||||
if node != nil && !common.Contains(node.Network(), network) {
|
||||
@@ -206,68 +215,74 @@ func (s *URLTest) selectBalanced(ctx context.Context, network string, destinatio
|
||||
|
||||
func (s *URLTest) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
s.group.Touch()
|
||||
var outbound adapter.Outbound
|
||||
if s.balancer != nil {
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP, N.NetworkUDP:
|
||||
outbound = s.selectBalanced(ctx, network, destination)
|
||||
default:
|
||||
return nil, E.Extend(N.ErrUnknownNetwork, network)
|
||||
}
|
||||
} else {
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP:
|
||||
outbound = s.group.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
outbound = s.group.selectedOutboundUDP
|
||||
default:
|
||||
return nil, E.Extend(N.ErrUnknownNetwork, network)
|
||||
}
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP, N.NetworkUDP:
|
||||
default:
|
||||
return nil, E.Extend(N.ErrUnknownNetwork, network)
|
||||
}
|
||||
// lx: health board §5.B — a failed dial no longer fails the user's connection outright:
|
||||
// the member is marked dead on the board and the dial is retried through the next
|
||||
// candidate (at most dialAttemptsMax members, within the context deadline), for every
|
||||
// mode. The old behaviour deleted the history entry in least_test (making a dead member
|
||||
// indistinguishable from an untested one) and did nothing at all in balanced modes
|
||||
// (plan §2 Д1/Д5).
|
||||
var tried map[string]bool
|
||||
var lastErr error
|
||||
for range dialAttemptsMax {
|
||||
outbound := s.dialSelect(ctx, network, destination, tried)
|
||||
if outbound == nil {
|
||||
outbound, _ = s.group.Select(network)
|
||||
break
|
||||
}
|
||||
conn, err := outbound.DialContext(ctx, network, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
if tried == nil {
|
||||
tried = make(map[string]bool, dialAttemptsMax)
|
||||
}
|
||||
s.markDialFailure(outbound, tried, err)
|
||||
lastErr = err
|
||||
if ctx.Err() != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
if outbound == nil {
|
||||
return nil, E.New("missing supported outbound")
|
||||
if lastErr != nil {
|
||||
return nil, lastErr
|
||||
}
|
||||
conn, err := outbound.DialContext(ctx, network, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
// lx: SPEC 019 v2 — in round_robin a dial error must NOT touch the pool: the cause is
|
||||
// unknown (dead node vs. dead destination vs. local network drop). Only the health-check
|
||||
// changes pool membership. least_test keeps the upstream behaviour (drop the history).
|
||||
if s.balancer == nil {
|
||||
s.group.history.DeleteURLTestHistory(outbound.Tag())
|
||||
}
|
||||
return nil, err
|
||||
return nil, E.New("missing supported outbound")
|
||||
}
|
||||
|
||||
func (s *URLTest) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
s.group.Touch()
|
||||
var outbound adapter.Outbound
|
||||
if s.balancer != nil {
|
||||
outbound = s.selectBalanced(ctx, N.NetworkUDP, destination)
|
||||
} else {
|
||||
outbound = s.group.selectedOutboundUDP
|
||||
// lx: health board §5.B — same mark-fail + re-pick retry as DialContext, but only for
|
||||
// the ListenPacket call itself: once the packet conn is returned no retry is possible
|
||||
// (the UDP session is bound to its member from the first send).
|
||||
var tried map[string]bool
|
||||
var lastErr error
|
||||
for range dialAttemptsMax {
|
||||
outbound := s.dialSelect(ctx, N.NetworkUDP, destination, tried)
|
||||
if outbound == nil {
|
||||
outbound, _ = s.group.Select(N.NetworkUDP)
|
||||
break
|
||||
}
|
||||
conn, err := outbound.ListenPacket(ctx, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewPacketConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
if tried == nil {
|
||||
tried = make(map[string]bool, dialAttemptsMax)
|
||||
}
|
||||
s.markDialFailure(outbound, tried, err)
|
||||
lastErr = err
|
||||
if ctx.Err() != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
if outbound == nil {
|
||||
return nil, E.New("missing supported outbound")
|
||||
if lastErr != nil {
|
||||
return nil, lastErr
|
||||
}
|
||||
conn, err := outbound.ListenPacket(ctx, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewPacketConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
// lx: SPEC 019 v2 — round_robin dial error leaves the pool untouched (see DialContext).
|
||||
if s.balancer == nil {
|
||||
s.group.history.DeleteURLTestHistory(outbound.Tag())
|
||||
}
|
||||
return nil, err
|
||||
return nil, E.New("missing supported outbound")
|
||||
}
|
||||
|
||||
func (s *URLTest) NewConnection(ctx context.Context, conn net.Conn, metadata adapter.InboundContext, onClose N.CloseHandlerFunc) {
|
||||
@@ -382,47 +397,12 @@ func (g *URLTestGroup) Close() error {
|
||||
}
|
||||
|
||||
func (g *URLTestGroup) Select(network string) (adapter.Outbound, bool) {
|
||||
var minDelay uint16
|
||||
var minOutbound adapter.Outbound
|
||||
switch network {
|
||||
case N.NetworkTCP:
|
||||
if g.selectedOutboundTCP != nil {
|
||||
if history := g.history.LoadURLTestHistory(RealTag(g.selectedOutboundTCP)); history != nil {
|
||||
minOutbound = g.selectedOutboundTCP
|
||||
minDelay = history.Delay
|
||||
}
|
||||
}
|
||||
case N.NetworkUDP:
|
||||
if g.selectedOutboundUDP != nil {
|
||||
if history := g.history.LoadURLTestHistory(RealTag(g.selectedOutboundUDP)); history != nil {
|
||||
minOutbound = g.selectedOutboundUDP
|
||||
minDelay = history.Delay
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
history := g.history.LoadURLTestHistory(RealTag(detour))
|
||||
if history == nil {
|
||||
continue
|
||||
}
|
||||
if minDelay == 0 || minDelay > history.Delay+g.tolerance {
|
||||
minDelay = history.Delay
|
||||
minOutbound = detour
|
||||
}
|
||||
}
|
||||
if minOutbound == nil {
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
return detour, false
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
return minOutbound, true
|
||||
// lx: health board §5.B — selection reads the board verdict instead of "has a history
|
||||
// entry": fresh-alive members ranked by delay first, untested members in config order
|
||||
// second, first-by-config last — a dead member is never picked while a live or untested
|
||||
// one exists. The logic lives in selectExcluding (urltest_health_lx.go) so the
|
||||
// dial-retry path can re-run it minus the members that just failed.
|
||||
return g.selectExcluding(network, nil)
|
||||
}
|
||||
|
||||
func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{}) {
|
||||
@@ -451,6 +431,16 @@ func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{})
|
||||
}
|
||||
}
|
||||
|
||||
// lx: health board §5.B/§5.C — CheckOutbounds is the FAST loop of the pair: an
|
||||
// ACTIVE group probes its own members on its own ticker (interval = the global
|
||||
// probe interval; failover keeps its 30s default), and a failed probe writes
|
||||
// MarkFailed to the shared health board (common/urltest) instead of deleting
|
||||
// the entry. The shater observatory is the complementary BACKGROUND loop: it
|
||||
// probes what no active group measures (idle groups' members, selector
|
||||
// candidates, chain exits), and its freshness gate skips any tag this loop
|
||||
// keeps current — no duplicate probes, and this loop's cadence is never
|
||||
// suppressed. Both loops write to the SAME board, so selection and the panel
|
||||
// see one source of truth no matter which prober found the death.
|
||||
func (g *URLTestGroup) CheckOutbounds(force bool) {
|
||||
_, _ = g.urlTest(g.ctx, force)
|
||||
}
|
||||
@@ -481,8 +471,10 @@ func (g *URLTestGroup) urlTest(ctx context.Context, force bool) (map[string]uint
|
||||
}
|
||||
|
||||
// testNodes runs the URL test over the given outbounds (skipping fresh history unless force),
|
||||
// stores/deletes history, and returns tag->delay for the live ones. lx: shared by least_test
|
||||
// and the round_robin force path.
|
||||
// stores successes and marks failures on the health board, and returns tag->delay for the live
|
||||
// ones. lx: shared by least_test and the round_robin force path; this is the probe
|
||||
// primitive of the fast active-group loop (see CheckOutbounds for the two-loop
|
||||
// contract with the shater observatory).
|
||||
func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbound, force bool) map[string]uint16 {
|
||||
result := make(map[string]uint16)
|
||||
b, _ := batch.New(ctx, batch.WithConcurrencyNum[any](10))
|
||||
@@ -495,7 +487,7 @@ func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbou
|
||||
continue
|
||||
}
|
||||
history := g.history.LoadURLTestHistory(realTag)
|
||||
if !force && history != nil && time.Since(history.Time) < g.interval {
|
||||
if !force && history != nil && time.Since(history.LastOK) < g.interval { // lx: health board §5.A — Time renamed to LastOK
|
||||
continue
|
||||
}
|
||||
checked[realTag] = true
|
||||
@@ -509,12 +501,18 @@ func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbou
|
||||
t, err := urltest.URLTest(testCtx, g.link, p)
|
||||
if err != nil {
|
||||
g.logger.Debug("outbound ", tag, " unavailable: ", err)
|
||||
g.history.DeleteURLTestHistory(realTag)
|
||||
// lx: health board §5.A — a failed check marks the entry instead of deleting
|
||||
// it (deletion stays reserved for members removed from the configuration).
|
||||
g.markFailedLogged(realTag, tag, "probe", err)
|
||||
} else {
|
||||
g.logger.Debug("outbound ", tag, " available: ", t, "ms")
|
||||
// lx: health board — log the dead -> alive flip before the success overwrites it.
|
||||
if g.history.Verdict(realTag, g.healthTTL()) == urltest.VerdictDead {
|
||||
g.logger.Info("outbound ", tag, " flipped dead -> alive (probe: ", t, "ms)")
|
||||
}
|
||||
g.history.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: t,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: t,
|
||||
})
|
||||
resultAccess.Lock()
|
||||
result[tag] = t
|
||||
@@ -757,7 +755,11 @@ func (g *URLTestGroup) rebuildPool() {
|
||||
results := make(map[string]candidate, len(g.outbounds))
|
||||
for _, detour := range g.outbounds {
|
||||
tag := detour.Tag()
|
||||
if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
realTag := RealTag(detour)
|
||||
// lx: health board §5.B — alive = fresh board verdict, not entry presence
|
||||
// (failures persist in history now).
|
||||
if g.history.Verdict(realTag, g.healthTTL()) == urltest.VerdictAlive {
|
||||
history := g.history.LoadURLTestHistory(realTag)
|
||||
results[tag] = candidate{tag: tag, delay: history.Delay, alive: true}
|
||||
} else {
|
||||
results[tag] = candidate{tag: tag, alive: false}
|
||||
@@ -819,8 +821,10 @@ func (g *URLTestGroup) rebuildPool() {
|
||||
g.balancer.setSlots(planTolerantPool(current, results, size, g.balancer.poolTolerance), tolerantLive)
|
||||
}
|
||||
|
||||
// seedPool fills the pool before the first health-check: prefer nodes with live history (the
|
||||
// process was not unloaded), else the first `size` nodes in config order. lx: SPEC 019 v2.
|
||||
// seedPool fills the pool before the first health-check: prefer nodes the board holds a
|
||||
// fresh-alive verdict for (the process was not unloaded), else the first `size` nodes in
|
||||
// config order. lx: SPEC 019 v2; health board §5.B — a persisted failure record must not
|
||||
// look like a warm node.
|
||||
func (g *URLTestGroup) seedPool() {
|
||||
if g.balancer == nil {
|
||||
return
|
||||
@@ -829,10 +833,14 @@ func (g *URLTestGroup) seedPool() {
|
||||
if size == 0 {
|
||||
return
|
||||
}
|
||||
// Nodes with existing history first (top by delay), then config order to fill.
|
||||
// Fresh-alive nodes first (top by delay), then config order to fill.
|
||||
withHistory := make([]candidate, 0, len(g.outbounds))
|
||||
for _, detour := range g.outbounds {
|
||||
if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
realTag := RealTag(detour)
|
||||
if g.history.Verdict(realTag, g.healthTTL()) != urltest.VerdictAlive {
|
||||
continue
|
||||
}
|
||||
if history := g.history.LoadURLTestHistory(realTag); history != nil {
|
||||
withHistory = append(withHistory, candidate{tag: detour.Tag(), delay: history.Delay, alive: true})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@ import (
|
||||
"sync/atomic"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
E "github.com/sagernet/sing/common/exceptions"
|
||||
@@ -46,6 +47,14 @@ type balancer struct {
|
||||
random bool // mode == random: pick a uniformly-random live slot per connection
|
||||
priority bool // priority == config order is the ranking (failover); enables fail-back
|
||||
|
||||
// verdict, if set, reads the health-board verdict for a slot tag (wired to
|
||||
// URLTestGroup.slotVerdict). Slot liveness is the verdict when the board has fresh
|
||||
// data — alive forces live, dead forces dead — and the health-check flag otherwise
|
||||
// (untested keeps the optimistic seed usable on cold start). This is what makes a
|
||||
// death recorded by ANY prober (group checker, observatory, failed dial) take effect
|
||||
// on the next pick instead of the next tick. Health board plan §5.B.
|
||||
verdict func(tag string) urltest.HealthVerdict
|
||||
|
||||
access sync.Mutex // guards slots
|
||||
slots []slot // the pool; len == min(poolSize, available nodes), index = fixed slot number
|
||||
counter atomic.Uint64
|
||||
@@ -133,7 +142,7 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
|
||||
}
|
||||
liveCount := 0
|
||||
for i := range b.slots {
|
||||
if b.slots[i].live {
|
||||
if b.slotIsLive(i) {
|
||||
liveCount++
|
||||
}
|
||||
}
|
||||
@@ -155,7 +164,7 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
|
||||
// picked deterministically from the same key — the flow still lands somewhere working.
|
||||
h := hashKey(b.stickyKey(ctx, destination))
|
||||
idx := int(h % uint64(n))
|
||||
if b.slots[idx].live {
|
||||
if b.slotIsLive(idx) {
|
||||
tag = b.slots[idx].tag
|
||||
} else {
|
||||
tag = b.nthLiveTag(int(h % uint64(liveCount)))
|
||||
@@ -172,11 +181,30 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
|
||||
return fallback
|
||||
}
|
||||
|
||||
// slotIsLive reports whether slot i is selectable: the board verdict wins when fresh
|
||||
// (alive → live, dead → dead), the last health-check flag decides for untested slots.
|
||||
// Caller must hold access. Health board plan §5.B.
|
||||
func (b *balancer) slotIsLive(i int) bool {
|
||||
s := b.slots[i]
|
||||
if s.tag == "" {
|
||||
return false
|
||||
}
|
||||
if b.verdict != nil {
|
||||
switch b.verdict(s.tag) {
|
||||
case urltest.VerdictAlive:
|
||||
return true
|
||||
case urltest.VerdictDead:
|
||||
return false
|
||||
}
|
||||
}
|
||||
return s.live
|
||||
}
|
||||
|
||||
// nthLiveTag returns the tag of the k-th live slot (0-based, in slot order). Caller must hold
|
||||
// access and pass k in [0, liveCount). Walks without allocating (pools may be large for random).
|
||||
func (b *balancer) nthLiveTag(k int) string {
|
||||
for i := range b.slots {
|
||||
if !b.slots[i].live {
|
||||
if !b.slotIsLive(i) {
|
||||
continue
|
||||
}
|
||||
if k == 0 {
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
// lx:begin health-board
|
||||
|
||||
// Health-board driven selection and dial retry for the urltest group (plan §5.B).
|
||||
//
|
||||
// The group used to equate "has a history entry" with "alive": failures deleted the
|
||||
// entry, so a dead member was indistinguishable from a never-measured one (plan §2 Д5)
|
||||
// and a stale success counted as alive forever (Д2). With failures now recorded on the
|
||||
// board (common/urltest MarkFailed), liveness is a read-time verdict with a TTL derived
|
||||
// from the group's check interval. Selection prefers fresh-alive members ranked by
|
||||
// delay, falls back to untested members in config order, and only then to the first
|
||||
// member by config; a failed user dial marks the member dead on the board and retries
|
||||
// the connection through the next candidate instead of failing outright (Д1).
|
||||
|
||||
package group
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
"github.com/sagernet/sing/common"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
)
|
||||
|
||||
// dialAttemptsMax bounds how many distinct members one connection may try: the first
|
||||
// pick plus up to two re-picks after a mark-fail. The context deadline still applies to
|
||||
// every attempt, so a short dial timeout cuts the sequence earlier.
|
||||
const dialAttemptsMax = 3
|
||||
|
||||
// healthTTLFloor keeps verdicts meaningful for groups with short check intervals: a
|
||||
// single missed tick must not flip a member to untested.
|
||||
const healthTTLFloor = 10 * time.Minute
|
||||
|
||||
// healthTTL is the freshness window for board verdicts: three check intervals (a member
|
||||
// that missed several consecutive checks is stale), never below healthTTLFloor.
|
||||
func (g *URLTestGroup) healthTTL() time.Duration {
|
||||
ttl := 3 * g.interval
|
||||
if ttl < healthTTLFloor {
|
||||
ttl = healthTTLFloor
|
||||
}
|
||||
return ttl
|
||||
}
|
||||
|
||||
// markFailedLogged records a failure for realTag on the board (the entry is kept — plan
|
||||
// §2 Д5) and logs the alive → dead verdict flip with its cause (probe or dial), the
|
||||
// single diagnostic trail for locating dead members and spotting flapping.
|
||||
func (g *URLTestGroup) markFailedLogged(realTag string, displayTag string, reason string, cause error) {
|
||||
previous := g.history.Verdict(realTag, g.healthTTL())
|
||||
g.history.MarkFailed(realTag)
|
||||
if previous == urltest.VerdictAlive {
|
||||
g.logger.Info("outbound ", displayTag, " flipped alive -> dead (", reason, ": ", cause, ")")
|
||||
}
|
||||
}
|
||||
|
||||
// slotVerdict reports the board verdict for a balancer slot tag. Slots hold member tags
|
||||
// as configured while the board is keyed by RealTag (a nested group's live leaf), so
|
||||
// resolve through the outbound manager first — same discipline as Pool().
|
||||
func (g *URLTestGroup) slotVerdict(tag string) urltest.HealthVerdict {
|
||||
historyTag := tag
|
||||
if node, loaded := g.outbound.Outbound(tag); loaded {
|
||||
historyTag = RealTag(node)
|
||||
}
|
||||
return g.history.Verdict(historyTag, g.healthTTL())
|
||||
}
|
||||
|
||||
// selectExcluding is Select with an exclusion set (RealTag keys) for the dial-retry
|
||||
// path: a member that just failed a dial is marked dead on the board AND excluded here,
|
||||
// so even the last-resort config-order fallback cannot re-pick it.
|
||||
func (g *URLTestGroup) selectExcluding(network string, exclude map[string]bool) (adapter.Outbound, bool) {
|
||||
ttl := g.healthTTL()
|
||||
var minDelay uint16
|
||||
var minOutbound adapter.Outbound
|
||||
// Keep the upstream hysteresis: the currently selected outbound only yields to a
|
||||
// member faster by more than tolerance — but only while it is still alive itself.
|
||||
var current adapter.Outbound
|
||||
switch network {
|
||||
case N.NetworkTCP:
|
||||
current = g.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
current = g.selectedOutboundUDP
|
||||
}
|
||||
if current != nil {
|
||||
currentTag := RealTag(current)
|
||||
if !exclude[currentTag] && g.history.Verdict(currentTag, ttl) == urltest.VerdictAlive {
|
||||
if history := g.history.LoadURLTestHistory(currentTag); history != nil {
|
||||
minOutbound = current
|
||||
minDelay = history.Delay
|
||||
}
|
||||
}
|
||||
}
|
||||
var firstUntested adapter.Outbound
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
realTag := RealTag(detour)
|
||||
if exclude[realTag] {
|
||||
continue
|
||||
}
|
||||
switch g.history.Verdict(realTag, ttl) {
|
||||
case urltest.VerdictAlive:
|
||||
history := g.history.LoadURLTestHistory(realTag)
|
||||
if history == nil {
|
||||
continue
|
||||
}
|
||||
if minDelay == 0 || minDelay > history.Delay+g.tolerance {
|
||||
minDelay = history.Delay
|
||||
minOutbound = detour
|
||||
}
|
||||
case urltest.VerdictUntested:
|
||||
if firstUntested == nil {
|
||||
firstUntested = detour
|
||||
}
|
||||
}
|
||||
}
|
||||
if minOutbound != nil {
|
||||
return minOutbound, true
|
||||
}
|
||||
// No fresh-alive member: an untested one (config order) is a better bet than a
|
||||
// known-dead one. When every member is dead, fall back to config order so the group
|
||||
// still dials something — the retry loop walks further members on failure.
|
||||
if firstUntested != nil {
|
||||
return firstUntested, false
|
||||
}
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
if exclude[RealTag(detour)] {
|
||||
continue
|
||||
}
|
||||
return detour, false
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
// dialSelect picks the member for one dial attempt, skipping members this connection has
|
||||
// already tried (and marked dead). In least_test mode the cached selection is used only
|
||||
// while its verdict is not dead — a member the board knows to be down is re-selected
|
||||
// around immediately instead of waiting for the next checker tick.
|
||||
func (s *URLTest) dialSelect(ctx context.Context, network string, destination M.Socksaddr, tried map[string]bool) adapter.Outbound {
|
||||
if s.balancer != nil {
|
||||
return s.selectBalanced(ctx, network, destination, tried)
|
||||
}
|
||||
var outbound adapter.Outbound
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP:
|
||||
outbound = s.group.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
outbound = s.group.selectedOutboundUDP
|
||||
}
|
||||
if outbound != nil {
|
||||
realTag := RealTag(outbound)
|
||||
if !tried[realTag] && s.group.history.Verdict(realTag, s.group.healthTTL()) != urltest.VerdictDead {
|
||||
return outbound
|
||||
}
|
||||
}
|
||||
outbound, _ = s.group.selectExcluding(network, tried)
|
||||
return outbound
|
||||
}
|
||||
|
||||
// markDialFailure records a failed dial: the member is marked dead on the board — every
|
||||
// reader (this group's next pick, the balancer slots via slotVerdict, the panel) sees it
|
||||
// immediately — and added to the connection's tried set so the retry never re-picks it.
|
||||
func (s *URLTest) markDialFailure(outbound adapter.Outbound, tried map[string]bool, err error) {
|
||||
realTag := RealTag(outbound)
|
||||
s.group.markFailedLogged(realTag, outbound.Tag(), "dial", err)
|
||||
tried[realTag] = true
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -0,0 +1,505 @@
|
||||
package group
|
||||
|
||||
// lx: health board §5.B tests — verdict-driven selection, slot liveness read-through,
|
||||
// and the dial-retry path, per mode × {alive, dead, untested, stale}.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/interrupt"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
)
|
||||
|
||||
// --- fixtures -----------------------------------------------------------------------
|
||||
|
||||
// healthNode is a dialable fake outbound: DialContext/ListenPacket succeed or fail per
|
||||
// the fail flag and record the dial order into dialed.
|
||||
type healthNode struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
fail bool
|
||||
dialed *[]string
|
||||
}
|
||||
|
||||
func (n *healthNode) Tag() string { return n.tag }
|
||||
func (n *healthNode) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||
|
||||
func (n *healthNode) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
}
|
||||
if n.fail {
|
||||
return nil, errors.New("dial refused")
|
||||
}
|
||||
left, right := net.Pipe()
|
||||
_ = right.Close()
|
||||
return left, nil
|
||||
}
|
||||
|
||||
func (n *healthNode) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
}
|
||||
if n.fail {
|
||||
return nil, errors.New("listen refused")
|
||||
}
|
||||
return net.ListenPacket("udp", "127.0.0.1:0")
|
||||
}
|
||||
|
||||
// fakeOutboundManager resolves tags over a fixed node set (only Outbound is used).
|
||||
type fakeOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
nodes map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *fakeOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
node, ok := m.nodes[tag]
|
||||
return node, ok
|
||||
}
|
||||
|
||||
func managerOf(nodes ...adapter.Outbound) *fakeOutboundManager {
|
||||
m := &fakeOutboundManager{nodes: make(map[string]adapter.Outbound, len(nodes))}
|
||||
for _, node := range nodes {
|
||||
m.nodes[node.Tag()] = node
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// healthTestGroup builds a minimal URLTestGroup (not started: no ticker, no checker).
|
||||
// interval 3m → healthTTL = 10m floor.
|
||||
func healthTestGroup(hist *urltest.HistoryStorage, manager adapter.OutboundManager, nodes ...adapter.Outbound) *URLTestGroup {
|
||||
return &URLTestGroup{
|
||||
ctx: context.Background(),
|
||||
outbound: manager,
|
||||
outbounds: nodes,
|
||||
history: hist,
|
||||
interval: 3 * time.Minute,
|
||||
tolerance: 50,
|
||||
logger: log.NewNOPFactory().Logger(),
|
||||
interruptGroup: interrupt.NewGroup(),
|
||||
}
|
||||
}
|
||||
|
||||
// healthURLTest wraps a group into a dialable *URLTest, wiring the balancer the same way
|
||||
// Start() does (slot liveness through the board verdict).
|
||||
func healthURLTest(g *URLTestGroup, bal *balancer, manager adapter.OutboundManager) *URLTest {
|
||||
g.balancer = bal
|
||||
if bal != nil {
|
||||
bal.verdict = g.slotVerdict
|
||||
}
|
||||
return &URLTest{
|
||||
group: g,
|
||||
balancer: bal,
|
||||
outbound: manager,
|
||||
logger: log.NewNOPFactory().Logger(),
|
||||
}
|
||||
}
|
||||
|
||||
// storeAlive records a fresh success 1s in the past: still well inside the TTL, but
|
||||
// strictly older than any failure the test triggers afterwards (Windows' coarse
|
||||
// monotonic clock can otherwise produce LastOK == LastFail ties within one tick).
|
||||
func storeAlive(hist *urltest.HistoryStorage, tag string, delay uint16) {
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: time.Now().Add(-time.Second), Delay: delay})
|
||||
}
|
||||
|
||||
// storeStale records a success far older than the TTL: verdict must degrade to untested.
|
||||
func storeStale(hist *urltest.HistoryStorage, tag string, delay uint16) {
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: time.Now().Add(-time.Hour), Delay: delay})
|
||||
}
|
||||
|
||||
// --- healthTTL ----------------------------------------------------------------------
|
||||
|
||||
func TestHealthTTLFloor(t *testing.T) {
|
||||
g := &URLTestGroup{interval: 3 * time.Minute}
|
||||
if ttl := g.healthTTL(); ttl != 10*time.Minute {
|
||||
t.Fatalf("healthTTL(3m) = %v, want the 10m floor", ttl)
|
||||
}
|
||||
g.interval = 5 * time.Minute
|
||||
if ttl := g.healthTTL(); ttl != 15*time.Minute {
|
||||
t.Fatalf("healthTTL(5m) = %v, want 3×interval = 15m", ttl)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Select (least_test): alive > untested > config order ---------------------------
|
||||
|
||||
func TestSelectPrefersAliveOverDeadAndUntested(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b, c := &balNode{tag: "a"}, &balNode{tag: "b"}, &balNode{tag: "c"}
|
||||
g := healthTestGroup(hist, nil, a, b, c)
|
||||
// a: dead but with a LOWER recorded delay than b (the old "has an entry" logic
|
||||
// would have picked it); b: alive; c: untested.
|
||||
storeAlive(hist, "a", 10)
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 100)
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v (exists %v), want alive b", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectRanksAliveByDelay(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
storeAlive(hist, "a", 200)
|
||||
storeAlive(hist, "b", 50) // 200 > 50+tolerance(50) → b wins
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v (exists %v), want faster alive b", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectHysteresisKeepsAliveCurrent(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
g.selectedOutboundTCP = a
|
||||
storeAlive(hist, "a", 100)
|
||||
storeAlive(hist, "b", 60) // within tolerance (100 ≤ 60+50) → keep a
|
||||
if selected, _ := g.Select(N.NetworkTCP); selected != adapter.Outbound(a) {
|
||||
t.Fatalf("Select = %v, want current a kept within tolerance", selected)
|
||||
}
|
||||
storeAlive(hist, "b", 40) // beyond tolerance (100 > 40+50) → switch to b
|
||||
if selected, _ := g.Select(N.NetworkTCP); selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v, want b beyond tolerance", selected)
|
||||
}
|
||||
}
|
||||
|
||||
// A dead current selection must NOT seed the hysteresis: any alive member wins.
|
||||
func TestSelectDeadCurrentLosesToAlive(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
g.selectedOutboundTCP = a
|
||||
storeAlive(hist, "a", 10)
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 500)
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v (exists %v), want alive b over dead current a", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectFallsBackToUntestedInConfigOrder(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b, c, d := &balNode{tag: "a"}, &balNode{tag: "b"}, &balNode{tag: "c"}, &balNode{tag: "d"}
|
||||
g := healthTestGroup(hist, nil, a, b, c, d)
|
||||
hist.MarkFailed("a")
|
||||
hist.MarkFailed("b")
|
||||
// c and d untested → first by config order (c), not exists.
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if exists || selected != adapter.Outbound(c) {
|
||||
t.Fatalf("Select = %v (exists %v), want first untested c", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectAllDeadFallsBackToConfigOrder(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
hist.MarkFailed("a")
|
||||
hist.MarkFailed("b")
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if exists || selected != adapter.Outbound(a) {
|
||||
t.Fatalf("Select = %v (exists %v), want first-by-config a", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectStaleSuccessCountsAsUntested(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
storeStale(hist, "a", 20) // success older than TTL → untested tier
|
||||
hist.MarkFailed("b")
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if exists || selected != adapter.Outbound(a) {
|
||||
t.Fatalf("Select = %v (exists %v), want stale a as untested over dead b", selected, exists)
|
||||
}
|
||||
// A fresh-alive member still beats the stale one.
|
||||
c := &balNode{tag: "c"}
|
||||
g.outbounds = append(g.outbounds, c)
|
||||
storeAlive(hist, "c", 300)
|
||||
selected, exists = g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(c) {
|
||||
t.Fatalf("Select = %v (exists %v), want fresh alive c over stale a", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectExcludingSkipsTriedMembers(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 20)
|
||||
selected, _ := g.selectExcluding(N.NetworkTCP, map[string]bool{"a": true})
|
||||
if selected != adapter.Outbound(b) {
|
||||
t.Fatalf("selectExcluding = %v, want b with a excluded", selected)
|
||||
}
|
||||
// Every member excluded → nothing to dial.
|
||||
selected, _ = g.selectExcluding(N.NetworkTCP, map[string]bool{"a": true, "b": true})
|
||||
if selected != nil {
|
||||
t.Fatalf("selectExcluding = %v, want nil with all excluded", selected)
|
||||
}
|
||||
}
|
||||
|
||||
// --- slot liveness reads the board verdict (RR / random / failover) -----------------
|
||||
|
||||
func TestSlotLivenessVerdictOverridesFlag(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
b := rrBalancer(t, 2, []string{"none"})
|
||||
b.verdict = func(tag string) urltest.HealthVerdict { return hist.Verdict(tag, 10*time.Minute) }
|
||||
_, resolve := resolveFrom("a", "b")
|
||||
// Both slots flagged live by the last check, but the board learned "a" died since
|
||||
// (observatory or a failed dial) — pick must skip it the very next connection.
|
||||
b.setSlots([]string{"a", "b"}, allLive("a", "b"))
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 30)
|
||||
for range 10 {
|
||||
picked := b.pick(context.Background(), destDomain("example.com"), nil, resolve)
|
||||
if picked == nil || picked.Tag() != "b" {
|
||||
t.Fatalf("pick = %v, want b (a is board-dead despite live flag)", picked)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlotLivenessVerdictAliveRevives(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
b := rrBalancer(t, 1, []string{"none"})
|
||||
b.verdict = func(tag string) urltest.HealthVerdict { return hist.Verdict(tag, 10*time.Minute) }
|
||||
_, resolve := resolveFrom("a")
|
||||
// Slot flagged dead by the last check, but the board holds a fresh success now:
|
||||
// the verdict wins and the slot is selectable again before the next tick.
|
||||
b.setSlots([]string{"a"}, nil)
|
||||
storeAlive(hist, "a", 20)
|
||||
picked := b.pick(context.Background(), destDomain("example.com"), nil, resolve)
|
||||
if picked == nil || picked.Tag() != "a" {
|
||||
t.Fatalf("pick = %v, want board-alive a despite dead flag", picked)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRandomPickSkipsBoardDeadSlot(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
b := randomBalancer(t, 3)
|
||||
b.verdict = func(tag string) urltest.HealthVerdict { return hist.Verdict(tag, 10*time.Minute) }
|
||||
_, resolve := resolveFrom("a", "b", "c")
|
||||
b.setSlots([]string{"a", "b", "c"}, allLive("a", "b", "c"))
|
||||
hist.MarkFailed("a")
|
||||
for range 50 {
|
||||
picked := b.pick(context.Background(), destDomain("example.com"), nil, resolve)
|
||||
if picked == nil || picked.Tag() == "a" {
|
||||
t.Fatalf("random pick returned board-dead a")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// failover (priority balancer, pool 1): the slot occupant went board-dead → the group
|
||||
// falls back to Select, which routes to a live member, never the dead occupant.
|
||||
func TestFailoverDeadSlotFallsBackToAliveMember(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
manager := managerOf(a, nodeB)
|
||||
g := healthTestGroup(hist, manager, a, nodeB)
|
||||
bal := &balancer{poolSize: 1, priority: true}
|
||||
s := healthURLTest(g, bal, manager)
|
||||
bal.setSlots([]string{"a"}, allLive("a"))
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 40)
|
||||
selected := s.dialSelect(context.Background(), N.NetworkTCP, destDomain("example.com"), nil)
|
||||
if selected == nil || selected.Tag() != "b" {
|
||||
t.Fatalf("dialSelect = %v, want alive fallback b for a dead failover slot", selected)
|
||||
}
|
||||
}
|
||||
|
||||
// --- dial retry: mark-fail + re-pick, ≤ 3 candidates, within deadline ---------------
|
||||
|
||||
func TestDialContextRetriesThroughNextAlive(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
g := healthTestGroup(hist, nil, a, nodeB)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
g.selectedOutboundTCP = a // the checker had picked a; it dies between ticks
|
||||
conn, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("DialContext failed despite live member b: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if len(dialed) != 2 || dialed[0] != "a" || dialed[1] != "b" {
|
||||
t.Fatalf("dial order = %v, want [a b]", dialed)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed dial", v)
|
||||
}
|
||||
if entry := hist.LoadURLTestHistory("a"); entry == nil {
|
||||
t.Fatalf("history entry for a was deleted; mark-fail must keep it")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDialContextStopsAfterThreeCandidates(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
nodes := make([]adapter.Outbound, 0, 4)
|
||||
for i, tag := range []string{"a", "b", "c", "d"} {
|
||||
nodes = append(nodes, &healthNode{tag: tag, fail: true, dialed: &dialed})
|
||||
storeAlive(hist, tag, uint16(10+80*i)) // spread beyond tolerance so order is a,b,c
|
||||
}
|
||||
g := healthTestGroup(hist, nil, nodes...)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
_, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err == nil {
|
||||
t.Fatalf("DialContext succeeded with every member failing")
|
||||
}
|
||||
if len(dialed) != dialAttemptsMax {
|
||||
t.Fatalf("dialed %d members (%v), want at most %d", len(dialed), dialed, dialAttemptsMax)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDialContextStopsWhenContextDone(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
g := healthTestGroup(hist, nil, a, nodeB)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
if _, err := s.DialContext(ctx, N.NetworkTCP, destDomain("example.com")); err == nil {
|
||||
t.Fatalf("DialContext succeeded past a done context")
|
||||
}
|
||||
if len(dialed) != 1 {
|
||||
t.Fatalf("dialed %v, want a single attempt within a done context", dialed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDialContextBalancedDemotesAndRetries(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
manager := managerOf(a, nodeB)
|
||||
g := healthTestGroup(hist, manager, a, nodeB)
|
||||
bal := rrBalancer(t, 2, []string{"none"})
|
||||
s := healthURLTest(g, bal, manager)
|
||||
bal.setSlots([]string{"a", "b"}, allLive("a", "b"))
|
||||
conn, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("balanced DialContext failed despite live member b: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if len(dialed) != 2 || dialed[0] != "a" || dialed[1] != "b" {
|
||||
t.Fatalf("dial order = %v, want [a b] (a demoted, b retried)", dialed)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed dial", v)
|
||||
}
|
||||
// The demotion is visible to the NEXT connection immediately: a is never dialed again.
|
||||
conn, err = s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("second balanced DialContext failed: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if dialed[len(dialed)-1] != "b" {
|
||||
t.Fatalf("dial order = %v, want the second connection to go straight to b", dialed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListenPacketRetriesBeforeFirstSend(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
g := healthTestGroup(hist, nil, a, nodeB)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
g.selectedOutboundUDP = a
|
||||
conn, err := s.ListenPacket(context.Background(), destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("ListenPacket failed despite live member b: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if len(dialed) != 2 || dialed[0] != "a" || dialed[1] != "b" {
|
||||
t.Fatalf("listen order = %v, want [a b]", dialed)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed listen", v)
|
||||
}
|
||||
}
|
||||
|
||||
// --- check failures mark the board, never delete ------------------------------------
|
||||
|
||||
func TestTestNodesMarksFailedKeepsEntry(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
storeAlive(hist, "a", 42)
|
||||
result := g.testNodes(context.Background(), []adapter.Outbound{a}, true)
|
||||
if len(result) != 0 {
|
||||
t.Fatalf("testNodes result = %v, want empty for a failing node", result)
|
||||
}
|
||||
entry := hist.LoadURLTestHistory("a")
|
||||
if entry == nil {
|
||||
t.Fatalf("history entry deleted on check failure; must be marked instead")
|
||||
}
|
||||
if entry.Delay != 42 {
|
||||
t.Fatalf("entry.Delay = %d, want the last-known 42 preserved", entry.Delay)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed check", v)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Pool / seedPool read the verdict, not entry presence ---------------------------
|
||||
|
||||
func TestPoolDelayRequiresAliveVerdict(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a"}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
bal := rrBalancer(t, 1, []string{"none"})
|
||||
s := healthURLTest(g, bal, manager)
|
||||
bal.setSlots([]string{"a"}, allLive("a"))
|
||||
storeAlive(hist, "a", 0) // live sub-ms → clamped to 1
|
||||
pool := s.Pool()
|
||||
if len(pool) != 1 || pool[0].Delay != 1 {
|
||||
t.Fatalf("Pool = %v, want alive slot a with clamped delay 1", pool)
|
||||
}
|
||||
hist.MarkFailed("a") // entry still present, but dead → delay must read 0
|
||||
pool = s.Pool()
|
||||
if len(pool) != 1 || pool[0].Delay != 0 {
|
||||
t.Fatalf("Pool = %v, want dead slot a reported with delay 0", pool)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSeedPoolSkipsBoardDeadNodes(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b, c := &balNode{tag: "a"}, &balNode{tag: "b"}, &balNode{tag: "c"}
|
||||
g := healthTestGroup(hist, nil, a, b, c)
|
||||
bal := rrBalancer(t, 2, []string{"none"})
|
||||
g.balancer = bal
|
||||
// a has an entry (fast, but DEAD); c is fresh-alive. The warm seed must be c, not a.
|
||||
storeAlive(hist, "a", 10)
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "c", 30)
|
||||
g.seedPool()
|
||||
tags := bal.poolTags()
|
||||
if len(tags) != 2 || tags[0] != "c" {
|
||||
t.Fatalf("seeded pool = %v, want the alive c seeded first", tags)
|
||||
}
|
||||
}
|
||||
@@ -16,6 +16,7 @@ import (
|
||||
"github.com/sagernet/sing/common"
|
||||
"github.com/sagernet/sing/common/buf"
|
||||
"github.com/sagernet/sing/common/control"
|
||||
E "github.com/sagernet/sing/common/exceptions" // lx:tproxy_writeback_connect
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
"github.com/sagernet/sing/common/udpnat2"
|
||||
@@ -68,7 +69,26 @@ func (t *TProxy) Start(stage adapter.StartStage) error {
|
||||
}
|
||||
|
||||
func (t *TProxy) Close() error {
|
||||
return t.listener.Close()
|
||||
err := t.listener.Close()
|
||||
// lx:begin tproxy_writeback_connect
|
||||
// Closing the listener stops INGRESS but leaves every live UDP NAT session in
|
||||
// the cache, and each session holds a write-back socket bound to its original
|
||||
// destination. Nothing else ever wakes those sessions: the cache evicts
|
||||
// lazily (on the next Get/Add), and after this inbound is gone there is no
|
||||
// next Get. For a DNS session answered in-engine by hijack-dns there is not
|
||||
// even an outbound connection whose failure could unwind it, so its socket
|
||||
// survives until a GC finalizer happens to reach it.
|
||||
//
|
||||
// That is invisible upstream, where an inbound is closed once at shutdown. On
|
||||
// this fork the engine is REBUILT on every config apply, so each apply would
|
||||
// strand another generation of sockets on the router's own LAN :53. Purge
|
||||
// evicts every session; the cache's OnEvict closes the conn, which unblocks
|
||||
// the session's routing goroutine and runs its onClose — the one place the
|
||||
// write-back socket is actually closed. Purge AFTER the listener so a packet
|
||||
// arriving mid-teardown cannot re-create a session behind us.
|
||||
t.udpNat.Purge()
|
||||
// lx:end tproxy_writeback_connect
|
||||
return err
|
||||
}
|
||||
|
||||
func (t *TProxy) NewConnection(ctx context.Context, conn net.Conn, metadata adapter.InboundContext, onClose N.CloseHandlerFunc) {
|
||||
@@ -121,18 +141,92 @@ type tproxyPacketWriter struct {
|
||||
conn *net.UDPConn
|
||||
}
|
||||
|
||||
// lx:begin tproxy_writeback_connect
|
||||
//
|
||||
// The TPROXY UDP write-back socket is bound to the ORIGINAL DESTINATION, so the
|
||||
// client sees the reply coming from the address it addressed. Upstream leaves
|
||||
// that socket UNCONNECTED (net.ListenPacket + WriteToUDPAddrPort), and that is
|
||||
// the bug this block exists for.
|
||||
//
|
||||
// An unconnected socket bound to <addr>:<port> is, as far as the kernel is
|
||||
// concerned, a RECEIVER for that address:port — and because the socket also
|
||||
// carries SO_REUSEADDR it silently joins the UDP demultiplex set of whatever
|
||||
// else is bound there. Nothing ever reads from it: this writer only sends. So
|
||||
// every datagram the kernel happens to hand it is lost.
|
||||
//
|
||||
// On a router that transparently intercepts LAN DNS towards its OWN address
|
||||
// (`nft ... udp dport 53 tproxy ...`) the original destination IS the router's
|
||||
// LAN address, so each intercepted DNS session parks another silent receiver on
|
||||
// <router-lan-ip>:53 right next to dnsmasq's socket — observed on the stand:
|
||||
// 33 such sockets against dnsmasq's one, several with a growing Recv-Q. The
|
||||
// host's own queries to that address take the loopback path, are never diverted,
|
||||
// and are therefore demultiplexed among all of them: they land in one of the
|
||||
// silent sockets at random and time out, permanently and unpredictably, while
|
||||
// every other DNS path on the box keeps working.
|
||||
//
|
||||
// Connecting the socket fixes it at the root. compute_score() in the kernel's
|
||||
// UDP lookup REJECTS a connected socket for any peer other than the connected
|
||||
// one, so a write-back socket can no longer be handed a datagram it will not
|
||||
// read. Nothing about the reply changes — same spoofed source address, same
|
||||
// single peer, one Write instead of one WriteTo.
|
||||
//
|
||||
// The unconnected path is kept verbatim for a destination that cannot be bound
|
||||
// (a domain socksaddr), so no existing case regresses.
|
||||
|
||||
// newTProxyWriteBack creates the connected write-back socket: local address =
|
||||
// the original destination (transparent bind), peer = the client. A package
|
||||
// variable so the behaviour can be tested without CAP_NET_ADMIN.
|
||||
var newTProxyWriteBack = func(w *tproxyPacketWriter, destination M.Socksaddr) (*net.UDPConn, error) {
|
||||
var dialer net.Dialer
|
||||
dialer.LocalAddr = destination.UDPAddr()
|
||||
dialer.Control = control.Append(dialer.Control, control.ReuseAddr())
|
||||
dialer.Control = control.Append(dialer.Control, redir.TProxyWriteBack())
|
||||
conn, err := w.listener.DialContext(dialer, w.ctx, "udp", w.source.String())
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
udpConn, loaded := conn.(*net.UDPConn)
|
||||
if !loaded {
|
||||
conn.Close()
|
||||
return nil, E.New("tproxy write back: unexpected connection type ", conn)
|
||||
}
|
||||
return udpConn, nil
|
||||
}
|
||||
|
||||
// lx:end tproxy_writeback_connect
|
||||
|
||||
func (w *tproxyPacketWriter) WritePacket(buffer *buf.Buffer, destination M.Socksaddr) error {
|
||||
defer buffer.Release()
|
||||
if w.listener.ListenOptions().NetNs == "" {
|
||||
conn := w.conn
|
||||
if w.destination == destination && conn != nil {
|
||||
_, err := conn.WriteToUDPAddrPort(buffer.Bytes(), w.source)
|
||||
// lx:begin tproxy_writeback_connect
|
||||
// The cached socket is CONNECTED to the client, so this is a plain
|
||||
// Write. A failed write also closes it: upstream only dropped the
|
||||
// reference, leaving the fd to the GC finalizer.
|
||||
_, err := conn.Write(buffer.Bytes())
|
||||
if err != nil {
|
||||
conn.Close()
|
||||
w.conn = nil
|
||||
}
|
||||
return err
|
||||
// lx:end tproxy_writeback_connect
|
||||
}
|
||||
}
|
||||
// lx:begin tproxy_writeback_connect
|
||||
if destination.IsIP() {
|
||||
udpConn, err := newTProxyWriteBack(w, destination)
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
if w.listener.ListenOptions().NetNs == "" && w.destination == destination {
|
||||
w.conn = udpConn
|
||||
} else {
|
||||
defer udpConn.Close()
|
||||
}
|
||||
return common.Error(udpConn.Write(buffer.Bytes()))
|
||||
}
|
||||
// lx:end tproxy_writeback_connect
|
||||
var listenConfig net.ListenConfig
|
||||
listenConfig.Control = control.Append(listenConfig.Control, control.ReuseAddr())
|
||||
listenConfig.Control = control.Append(listenConfig.Control, redir.TProxyWriteBack())
|
||||
|
||||
@@ -0,0 +1,234 @@
|
||||
package redirect
|
||||
|
||||
// Regression cover for the lx:tproxy_writeback_connect block in tproxy.go.
|
||||
//
|
||||
// The failure it guards against, seen on a live BPi-R3 Mini: `shaterd` held 33
|
||||
// UNCONNECTED UDP sockets on the router's own LAN address :53 — the write-back
|
||||
// sockets of intercepted DNS sessions — alongside dnsmasq's single socket on the
|
||||
// same address:port. Nothing reads a write-back socket, so every host-originated
|
||||
// query that the kernel demultiplexed into one of them was silently dropped
|
||||
// (Recv-Q climbing, sender timing out). Connecting the socket to the one peer it
|
||||
// ever talks to removes it from the demultiplex set for everybody else.
|
||||
//
|
||||
// The real socket needs CAP_NET_ADMIN (IP_TRANSPARENT) and a foreign bind, so
|
||||
// these tests drive the seam (newTProxyWriteBack) with an ordinary connected
|
||||
// loopback socket. That is enough to pin both properties that actually broke:
|
||||
//
|
||||
// - the write path uses the CONNECTED form. If WritePacket ever goes back to
|
||||
// WriteToUDPAddrPort, Go returns ErrWriteToConnected on a connected socket
|
||||
// and these tests fail — i.e. the test cannot pass with an unconnected
|
||||
// write-back socket, which is exactly the regression.
|
||||
// - repeated writes to the same destination REUSE one socket instead of
|
||||
// accumulating a new one per packet.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"net"
|
||||
"net/netip"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/common/listener"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing/common/buf"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
"github.com/sagernet/sing/common/udpnat2"
|
||||
)
|
||||
|
||||
// writeBackHarness stands up a loopback "client" socket and a tproxyPacketWriter
|
||||
// whose socket factory returns a plain connected UDP socket aimed at it. It
|
||||
// returns the writer, the client socket and a pointer to the factory call count.
|
||||
func writeBackHarness(t *testing.T) (*tproxyPacketWriter, *net.UDPConn, *int) {
|
||||
t.Helper()
|
||||
|
||||
client, err := net.ListenUDP("udp", &net.UDPAddr{IP: net.IPv4(127, 0, 0, 1)})
|
||||
if err != nil {
|
||||
t.Fatalf("client socket: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { client.Close() })
|
||||
|
||||
source := netip.MustParseAddrPort(client.LocalAddr().String())
|
||||
w := &tproxyPacketWriter{
|
||||
ctx: context.Background(),
|
||||
source: source,
|
||||
// A listener with empty options is enough: WritePacket only reads
|
||||
// ListenOptions().NetNs, and the factory is stubbed below.
|
||||
listener: listener.New(listener.Options{
|
||||
Context: context.Background(),
|
||||
Listen: option.ListenOptions{},
|
||||
}),
|
||||
destination: M.SocksaddrFrom(netip.MustParseAddr("127.0.0.1"), 53),
|
||||
}
|
||||
|
||||
calls := 0
|
||||
orig := newTProxyWriteBack
|
||||
t.Cleanup(func() { newTProxyWriteBack = orig })
|
||||
newTProxyWriteBack = func(w *tproxyPacketWriter, destination M.Socksaddr) (*net.UDPConn, error) {
|
||||
calls++
|
||||
// The real implementation binds the ORIGINAL DESTINATION transparently
|
||||
// and connects to w.source; here we only reproduce the connect half,
|
||||
// which is the property under test.
|
||||
conn, err := net.DialUDP("udp", nil, net.UDPAddrFromAddrPort(w.source))
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return conn, nil
|
||||
}
|
||||
return w, client, &calls
|
||||
}
|
||||
|
||||
// readOne reads one datagram from the client socket with a short deadline.
|
||||
func readOne(t *testing.T, client *net.UDPConn) string {
|
||||
t.Helper()
|
||||
_ = client.SetReadDeadline(time.Now().Add(2 * time.Second))
|
||||
b := make([]byte, 512)
|
||||
n, _, err := client.ReadFrom(b)
|
||||
if err != nil {
|
||||
t.Fatalf("client read: %v", err)
|
||||
}
|
||||
return string(b[:n])
|
||||
}
|
||||
|
||||
// TestWriteBackUsesConnectedSocket: the reply must go out over a CONNECTED
|
||||
// socket. On the pre-fix code the write is WriteToUDPAddrPort, which Go refuses
|
||||
// on a connected socket ("use of WriteTo with pre-connected connection"), so
|
||||
// this test is red for exactly the shape that caused the outage.
|
||||
func TestWriteBackUsesConnectedSocket(t *testing.T) {
|
||||
w, client, calls := writeBackHarness(t)
|
||||
|
||||
if err := w.WritePacket(buf.As([]byte("first")).ToOwned(), w.destination); err != nil {
|
||||
t.Fatalf("WritePacket: %v", err)
|
||||
}
|
||||
if got := readOne(t, client); got != "first" {
|
||||
t.Fatalf("client got %q, want %q", got, "first")
|
||||
}
|
||||
if *calls != 1 {
|
||||
t.Fatalf("expected one write-back socket, got %d", *calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWriteBackReusesOneSocket: a session that keeps answering the same client
|
||||
// must keep ONE socket, not open a fresh one per datagram. The live box carried
|
||||
// one socket per intercepted DNS session already; one per PACKET would turn a
|
||||
// nuisance into an fd exhaustion.
|
||||
func TestWriteBackReusesOneSocket(t *testing.T) {
|
||||
w, client, calls := writeBackHarness(t)
|
||||
|
||||
for i, payload := range []string{"a", "b", "c", "d"} {
|
||||
if err := w.WritePacket(buf.As([]byte(payload)).ToOwned(), w.destination); err != nil {
|
||||
t.Fatalf("WritePacket %d: %v", i, err)
|
||||
}
|
||||
if got := readOne(t, client); got != payload {
|
||||
t.Fatalf("packet %d: client got %q, want %q", i, got, payload)
|
||||
}
|
||||
}
|
||||
if *calls != 1 {
|
||||
t.Fatalf("four packets to one destination must share one socket, got %d sockets", *calls)
|
||||
}
|
||||
if w.conn == nil {
|
||||
t.Fatalf("the write-back socket must be cached on the writer for reuse")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWriteBackClosesSocketOnWriteFailure: upstream dropped the reference to a
|
||||
// failed socket without closing it, leaving the fd to the GC finalizer. On a box
|
||||
// that already parks one socket per DNS session that is the wrong direction.
|
||||
func TestWriteBackClosesSocketOnWriteFailure(t *testing.T) {
|
||||
w, client, _ := writeBackHarness(t)
|
||||
|
||||
if err := w.WritePacket(buf.As([]byte("warm")).ToOwned(), w.destination); err != nil {
|
||||
t.Fatalf("WritePacket: %v", err)
|
||||
}
|
||||
readOne(t, client)
|
||||
|
||||
cached := w.conn
|
||||
if cached == nil {
|
||||
t.Fatalf("expected a cached socket after the first write")
|
||||
}
|
||||
// Close it behind WritePacket's back so the next write fails, exactly as a
|
||||
// dead peer or a torn-down plane would make it fail.
|
||||
cached.Close()
|
||||
|
||||
if err := w.WritePacket(buf.As([]byte("boom")).ToOwned(), w.destination); err == nil {
|
||||
t.Fatalf("a write on a closed socket must report the failure")
|
||||
}
|
||||
if w.conn != nil {
|
||||
t.Fatalf("a failed write must drop the cached socket")
|
||||
}
|
||||
// A second Close on an already-closed conn is an error, which is how we know
|
||||
// WritePacket closed it rather than merely forgetting it.
|
||||
if err := cached.Close(); err == nil {
|
||||
t.Fatalf("WritePacket must CLOSE the failed socket, not just nil the field")
|
||||
}
|
||||
}
|
||||
|
||||
// --- Close() must release the UDP NAT sessions --------------------------------
|
||||
|
||||
// natSessionHandler stands in for the router: it drains the session conn until
|
||||
// it errors (which is what Close does to it) and then reports onClose, exactly
|
||||
// as the real routing goroutine does. onClose is where the write-back socket is
|
||||
// closed, so "onClose fired" is the observable proof the socket was released.
|
||||
type natSessionHandler struct {
|
||||
released chan error
|
||||
}
|
||||
|
||||
func (h *natSessionHandler) NewPacketConnectionEx(ctx context.Context, conn N.PacketConn, source M.Socksaddr, destination M.Socksaddr, onClose N.CloseHandlerFunc) {
|
||||
go func() {
|
||||
for {
|
||||
buffer := buf.NewSize(1024)
|
||||
_, err := conn.ReadPacket(buffer)
|
||||
buffer.Release()
|
||||
if err != nil {
|
||||
if onClose != nil {
|
||||
onClose(err)
|
||||
}
|
||||
h.released <- err
|
||||
return
|
||||
}
|
||||
}
|
||||
}()
|
||||
}
|
||||
|
||||
// TestTProxyCloseReleasesNatSessions pins the apply-level half of the leak.
|
||||
//
|
||||
// Closing the inbound used to close the listener only. The NAT cache evicts
|
||||
// lazily, so after the inbound is gone nothing ever touches it again and every
|
||||
// live session — with the write-back socket it holds on the router's own
|
||||
// LAN :53 — was left to a GC finalizer. On this fork the engine is rebuilt on
|
||||
// every config apply, so that is one stranded generation of sockets per apply.
|
||||
func TestTProxyCloseReleasesNatSessions(t *testing.T) {
|
||||
handler := &natSessionHandler{released: make(chan error, 1)}
|
||||
tp := &TProxy{
|
||||
ctx: context.Background(),
|
||||
listener: listener.New(listener.Options{
|
||||
Context: context.Background(),
|
||||
Listen: option.ListenOptions{},
|
||||
}),
|
||||
}
|
||||
prepared := 0
|
||||
tp.udpNat = udpnat.New(handler, func(source M.Socksaddr, destination M.Socksaddr, userData any) (bool, context.Context, N.PacketWriter, N.CloseHandlerFunc) {
|
||||
prepared++
|
||||
return true, context.Background(), nil, func(error) {}
|
||||
}, time.Minute, false)
|
||||
|
||||
tp.udpNat.NewPacket(
|
||||
[][]byte{{0x00}},
|
||||
M.SocksaddrFrom(netip.MustParseAddr("10.67.0.2"), 40000),
|
||||
M.SocksaddrFrom(netip.MustParseAddr("10.67.0.1"), 53),
|
||||
nil,
|
||||
)
|
||||
if prepared != 1 {
|
||||
t.Fatalf("expected one NAT session, got %d", prepared)
|
||||
}
|
||||
|
||||
if err := tp.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
select {
|
||||
case <-handler.released:
|
||||
case <-time.After(5 * time.Second):
|
||||
t.Fatal("Close must release every live NAT session (and with it the write-back socket it holds)")
|
||||
}
|
||||
}
|
||||
@@ -159,7 +159,10 @@ func (r *Router) startIdleSuspend() error {
|
||||
if period < idleTickFloor {
|
||||
period = idleTickFloor
|
||||
}
|
||||
go r.idleSuspendLoop(period)
|
||||
// Pass the stop channel by value so the loop never reads the r.idleStop field
|
||||
// (stopIdleSuspend nils it right after close): a closed local channel stays
|
||||
// ready in select, so the loop exits even when Close races an in-flight tick.
|
||||
go r.idleSuspendLoop(period, r.idleStop)
|
||||
return nil
|
||||
}
|
||||
|
||||
@@ -176,12 +179,12 @@ func (r *Router) stopIdleSuspend() {
|
||||
// asks each WG/AWG endpoint to suspend itself if it is unreachable AND idle past
|
||||
// the threshold. Per endpoint this is a single map lookup + an atomic idle
|
||||
// comparison; the endpoint owns the suspend/CAS/log decision.
|
||||
func (r *Router) idleSuspendLoop(period time.Duration) {
|
||||
func (r *Router) idleSuspendLoop(period time.Duration, stop <-chan struct{}) {
|
||||
ticker := time.NewTicker(period)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-r.idleStop:
|
||||
case <-stop:
|
||||
return
|
||||
case <-ticker.C:
|
||||
r.suspendIdleEndpoints(r.reachableOutbounds())
|
||||
|
||||
@@ -27,8 +27,11 @@
|
||||
#
|
||||
# VERSION Version string stamped into constant.Version. Resolution order:
|
||||
# 1) this positional arg, if given
|
||||
# 2) $SHATER_VERSION, if set
|
||||
# 3) `git describe --tags` (nearest tag + commit)
|
||||
# 2) $SHATER_VERSION, if set (CI sets it from ci/version.sh)
|
||||
# 3) `ci/version.sh --binary` — THE single source of truth shared
|
||||
# with the package version (vX.Y.Z-rR[-g<sha>], derived from
|
||||
# the git tag exactly like PKG_VERSION/PKG_RELEASE), so the
|
||||
# string the panel shows always matches `apk info shaterd`
|
||||
# 4) fallback: v0.2.0-dev
|
||||
# --fast Skip `npm ci` when panel/node_modules already exists (dev speed-up).
|
||||
#
|
||||
@@ -69,8 +72,10 @@ if [ -n "$VERSION_ARG" ]; then
|
||||
VERSION="$VERSION_ARG"
|
||||
elif [ -n "${SHATER_VERSION:-}" ]; then
|
||||
VERSION="$SHATER_VERSION"
|
||||
elif VERSION="$(git -C "$REPO" describe --tags 2>/dev/null)"; then
|
||||
: # git describe succeeded
|
||||
elif VERSION="$(sh "$REPO/ci/version.sh" --binary)"; then
|
||||
# Same computation the PACKAGE version comes from (ci/version.sh), so the
|
||||
# binary's constant.Version and the .ipk/.apk version can never drift apart.
|
||||
: # ci/version.sh always succeeds (it falls back to 0.0.0 without git)
|
||||
else
|
||||
VERSION="v0.2.0-dev"
|
||||
fi
|
||||
|
||||
@@ -163,10 +163,15 @@ func (t *adaptiveTimer) poll() {
|
||||
return
|
||||
}
|
||||
if t.timerConfig.policyMode == policyModeNetworkExtension {
|
||||
// lx:begin sec-oomcleanup
|
||||
// A trigger schedules a deferred FreeOSMemory for the next poll; run it
|
||||
// here and clear the flag (was inverted: it re-set true and so the free
|
||||
// after a trigger never happened).
|
||||
if t.cleanupTriggered {
|
||||
runtimeDebug.FreeOSMemory()
|
||||
t.cleanupTriggered = true
|
||||
t.cleanupTriggered = false
|
||||
}
|
||||
// lx:end sec-oomcleanup
|
||||
}
|
||||
if t.pendingPressureBaseline {
|
||||
t.pressureBaseline = sample
|
||||
@@ -202,7 +207,10 @@ func (t *adaptiveTimer) poll() {
|
||||
if !triggered {
|
||||
return
|
||||
}
|
||||
t.cleanupTriggered = false
|
||||
// lx:begin sec-oomcleanup
|
||||
// Schedule the deferred FreeOSMemory for the next network-extension poll.
|
||||
t.cleanupTriggered = true
|
||||
// lx:end sec-oomcleanup
|
||||
t.onTriggered(sample.usage)
|
||||
if rateTriggered {
|
||||
if t.killerDisabled {
|
||||
|
||||
+101
-120
@@ -31,6 +31,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/generate"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
@@ -168,106 +169,34 @@ func (a *Applier) UpdateRuleSet(tag string) error {
|
||||
return a.eng.UpdateRuleSet(tag)
|
||||
}
|
||||
|
||||
// TestAllNodes launches the engine's one-shot probe of every node (manual "Test
|
||||
// all nodes", feedback #3), returning started=false when a run is already in flight
|
||||
// or the engine is absent. It delegates straight to the engine and takes NEITHER
|
||||
// the apply mutex nor the flock, so kicking off a test never blocks behind an Apply.
|
||||
// configureObservatory (re)installs the engine's background observatory from the
|
||||
// model and the JUST-GENERATED options. It is called on every SUCCESSFUL apply,
|
||||
// which is exactly when the facts it depends on — the applied option set (the
|
||||
// reachability plan's source), the global probe URL/interval and the GroupHealth
|
||||
// master switch — can have changed.
|
||||
//
|
||||
// It reads the model itself to resolve each group's probe URL and egress binding: a
|
||||
// manual run must measure every tag with the URL of the group that owns it, or it
|
||||
// overwrites that group's entries with numbers taken against a different server (see
|
||||
// engine/probeplan.go). A UCI read failure degrades to "no group opinions, global
|
||||
// default URL" so a test can always be kicked off.
|
||||
func (a *Applier) TestAllNodes() (started bool) {
|
||||
if a.eng == nil {
|
||||
return false
|
||||
}
|
||||
specs, fallback := a.probeSpecs()
|
||||
return a.eng.TestAllNodes(specs, fallback)
|
||||
}
|
||||
|
||||
// probeSpecs resolves the per-group probe facts (egress binding + effective probe URL)
|
||||
// and the global fallback URL from UCI. A read failure yields (nil, "") — every tag is
|
||||
// then measured with urltest's built-in default, which is what the generator would have
|
||||
// used anyway when nothing is configured.
|
||||
func (a *Applier) probeSpecs() (map[string]engine.GroupProbeSpec, string) {
|
||||
m, err := model.ReadUCI()
|
||||
if err != nil || m == nil {
|
||||
return nil, ""
|
||||
}
|
||||
return groupProbeSpecs(m), strings.TrimSpace(m.Globals.ProbeURL)
|
||||
}
|
||||
|
||||
// groupProbeSpecs mirrors generate.groupProbeURL's precedence — the group's own probe
|
||||
// URL, else globals' — for every group in the model, alongside its egress binding.
|
||||
//
|
||||
// It deliberately stops at "" rather than substituting the gstatic default: "" means
|
||||
// "this group has no opinion", which is what lets the planner tell a group that shares
|
||||
// the global instrument from one that chose its own (only the latter can create the
|
||||
// ambiguity resolveProbeURL has to arbitrate).
|
||||
func groupProbeSpecs(m *model.Model) map[string]engine.GroupProbeSpec {
|
||||
if m == nil || len(m.Groups) == 0 {
|
||||
return nil
|
||||
}
|
||||
global := strings.TrimSpace(m.Globals.ProbeURL)
|
||||
out := make(map[string]engine.GroupProbeSpec, len(m.Groups))
|
||||
for _, g := range m.Groups {
|
||||
url := strings.TrimSpace(g.ProbeURL)
|
||||
if url == "" {
|
||||
url = global
|
||||
}
|
||||
out[g.Name] = engine.GroupProbeSpec{
|
||||
Egress: strings.TrimSpace(g.Egress),
|
||||
ProbeURL: url,
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// configureSweep (re)installs the engine's scheduled health sweep from the model. It is
|
||||
// called on every SUCCESSFUL apply, which is exactly when the facts it depends on — the
|
||||
// group set, their egress bindings, their probe URLs and the schedule itself — can have
|
||||
// changed.
|
||||
//
|
||||
// The schedule comes from Globals.SweepInterval via model.SweepSchedule, the same
|
||||
// function ValidateGlobals warns from, so what the operator is told and what the engine
|
||||
// does cannot diverge. An UNSET interval means ON at the engine's default tick: urltest
|
||||
// groups probe only while in use and selector groups never probe at all, so with the
|
||||
// sweep off a 376-node subscription reads almost entirely "untested". A zero Interval is
|
||||
// passed through deliberately — it tells engine.SweepConfig.withDefaults to apply its
|
||||
// own default, keeping that number in one place. See engine/sweep.go for the cost model.
|
||||
func (a *Applier) configureSweep(m *model.Model) {
|
||||
// The engine keeps its walk position when the rebuilt plan is identical
|
||||
// (engine.ConfigureObservatory), so the every-minute cron reconcile cannot
|
||||
// restart the cycle. GroupHealth=false — the operator's master switch — disables
|
||||
// background probing entirely.
|
||||
func (a *Applier) configureObservatory(m *model.Model, opts option.Options) {
|
||||
if a.eng == nil {
|
||||
return
|
||||
}
|
||||
interval, enabled, _ := m.Globals.SweepSchedule()
|
||||
// GroupHealth is the operator's master-switch for our sweep. When off, disable it
|
||||
// exactly like SweepInterval="off" — no warning, this is an explicit choice — while
|
||||
// PRESERVING the interval (so flipping the toggle back on restores the schedule).
|
||||
if !m.Globals.GroupHealth {
|
||||
enabled = false
|
||||
}
|
||||
a.eng.ConfigureSweep(engine.SweepConfig{
|
||||
Enabled: enabled,
|
||||
Interval: interval,
|
||||
Specs: groupProbeSpecs(m),
|
||||
FallbackURL: strings.TrimSpace(m.Globals.ProbeURL),
|
||||
interval, _ := model.ParseDuration(m.Globals.ProbeInterval)
|
||||
a.eng.ConfigureObservatory(engine.ObservatoryConfig{
|
||||
Enabled: m.Globals.GroupHealth,
|
||||
Options: opts,
|
||||
ProbeURL: strings.TrimSpace(m.Globals.ProbeURL),
|
||||
ProbeInterval: interval,
|
||||
})
|
||||
}
|
||||
|
||||
// NodeTestStatus reports the engine's probe-all progress (running + done/total). A
|
||||
// nil engine reads as idle. Lock-free like the engine method, safe to poll hot.
|
||||
func (a *Applier) NodeTestStatus() (running bool, done, total int) {
|
||||
if a.eng == nil {
|
||||
return false, 0, 0
|
||||
}
|
||||
return a.eng.NodeTestStatus()
|
||||
}
|
||||
|
||||
// TestGroups launches the engine's one-shot per-group test (delay + exit address,
|
||||
// F2), returning started=false when a run is already in flight or the engine is
|
||||
// absent. names empty/nil = every group. Like TestAllNodes it takes NEITHER the
|
||||
// apply mutex nor the flock, so kicking off a test never blocks behind an Apply.
|
||||
// TestGroups launches the engine's one-shot exit test (delay + exit address, F2)
|
||||
// of the named groups/chains, returning started=false when a run is already in
|
||||
// flight or the engine is absent. names empty/nil = every group and every chain.
|
||||
// It takes NEITHER the apply mutex nor the flock, so kicking off a test never
|
||||
// blocks behind an Apply.
|
||||
func (a *Applier) TestGroups(names []string, probeURL string) (started bool) {
|
||||
if a.eng == nil {
|
||||
return false
|
||||
@@ -275,8 +204,9 @@ func (a *Applier) TestGroups(names []string, probeURL string) (started bool) {
|
||||
return a.eng.TestGroups(names, probeURL)
|
||||
}
|
||||
|
||||
// GroupTestStatus reports the engine's group-test progress, the set of groups the run
|
||||
// covers, and its results. A nil engine reads as idle with empty (never nil) slices, so
|
||||
// GroupTestStatus reports the engine's exit-test progress, the set of targets
|
||||
// (groups and chains) the run covers, and its results. A nil engine reads as idle
|
||||
// with empty (never nil) slices, so
|
||||
// the panel can render it unconditionally.
|
||||
func (a *Applier) GroupTestStatus() (running bool, done, total int, scope []string, results []engine.GroupTestResult) {
|
||||
if a.eng == nil {
|
||||
@@ -306,6 +236,41 @@ func (a *Applier) GroupHealthOne(name string) (engine.GroupHealth, bool) {
|
||||
return a.eng.GroupHealthOne(name)
|
||||
}
|
||||
|
||||
// ChainHealth reports each configured chain's reachability (used/unused) for the
|
||||
// Targets page's "unused" badge (plan §5.E) — the chain analogue of GroupHealth.Used.
|
||||
// names come from the desired-state model (model.ReadUCI — the same source GET
|
||||
// /api/config lists, so every chain the panel shows gets a row, including one no
|
||||
// rule references and that the running box therefore never materialised); the engine
|
||||
// supplies the observatory's published used-set. A nil engine reads as empty.
|
||||
//
|
||||
// Takes NEITHER the apply mutex nor the flock, so it is safe to poll while an apply
|
||||
// is in flight — exactly like GroupHealth. A failed UCI read (no `uci` binary, e.g.
|
||||
// off-router) yields an empty slice rather than an error, so the endpoint stays a
|
||||
// pure read.
|
||||
func (a *Applier) ChainHealth() []engine.ChainHealth {
|
||||
if a.eng == nil {
|
||||
return []engine.ChainHealth{}
|
||||
}
|
||||
return a.eng.ChainHealth(chainNamesFromModel())
|
||||
}
|
||||
|
||||
// chainNamesFromModel reads the desired-state model best-effort and returns the
|
||||
// configured chain names in model order. A failed read (off-router, no `uci`) yields
|
||||
// nil — the engine then reports an empty chain list, never an error.
|
||||
func chainNamesFromModel() []string {
|
||||
m, err := model.ReadUCI()
|
||||
if err != nil || m == nil {
|
||||
return nil
|
||||
}
|
||||
names := make([]string, 0, len(m.Chains))
|
||||
for _, c := range m.Chains {
|
||||
if name := strings.TrimSpace(c.Name); name != "" {
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// HTTPClient returns an http.Client that dials THROUGH the running engine's
|
||||
// outbound named by `via` (feedback #1/#8). It delegates to the engine; a stopped
|
||||
// engine yields engine.ErrEngineStopped and an unknown tag engine.ErrOutboundUnknown.
|
||||
@@ -317,12 +282,14 @@ func (a *Applier) HTTPClient(via string) (*http.Client, error) {
|
||||
}
|
||||
|
||||
// UpdateSubscription fetches the named subscription, folds the parsed nodes into
|
||||
// UCI (replacing exactly that sub's cache), and reconciles so the running engine
|
||||
// picks them up (feedback #8). When the subscription's FetchVia=="proxy" the fetch
|
||||
// is routed THROUGH the engine outbound named by its FetchDetour (group:/node:/
|
||||
// egress:/direct); otherwise the fetch is DIRECT. Zero-node safety is inherited
|
||||
// from subscribe.UpdateSubscription (a bad body leaves the cache untouched); UCI is
|
||||
// only written on a successful parse. Returns the node count now cached.
|
||||
// the model (replacing exactly that sub's cache), persists them to the sub's
|
||||
// JSON cache file (model.SaveSubCache) + the userinfo counters to UCI, and
|
||||
// reconciles so the running engine picks them up (feedback #8). When the
|
||||
// subscription's FetchVia=="proxy" the fetch is routed THROUGH the engine
|
||||
// outbound named by its FetchDetour (group:/node:/egress:/direct); otherwise the
|
||||
// fetch is DIRECT. Zero-node safety is inherited from subscribe.UpdateSubscription
|
||||
// (a bad body leaves the cache untouched); nothing is written on a failed parse.
|
||||
// Returns the node count now cached.
|
||||
//
|
||||
// It does NOT take the apply mutex itself — the fetch/parse run lock-free and the
|
||||
// final Reconcile takes the flock + mutex like any apply.
|
||||
@@ -373,14 +340,20 @@ func (a *Applier) UpdateSubscription(name string) (added int, err error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
// Fold the account state into the model BEFORE WriteUCI, so quota/expiry land on
|
||||
// disk in the same write as the refreshed nodes. Failing to record statistics
|
||||
// must never fail a subscription update that actually succeeded — the nodes are
|
||||
// the payload, the counters are a bonus — so the error is logged, not returned.
|
||||
// Fold the account state in BEFORE the writes, so quota/expiry land in the
|
||||
// same pass as the refreshed nodes. Failing to record statistics must never
|
||||
// fail a subscription update that actually succeeded — the nodes are the
|
||||
// payload, the counters are a bonus — so the error is logged, not returned.
|
||||
// (It can only fire on an unknown subscription name, which we just resolved.)
|
||||
if _, serr := subscribe.StoreUserInfo(m, name, info, time.Now()); serr != nil {
|
||||
a.log.Warn("subscription ", name, ": store user info: ", serr)
|
||||
}
|
||||
// Persistence is SPLIT: the refreshed node set goes to this subscription's
|
||||
// JSON cache file; UCI only takes the userinfo counters (RenderUCIExport does
|
||||
// not emit FromSub nodes), so a refresh never rewrites the whole config.
|
||||
if err = model.SaveSubCache(name, m.NodesFromSub(name)); err != nil {
|
||||
return 0, fmt.Errorf("write subscription cache: %w", err)
|
||||
}
|
||||
if err = model.WriteUCI(m); err != nil {
|
||||
return 0, fmt.Errorf("write UCI: %w", err)
|
||||
}
|
||||
@@ -515,6 +488,23 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
return changed, err
|
||||
}
|
||||
|
||||
// The plane is COMPLETE only here: table + policy routing + sysctls. ApplyNft
|
||||
// already flushed the DNS conntrack when it loaded the ruleset, but that is
|
||||
// one step too early — ApplyRouting is idempotent BY del-then-add, so it opens
|
||||
// a window in which the fwmark rule is momentarily absent, and any DNS flow
|
||||
// that crosses that window is tracked against a plane that is still being
|
||||
// assembled. Flushing once more now that every piece is in place is what makes
|
||||
// "no entry survives the transition" actually true. Only on a real change (the
|
||||
// fast path assembled nothing), best-effort, and cheap: the :53 entry count is
|
||||
// bounded by the number of clients.
|
||||
if !nftCurrent {
|
||||
if n, ferr := netplane.FlushDNSConntrack(); ferr != nil {
|
||||
a.log.Debug("flush DNS conntrack after plane change: ", ferr)
|
||||
} else if n > 0 {
|
||||
a.log.Debug("plane changed: dropped ", n, " stale DNS conntrack entries")
|
||||
}
|
||||
}
|
||||
|
||||
// (4) success. Bump the effective-state generation ONLY when something really
|
||||
// moved: a no-op reconcile must not invalidate an armed commit-confirm window
|
||||
// (cron reconciles every minute — counting those would cancel every rollback
|
||||
@@ -538,10 +528,10 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
// hotplug event) is what buries a real warning under a thousand identical
|
||||
// lines a day and evicts incident history from the in-memory ring buffer.
|
||||
a.logWarningsIfChanged(ws)
|
||||
// (Re)arm the scheduled health sweep against the config that is now running: the
|
||||
// group set, their egress bindings and their probe URLs are exactly what the sweep
|
||||
// plan is built from, and this is the only place they can change.
|
||||
a.configureSweep(m)
|
||||
// (Re)configure the observatory against the config that is now running: the
|
||||
// applied options are exactly what its reachability plan is built from, and
|
||||
// this is the only place they can change.
|
||||
a.configureObservatory(m, opts)
|
||||
raiseActiveFlag(a.log)
|
||||
return changed, nil
|
||||
}
|
||||
@@ -697,9 +687,9 @@ func (a *Applier) Teardown() error {
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
|
||||
// Stop the background sweep BEFORE closing the engine, and wait for an in-flight
|
||||
// Stop the observatory BEFORE closing the engine, and wait for an in-flight
|
||||
// tick: a teardown must not leave probe dials racing a box that is going away.
|
||||
a.eng.StopSweep()
|
||||
a.eng.StopObservatory()
|
||||
|
||||
var firstErr error
|
||||
if err := a.eng.Close(); err != nil {
|
||||
@@ -1122,12 +1112,3 @@ func effectivePanelPort(configured int) int {
|
||||
|
||||
// MarshalJSON is a convenience so callers can json.Marshal a Status directly.
|
||||
func (s Status) JSON() ([]byte, error) { return json.MarshalIndent(s, "", " ") }
|
||||
|
||||
// SweepStatus reports the engine's scheduled health-sweep progress (see
|
||||
// engine/sweep.go). A nil engine reads as disabled and idle.
|
||||
func (a *Applier) SweepStatus() (enabled bool, cursor, total int, cycles uint64) {
|
||||
if a.eng == nil {
|
||||
return false, 0, 0, 0
|
||||
}
|
||||
return a.eng.SweepStatus()
|
||||
}
|
||||
|
||||
+18
-116
@@ -2,132 +2,34 @@ package apply
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// groupProbeSpecs must mirror generate.groupProbeURL's precedence — the group's own
|
||||
// probe URL, else globals' — because that is the URL the group ACTUALLY measures its
|
||||
// members with. If the two ever drift, a manual run or the scheduled sweep starts
|
||||
// overwriting a group's entries with numbers taken against a different server, which
|
||||
// is precisely what the probe plan exists to prevent.
|
||||
func TestGroupProbeSpecsPrecedence(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.Globals{ProbeURL: "https://global.example/204"},
|
||||
Groups: []model.Group{
|
||||
{Name: "inherits"},
|
||||
{Name: "own", ProbeURL: "https://own.example/probe"},
|
||||
{Name: "bound", Egress: " wg0 "},
|
||||
{Name: "padded", ProbeURL: " https://padded.example/probe "},
|
||||
},
|
||||
}
|
||||
got := groupProbeSpecs(m)
|
||||
if len(got) != 4 {
|
||||
t.Fatalf("specs = %d entries, want 4", len(got))
|
||||
}
|
||||
if u := got["inherits"].ProbeURL; u != "https://global.example/204" {
|
||||
t.Errorf("a group with no URL of its own = %q, want the global one", u)
|
||||
}
|
||||
if u := got["own"].ProbeURL; u != "https://own.example/probe" {
|
||||
t.Errorf("a group's own URL = %q, want it to win over the global one", u)
|
||||
}
|
||||
if u := got["padded"].ProbeURL; u != "https://padded.example/probe" {
|
||||
t.Errorf("padded URL = %q, want it trimmed (a stray space would look like a different instrument)", u)
|
||||
}
|
||||
if e := got["bound"].Egress; e != "wg0" {
|
||||
t.Errorf("egress = %q, want it trimmed — it is a dedup key component", e)
|
||||
}
|
||||
if e := got["own"].Egress; e != "" {
|
||||
t.Errorf("unbound group egress = %q, want empty", e)
|
||||
}
|
||||
}
|
||||
|
||||
// No globals URL means "no opinion" all the way down: the planner then falls back to
|
||||
// urltest's built-in default, exactly as the generator does when nothing is configured.
|
||||
func TestGroupProbeSpecsNoOpinion(t *testing.T) {
|
||||
got := groupProbeSpecs(&model.Model{Groups: []model.Group{{Name: "a"}}})
|
||||
if u := got["a"].ProbeURL; u != "" {
|
||||
t.Fatalf("ProbeURL = %q, want \"\" (no opinion) so it cannot outvote a group that has one", u)
|
||||
}
|
||||
if groupProbeSpecs(nil) != nil {
|
||||
t.Fatal("nil model must yield nil specs")
|
||||
}
|
||||
if groupProbeSpecs(&model.Model{}) != nil {
|
||||
t.Fatal("a model with no groups must yield nil specs")
|
||||
}
|
||||
}
|
||||
|
||||
// configureSweep must honour Globals.SweepInterval end to end — the knob is worthless
|
||||
// if the value parses correctly in the model and is then ignored by the consumer.
|
||||
func TestConfigureSweepHonoursInterval(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
interval string
|
||||
wantEnabled bool
|
||||
}{
|
||||
{"unset is ON at the engine default", "", true},
|
||||
{"an explicit tick is ON", "30s", true},
|
||||
{"below the floor is still ON (clamped, not refused)", "1s", true},
|
||||
{"garbage is still ON (a typo must not disable health data)", "soon", true},
|
||||
{"0 is OFF", "0", false},
|
||||
{"off is OFF", "off", false},
|
||||
{"disabled is OFF", "disabled", false},
|
||||
{"none is OFF", "none", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
eng := engine.New()
|
||||
a := New(eng, nil)
|
||||
// GroupHealth: true is the default (DefaultGlobals seed); set it explicitly here
|
||||
// because a bare model.Globals{} zeroes it, which would gate the sweep off and
|
||||
// mask what this test is actually about (the SweepInterval grammar).
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: c.interval, GroupHealth: true}})
|
||||
enabled, _, _, _ := eng.SweepStatus()
|
||||
if enabled != c.wantEnabled {
|
||||
t.Errorf("%s: sweep_interval=%q -> enabled=%v, want %v", c.name, c.interval, enabled, c.wantEnabled)
|
||||
}
|
||||
eng.StopSweep()
|
||||
}
|
||||
}
|
||||
|
||||
// The GroupHealth master-switch gates the sweep independently of SweepInterval: when it
|
||||
// is off the sweep is OFF even for a perfectly valid interval, and the interval value is
|
||||
// still passed through untouched (flipping the toggle back on restores the schedule).
|
||||
func TestConfigureSweepGroupHealthGate(t *testing.T) {
|
||||
// GroupHealth off + a valid interval -> sweep disabled.
|
||||
// The GroupHealth master-switch is the ONLY thing that gates the observatory:
|
||||
// off means off, on means background probing of everything the applied config's
|
||||
// rules reach, on the engine's own schedule.
|
||||
func TestConfigureObservatoryGroupHealthGate(t *testing.T) {
|
||||
// GroupHealth off -> observatory disabled.
|
||||
eng := engine.New()
|
||||
a := New(eng, nil)
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "30s", GroupHealth: false}})
|
||||
if enabled, _, _, _ := eng.SweepStatus(); enabled {
|
||||
t.Fatalf("GroupHealth=false with sweep_interval=30s -> enabled=true, want the sweep gated off")
|
||||
a.configureObservatory(&model.Model{Globals: model.Globals{GroupHealth: false}}, option.Options{})
|
||||
if enabled, _, _ := eng.ObservatoryStatus(); enabled {
|
||||
t.Fatalf("GroupHealth=false -> enabled=true, want the observatory gated off")
|
||||
}
|
||||
eng.StopSweep()
|
||||
eng.StopObservatory()
|
||||
|
||||
// GroupHealth on (the default) + the same interval -> sweep enabled, as before.
|
||||
// GroupHealth on (the default) -> observatory enabled.
|
||||
eng = engine.New()
|
||||
a = New(eng, nil)
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "30s", GroupHealth: true}})
|
||||
if enabled, _, _, _ := eng.SweepStatus(); !enabled {
|
||||
t.Fatalf("GroupHealth=true with sweep_interval=30s -> enabled=false, want the sweep on")
|
||||
}
|
||||
eng.StopSweep()
|
||||
}
|
||||
|
||||
// The interval actually reaches the engine (and the floor clamp with it), rather than
|
||||
// every value collapsing onto the default.
|
||||
func TestConfigureSweepPassesTickThrough(t *testing.T) {
|
||||
eng := engine.New()
|
||||
a := New(eng, nil)
|
||||
defer eng.StopSweep()
|
||||
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "45s", GroupHealth: true}})
|
||||
if got := eng.SweepTickForTest(); got != 45*time.Second {
|
||||
t.Errorf("tick = %v, want 45s", got)
|
||||
}
|
||||
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "1s", GroupHealth: true}})
|
||||
if got := eng.SweepTickForTest(); got != model.SweepIntervalMin {
|
||||
t.Errorf("tick = %v, want the %v floor", got, model.SweepIntervalMin)
|
||||
a.configureObservatory(&model.Model{Globals: model.Globals{GroupHealth: true}}, option.Options{})
|
||||
if enabled, _, _ := eng.ObservatoryStatus(); !enabled {
|
||||
t.Fatalf("GroupHealth=true -> enabled=false, want the observatory on")
|
||||
}
|
||||
eng.StopObservatory()
|
||||
|
||||
// A nil engine is a no-op, not a panic (offline daemon stub).
|
||||
(&Applier{}).configureObservatory(&model.Model{}, option.Options{})
|
||||
}
|
||||
|
||||
@@ -11,10 +11,9 @@ package apply
|
||||
//
|
||||
// This file computes the difference, by walking the FULLY RESOLVED routing rules
|
||||
// that generate produced. Using generate's output rather than the raw model is
|
||||
// deliberate: preset packs (ru-bypass), WAN-profile overrides and time-scheduled
|
||||
// rules have all been folded in by then, so a preset-driven "Russian addresses go
|
||||
// direct" is accounted for exactly like a hand-written rule. The raw model would
|
||||
// have missed all three.
|
||||
// deliberate: WAN-profile overrides and time-scheduled rules have both been
|
||||
// folded in by then, so an override-driven "go direct" is accounted for exactly
|
||||
// like a hand-written rule. The raw model would have missed both.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
@@ -22,6 +22,16 @@ func tunnelModel() *model.Model {
|
||||
return m
|
||||
}
|
||||
|
||||
// pinnedIPSet declares the inline `type=ipcidr` rule-set a rule pins its
|
||||
// destination addresses with. Since schema v2 a routing rule has no dst_ip of its
|
||||
// own: addresses are a `config ruleset`, so the generated rule carries a rule_set
|
||||
// reference and the plan resolves the actual prefixes through the lookup below —
|
||||
// i.e. through the RUNNING engine in production (engine.RuleSetIPCIDRs), not out
|
||||
// of the config text. planFor's table stands in for that.
|
||||
func pinnedIPSet(name string, cidrs ...string) model.Ruleset {
|
||||
return model.Ruleset{Name: name, Type: "ipcidr", Source: "inline", Entries: cidrs}
|
||||
}
|
||||
|
||||
// planFor generates m and builds the untunnelable plan, resolving rule-set tags
|
||||
// from the supplied table. A tag absent from the table reports "not loaded".
|
||||
func planFor(t *testing.T, m *model.Model, sets map[string][]string) *netplane.UntunnelablePlan {
|
||||
@@ -58,11 +68,12 @@ func renderPlan(t *testing.T, m *model.Model, plan *netplane.UntunnelablePlan) s
|
||||
// only 8.8.8.8 may be un-pingable and the rest of the internet must answer.
|
||||
func TestOnlyPinnedAddressIsTunnelled(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("pin", "8.8.8.8/32")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "pin", Enabled: true, Order: 10, DstIP: []string{"8.8.8.8/32"}, Target: "group:auto"},
|
||||
{Name: "pin", Enabled: true, Order: 10, DstRuleset: []string{"pin"}, Target: "group:auto"},
|
||||
{Name: "rest", Enabled: true, Order: 99, Target: "direct"},
|
||||
}
|
||||
plan := planFor(t, m, nil)
|
||||
plan := planFor(t, m, map[string][]string{"rs-pin": {"8.8.8.8/32"}})
|
||||
|
||||
if !plan.DefaultAllow {
|
||||
t.Fatalf("a catch-all `direct` rule must make the default ALLOW; plan=%+v", plan)
|
||||
@@ -224,11 +235,12 @@ func TestDomainRuleDoesNotAffectUntunnelable(t *testing.T) {
|
||||
// TestBlockedRuleDenies: an explicitly blocked destination stays dropped.
|
||||
func TestBlockedRuleDenies(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("bad", "203.0.113.0/24")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "bad", Enabled: true, Order: 10, DstIP: []string{"203.0.113.0/24"}, Target: "block"},
|
||||
{Name: "bad", Enabled: true, Order: 10, DstRuleset: []string{"bad"}, Target: "block"},
|
||||
{Name: "rest", Enabled: true, Order: 99, Target: "direct"},
|
||||
}
|
||||
plan := planFor(t, m, nil)
|
||||
plan := planFor(t, m, map[string][]string{"rs-bad": {"203.0.113.0/24"}})
|
||||
if len(plan.Matches) != 1 || plan.Matches[0].Allow {
|
||||
t.Fatalf("a blocked destination must deny untunnelable traffic too: %+v", plan.Matches)
|
||||
}
|
||||
@@ -361,11 +373,12 @@ func TestPlanNeverAcceptsTCPOrUDP(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m.Globals.IPv6 = true
|
||||
m.Globals.Untunnelable = policy
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("pin", "8.8.8.8/32")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "pin", Enabled: true, Order: 10, DstIP: []string{"8.8.8.8/32"}, Target: "group:auto"},
|
||||
{Name: "pin", Enabled: true, Order: 10, DstRuleset: []string{"pin"}, Target: "group:auto"},
|
||||
{Name: "rest", Enabled: true, Order: 99, Target: "direct"},
|
||||
}
|
||||
fwd := renderPlan(t, m, planFor(t, m, nil))
|
||||
fwd := renderPlan(t, m, planFor(t, m, map[string][]string{"rs-pin": {"8.8.8.8/32"}}))
|
||||
|
||||
// Only the lines the untunnelable policy emits are in scope: the tproxy
|
||||
// diverts in prerouting legitimately match TCP/UDP, which is their job.
|
||||
@@ -424,12 +437,13 @@ func TestLocalPlaneSurvivesEveryPlan(t *testing.T) {
|
||||
// only grant those clients, not everyone.
|
||||
func TestSourceScopedRuleNarrowsTheAllow(t *testing.T) {
|
||||
m := tunnelModel()
|
||||
m.Rulesets = []model.Ruleset{pinnedIPSet("lab", "198.51.100.0/24")}
|
||||
m.Rules = []model.Rule{
|
||||
{Name: "lab", Enabled: true, Order: 10, Src: []string{"192.168.9.0/24"},
|
||||
DstIP: []string{"198.51.100.0/24"}, Target: "direct"},
|
||||
DstRuleset: []string{"lab"}, Target: "direct"},
|
||||
{Name: "dflt", Enabled: true, Order: 99, Target: "group:auto"},
|
||||
}
|
||||
plan := planFor(t, m, nil)
|
||||
plan := planFor(t, m, map[string][]string{"rs-lab": {"198.51.100.0/24"}})
|
||||
if len(plan.Matches) != 1 {
|
||||
t.Fatalf("expected one step, got %+v", plan.Matches)
|
||||
}
|
||||
|
||||
@@ -91,6 +91,15 @@ var criticalMarkers = []string{
|
||||
"not covered", // an interface outside the fail-closed guard
|
||||
"fail-closed", // ''
|
||||
"REJECTED", // an unusable interface name
|
||||
// A condition-less rule retired by a later condition-less rule whose target is
|
||||
// `direct` (generate/route.go warnUnreachableRules): the operator's default
|
||||
// policy — a tunnel, or a block — is not the one the router uses, so everything
|
||||
// unmatched leaves on the plain WAN. The marker is the full clause, not the
|
||||
// shorter "leaves over the plain WAN" that ruleKillFallback's kill=open note
|
||||
// also contains: that one is a DELIBERATE per-rule bypass the operator asked
|
||||
// for, and grading it critical here would be a different decision made by
|
||||
// accident.
|
||||
"leaves over the plain WAN with your real IP address",
|
||||
}
|
||||
|
||||
// protectionSections are entity kinds whose whole purpose is to block or divert
|
||||
@@ -104,7 +113,6 @@ var protectionSections = map[string]bool{
|
||||
"ruleset": true,
|
||||
"blocklist": true,
|
||||
"allowlist": true,
|
||||
"preset": true,
|
||||
}
|
||||
|
||||
// infoMarkers identify operational notes that are not protection gaps.
|
||||
|
||||
@@ -240,3 +240,45 @@ func blockGlobals() model.Globals {
|
||||
g.Untunnelable = netplane.UntunnelableBlock
|
||||
return g
|
||||
}
|
||||
|
||||
// TestCollectWarningsGradesUnreachableRules (B1): a condition-less rule retired by
|
||||
// a later condition-less one is graded by CONSEQUENCE, not by the mere fact that a
|
||||
// setting is dead.
|
||||
//
|
||||
// - the surviving default is `direct` while the retired one wanted a tunnel:
|
||||
// the operator's default policy is not in effect and everything unmatched
|
||||
// leaves on the plain WAN — critical, the panel must show it loudly;
|
||||
// - the surviving default is the tunnel and the dead one was `direct`: a dead
|
||||
// knob, nothing is leaking — warning.
|
||||
//
|
||||
// Both must be attributed to section "rule" + the rule's name, so the panel can
|
||||
// deep-link to the offending row instead of printing prose.
|
||||
func TestCollectWarningsGradesUnreachableRules(t *testing.T) {
|
||||
const leak = `rule "default": it has no conditions, so it sets the default for ALL traffic — ` +
|
||||
`but rule "fallback" (order 100) has none either and comes after it, so "direct" wins and ` +
|
||||
`this rule's target "group:auto" is never applied. Everything no other rule matches leaves ` +
|
||||
`over the plain WAN with your real IP address. Delete one of the two, or give this one a condition`
|
||||
const dead = `rule "default": it has no conditions, so it sets the default for ALL traffic — ` +
|
||||
`but rule "default" (order 100) has none either and comes after it, so the default the ` +
|
||||
`router uses is "group:auto" and this rule's target "direct" is never applied. Delete one ` +
|
||||
`of the two, or give this one a condition so it can match something`
|
||||
|
||||
check := func(text, wantSeverity string) {
|
||||
t.Helper()
|
||||
found := false
|
||||
for _, w := range collectWarnings(blockGlobals(), []string{text}, nil, nil) {
|
||||
if w.Section != "rule" || w.Name != "default" {
|
||||
continue
|
||||
}
|
||||
found = true
|
||||
if w.Severity != wantSeverity {
|
||||
t.Errorf("severity = %q, want %q for: %s", w.Severity, wantSeverity, text)
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Fatalf("never-applied warning not attributed to rule %q: %s", "default", text)
|
||||
}
|
||||
}
|
||||
check(leak, SeverityCritical)
|
||||
check(dead, SeverityWarning)
|
||||
}
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
package main
|
||||
|
||||
// Control-plane log formatting (the IO-free half, so it is unit-testable on any
|
||||
// dev host — same split as profilewatch.go).
|
||||
//
|
||||
// The daemon's own logger used to be built with a bare log.Formatter, whose
|
||||
// DisableColors zero value is false: every control-plane line went to procd's
|
||||
// stderr — i.e. straight into syslog — carrying aurora escape sequences. See
|
||||
// shater/logsink/color.go for why that is a defect and not a cosmetic.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/shater/logsink"
|
||||
)
|
||||
|
||||
// controlLogFormatter builds the control-plane logger's formatter. out is the
|
||||
// process's real stderr (the sink's syslog half): colours are emitted only when
|
||||
// that is a terminal, never when it is procd/syslog or a log file.
|
||||
func controlLogFormatter(baseTime time.Time, out *os.File) log.Formatter {
|
||||
return log.Formatter{
|
||||
BaseTime: baseTime,
|
||||
DisableColors: !logsink.IsTTY(out),
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"context"
|
||||
"os"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
)
|
||||
|
||||
// TestControlLogFormatterNoANSIOffTTY pins the control-plane half of the syslog
|
||||
// colour leak: when the daemon's stderr is not a terminal — which is ALWAYS the
|
||||
// case under procd, where stderr is the pipe procd relays to syslog — no line
|
||||
// the daemon formats may contain an ESC (0x1b).
|
||||
func TestControlLogFormatterNoANSIOffTTY(t *testing.T) {
|
||||
r, w, err := os.Pipe()
|
||||
if err != nil {
|
||||
t.Fatalf("pipe: %v", err)
|
||||
}
|
||||
t.Cleanup(func() { _ = r.Close(); _ = w.Close() })
|
||||
|
||||
f := controlLogFormatter(time.Now(), w)
|
||||
if !f.DisableColors {
|
||||
t.Fatalf("colours enabled for a non-terminal stderr")
|
||||
}
|
||||
// A context ID exercises the second colouring branch of log/format.go (the
|
||||
// 256-colour connection id), which is what produced ESC[38;5;193m on the router.
|
||||
ctx := log.ContextWithNewID(context.Background())
|
||||
for _, level := range []log.Level{log.LevelError, log.LevelWarn, log.LevelInfo, log.LevelDebug, log.LevelTrace} {
|
||||
line := f.Format(ctx, level, "dns", "exchange failed for example.com. IN AAAA: unexpected EOF", time.Now())
|
||||
if i := strings.IndexByte(line, 0x1b); i >= 0 {
|
||||
t.Errorf("level %v: formatted line carries an ANSI escape at byte %d: %q", level, i, line)
|
||||
}
|
||||
}
|
||||
}
|
||||
+93
-13
@@ -79,7 +79,7 @@ func dispatch(args []string) int {
|
||||
case "mint-token":
|
||||
return cmdMintToken()
|
||||
case "nodes":
|
||||
return cmdReadStub("nodes", "[]")
|
||||
return cmdNodes()
|
||||
case "stats":
|
||||
return cmdReadStub("stats", "{}")
|
||||
case "blocklist":
|
||||
@@ -124,7 +124,7 @@ usage: shaterd <verb>
|
||||
rollback roll back to the last-good config
|
||||
status print daemon/data-plane status as JSON
|
||||
mint-token mint a single-use panel handoff token (JSON) via the daemon
|
||||
nodes print nodes as JSON
|
||||
nodes print the configured nodes as JSON (manual + subscription)
|
||||
stats print stats as JSON
|
||||
blocklist update force a DNS-filter blocklist refresh (reconcile; no-op if down)
|
||||
schedule due re-evaluate time-scheduled rules now (reconcile; no-op if down)
|
||||
@@ -158,8 +158,11 @@ func cmdRun() int {
|
||||
// The level starts at trace (nothing read yet) and is corrected from
|
||||
// Globals.LogLevel as soon as UCI is read — the control plane now RESPECTS
|
||||
// the configured level instead of the old unconditional trace.
|
||||
// The formatter colours only when stderr is a real terminal
|
||||
// (controlLogFormatter -> logsink.IsTTY): under procd stderr IS syslog, and
|
||||
// ANSI escapes there break `logread | grep ERROR` and every log collector.
|
||||
logFactory := log.NewDefaultFactory(context.Background(),
|
||||
log.Formatter{BaseTime: time.Now()}, sink, "", nil, false)
|
||||
controlLogFormatter(time.Now(), os.Stderr), sink, "", nil, false)
|
||||
log.SetStdLogger(logFactory.Logger())
|
||||
logger := log.StdLogger()
|
||||
|
||||
@@ -172,10 +175,23 @@ func cmdRun() int {
|
||||
logFactory.SetLevel(controlLogLevel(g.LogLevel))
|
||||
}
|
||||
|
||||
// Refuse to start a second daemon (race-safe single-owner guard). procd's
|
||||
// term_timeout ensures the previous instance exits before reload restarts us.
|
||||
if pid, ok := daemonAlive(); ok && pid != os.Getpid() {
|
||||
logger.Error("another shaterd is already running (pid ", pid, ") — refusing to start")
|
||||
// Single-owner guard: never run two daemons at once. On a `restart` procd's
|
||||
// `service delete` is ASYNCHRONOUS — the ubus call returns immediately while
|
||||
// the outgoing instance is still running its honest teardown — and the
|
||||
// following `service add` starts us straight away, so a predecessor being
|
||||
// alive here is the NORMAL restart case, not an error.
|
||||
//
|
||||
// This used to exit(1) on the spot and lean on procd's `respawn ... 5 ...` to
|
||||
// try again five seconds later. That is a blind retry, not synchronisation:
|
||||
// it neither knows nor waits for the predecessor's teardown to finish, and it
|
||||
// turns every restart into at least one logged crash plus a five-second hole
|
||||
// in which the LAN has no plane at all. Waiting for the predecessor to exit
|
||||
// makes `restart` behave exactly like `stop` + pause + `start`: our apply is
|
||||
// then strictly ordered AFTER the previous teardown, which is the whole point
|
||||
// of the guard.
|
||||
if pid, waiting := waitForPredecessor(predecessorBudget, predecessorPoll, logger); waiting {
|
||||
logger.Error("another shaterd is still running (pid ", pid, ") after waiting ",
|
||||
predecessorBudget, " for it to exit — refusing to start")
|
||||
return 1
|
||||
}
|
||||
if err := writePidfile(); err != nil {
|
||||
@@ -497,9 +513,12 @@ func watchNewDevices(n *alert.Notifier, l log.ContextLogger) {
|
||||
continue // baseline poll: seed seen[], alert nothing
|
||||
}
|
||||
for _, d := range fresh {
|
||||
label := d.Hostname
|
||||
label := strings.TrimSpace(d.Hostname)
|
||||
// No hostname: the old fallback used the IP as the "name", producing
|
||||
// "10.67.0.223 (mac) at 10.67.0.223". Lead with the MAC instead.
|
||||
body := fmt.Sprintf("%s (%s) at %s", label, d.MAC, d.IP)
|
||||
if label == "" {
|
||||
label = d.IP
|
||||
body = fmt.Sprintf("%s at %s", d.MAC, d.IP)
|
||||
}
|
||||
l.Info("new device on LAN: ", d.MAC, " ", d.IP, " ", label)
|
||||
// Key on the MAC: two different devices joining within the dedup window
|
||||
@@ -509,7 +528,7 @@ func watchNewDevices(n *alert.Notifier, l log.ContextLogger) {
|
||||
n.FireIncident(alert.Incident{
|
||||
Events: []string{"new_device"},
|
||||
Title: "New device on the network",
|
||||
Body: fmt.Sprintf("%s (%s) at %s", label, d.MAC, d.IP),
|
||||
Body: body,
|
||||
Key: "new_device:" + d.MAC,
|
||||
})
|
||||
}
|
||||
@@ -544,7 +563,7 @@ func watchActiveProfile(applier *apply.Applier, logger log.ContextLogger) {
|
||||
logger.Debug("profile watch: read UCI failed — skip: ", err)
|
||||
continue
|
||||
}
|
||||
desired := desiredActiveProfile(time.Now(), m.Profiles, m.Globals.ActiveProfile, dev, netplane.IfaceDevice)
|
||||
desired := desiredActiveProfile(m.Profiles, m.Globals.ActiveProfile, dev, netplane.IfaceDevice)
|
||||
if desired == m.Globals.ActiveProfile {
|
||||
continue // no change — anti-flap
|
||||
}
|
||||
@@ -729,12 +748,21 @@ func cmdSubUpdate(name string) int {
|
||||
failed = true
|
||||
continue
|
||||
}
|
||||
// The refreshed set is durable in the sub's own JSON cache file; UCI
|
||||
// (below) only drains legacy sections — RenderUCIExport does not emit
|
||||
// FromSub nodes, so a refresh no longer rewrites /etc/config/shater.
|
||||
if serr := model.SaveSubCache(s.Name, m.NodesFromSub(s.Name)); serr != nil {
|
||||
logger.Error("sub ", s.Name, ": ", serr)
|
||||
failed = true
|
||||
continue
|
||||
}
|
||||
updated++
|
||||
summary = append(summary, fmt.Sprintf("%s: %d nodes", s.Name, added))
|
||||
}
|
||||
|
||||
// Only persist when something actually changed — a total failure must leave the
|
||||
// on-disk config (and its existing node cache) untouched.
|
||||
// on-disk config (and the existing node caches) untouched. The write is what
|
||||
// migrates an old config: legacy from_sub sections are dropped by the render.
|
||||
if updated > 0 {
|
||||
if werr := model.WriteUCI(m); werr != nil {
|
||||
fmt.Fprintln(os.Stderr, "shaterd sub update: write UCI:", werr)
|
||||
@@ -820,6 +848,42 @@ func cmdStatus() int {
|
||||
return 0
|
||||
}
|
||||
|
||||
// cmdNodes prints the node inventory as a JSON array (see nodes.go for the
|
||||
// shape and for why this is a view rather than the raw model.Node).
|
||||
//
|
||||
// Shape of the call mirrors cmdStatus: ask the RUNNING daemon first — it is the
|
||||
// process that owns the engine, so its answer is the inventory the live box was
|
||||
// built from — and fall back to reading the same on-disk desired state directly
|
||||
// when there is no daemon (or it did not answer). Both sides call the SAME
|
||||
// nodesJSON(), so the fallback cannot report something the daemon would not.
|
||||
//
|
||||
// The one thing it will never do is print `[]` because it could not find out:
|
||||
// a read failure goes to stderr and exits non-zero, so "empty" on stdout with
|
||||
// exit 0 means "no nodes are configured" and nothing else.
|
||||
func cmdNodes() int {
|
||||
if _, ok := daemonAlive(); ok {
|
||||
resp, err := ctlRequest("nodes")
|
||||
switch {
|
||||
case err != nil:
|
||||
fmt.Fprintf(os.Stderr, "shaterd nodes: %v — reading the on-disk config instead\n", err)
|
||||
case strings.HasPrefix(strings.TrimSpace(resp), "["):
|
||||
fmt.Println(strings.TrimSpace(resp))
|
||||
return 0
|
||||
default:
|
||||
// The daemon answered, but with an error object rather than a list.
|
||||
fmt.Fprintf(os.Stderr, "shaterd nodes: daemon: %s\n", strings.TrimSpace(resp))
|
||||
}
|
||||
}
|
||||
b, err := nodesJSON()
|
||||
if err != nil {
|
||||
fmt.Fprintf(os.Stderr, "shaterd nodes: %v\n", err)
|
||||
fmt.Println("[]") // stdout stays parseable JSON; the exit code carries the failure
|
||||
return 1
|
||||
}
|
||||
fmt.Println(string(b))
|
||||
return 0
|
||||
}
|
||||
|
||||
// cmdMintToken asks the RUNNING daemon (over the control socket) for a single-use
|
||||
// panel handoff token and prints the daemon's JSON reply verbatim — {"token":"..."}
|
||||
// on success, {"error":"..."} otherwise. It is the CLI shim the LuCI/rpcd layer
|
||||
@@ -857,6 +921,13 @@ func printJSONError(msg string) {
|
||||
fmt.Println(string(b))
|
||||
}
|
||||
|
||||
// cmdReadStub relays a read-only verb to the daemon and prints def when there is
|
||||
// no daemon to ask. It is ONLY valid where def is the truth in the daemon-down
|
||||
// case: `stats` counts what the LIVE engine saw, so with no engine there really
|
||||
// is nothing counted and `{}` says exactly that. It is NOT a way to make a verb
|
||||
// look implemented — `nodes` used to be routed through here with def="[]" while
|
||||
// hundreds of nodes sat on disk (see nodes.go). Anything whose data outlives the
|
||||
// daemon belongs in its own command that reads that data.
|
||||
func cmdReadStub(cmd, def string) int {
|
||||
if _, ok := daemonAlive(); ok {
|
||||
if resp, err := ctlRequest(cmd); err == nil {
|
||||
@@ -928,7 +999,16 @@ func handleCtl(conn net.Conn, a *apply.Applier, ps *panel.Server, sa stats.Stats
|
||||
case "rollback":
|
||||
writeResult(conn, false, a.Rollback())
|
||||
case "nodes":
|
||||
writeLine(conn, "[]")
|
||||
// Same assembly as the CLI fallback and as GET /api/config: model.ReadUCI
|
||||
// (UCI + the per-subscription caches) projected onto the printable view.
|
||||
b, err := nodesJSON()
|
||||
if err != nil {
|
||||
l.Warn("control socket: nodes: ", err)
|
||||
e, _ := json.Marshal(map[string]string{"error": err.Error()})
|
||||
writeLine(conn, string(e))
|
||||
return
|
||||
}
|
||||
writeLine(conn, string(b))
|
||||
case "stats":
|
||||
if sa == nil {
|
||||
writeLine(conn, "{}")
|
||||
|
||||
@@ -0,0 +1,132 @@
|
||||
package main
|
||||
|
||||
// `shaterd nodes` — the node inventory, as JSON.
|
||||
//
|
||||
// This verb used to answer `[]` unconditionally (a `cmdReadStub("nodes", "[]")`
|
||||
// on the CLI side and a hard-coded `writeLine(conn, "[]")` in the daemon's
|
||||
// control-socket handler), while `shaterd --help` advertised "print nodes as
|
||||
// JSON". The data was never missing: /etc/shater/subs/*.json holds every
|
||||
// subscription-fetched node and `GET /api/config` reports the full merged set —
|
||||
// on a live router that is hundreds of nodes answered as "none". An empty list
|
||||
// is indistinguishable from a truthful "no nodes are configured", so the verb
|
||||
// did not fail loudly, it lied quietly. Same class as the honesty fixes in
|
||||
// 9dc954029 / aec82d444; the cure is the same: report what is actually there.
|
||||
//
|
||||
// # Data path (deliberately the SAME one /api/config uses)
|
||||
//
|
||||
// model.ReadUCI() = `uci export shater` (manual nodes + everything else) +
|
||||
// model.MergeSubCaches (the per-subscription JSON caches). That single call is
|
||||
// what the panel's handleConfigGet serves, what generate builds the engine from
|
||||
// and what `sub update` writes back — so `shaterd nodes` cannot drift from the
|
||||
// panel or from the running engine, because there is no second assembly here to
|
||||
// drift. This file only PROJECTS that model onto a small, printable view.
|
||||
//
|
||||
// # Why a view and not the raw model.Node
|
||||
//
|
||||
// model.Node carries the share-link URI, and a share link is a credential
|
||||
// (uuid/password in the query string). /api/config may return it — that path is
|
||||
// session-authenticated and the panel needs the URI to edit a node — but a CLI
|
||||
// verb whose output gets piped into support tickets, `logger`, and cron mail
|
||||
// must not spray credentials. The view therefore reports the *derived* facts a
|
||||
// share link answers (protocol, server, port) and drops the secret-bearing URI.
|
||||
// Nodes whose URI does not parse are still listed, with the parse error in
|
||||
// `parse_error`: the engine skips exactly those nodes (generate/outbound.go),
|
||||
// and a node that is configured-but-unusable is precisely what an operator
|
||||
// needs to see — hiding it would be the same lie in a smaller coat.
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
"github.com/sagernet/sing-box/shater/parse"
|
||||
)
|
||||
|
||||
// nodeView is ONE node as the `nodes` verb reports it. Field set: the model's
|
||||
// own facts (name/enabled/sub/egress/stale/fingerprint) plus what the share link
|
||||
// decodes to (protocol/server/port). Everything is omitempty-free where a reader
|
||||
// would have to distinguish "absent" from "false"/"zero" — `enabled` and `stale`
|
||||
// are always present so a consumer never has to guess.
|
||||
type nodeView struct {
|
||||
Name string `json:"name"`
|
||||
Enabled bool `json:"enabled"`
|
||||
// Sub is the subscription this node came from; "" means a manual node
|
||||
// (model.Node.FromSub semantics, unchanged).
|
||||
Sub string `json:"sub"`
|
||||
// Protocol is the parsed protocol (vless|vmess|trojan|shadowsocks|wireguard…).
|
||||
// When the URI does not parse it falls back to the bare URI scheme so the
|
||||
// operator still sees what kind of thing failed; "" only when the URI is empty.
|
||||
Protocol string `json:"protocol"`
|
||||
Server string `json:"server,omitempty"`
|
||||
Port uint16 `json:"port,omitempty"`
|
||||
// Egress is the `config egress` this node's own upstream is bound to
|
||||
// (multi-WAN); "" = default route.
|
||||
Egress string `json:"egress,omitempty"`
|
||||
// Stale marks a cached subscription node whose subscription failed to refresh.
|
||||
Stale bool `json:"stale"`
|
||||
Fingerprint string `json:"fingerprint,omitempty"`
|
||||
// ParseError is the reason the engine will SKIP this node, verbatim from
|
||||
// parse.ParseShareLink. Empty on every usable node.
|
||||
ParseError string `json:"parse_error,omitempty"`
|
||||
}
|
||||
|
||||
// readModel is the model source. A package var so the tests can exercise the
|
||||
// assembly without a router's `uci` binary; production always reads the real
|
||||
// merged desired state.
|
||||
var readModel = model.ReadUCI
|
||||
|
||||
// nodesJSON reads the desired state and renders the node inventory as a JSON
|
||||
// array. The array is never `null`: an empty configuration marshals to `[]`,
|
||||
// which is the ONE case where `[]` is the truth.
|
||||
func nodesJSON() ([]byte, error) {
|
||||
m, err := readModel()
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return json.Marshal(nodeViews(m))
|
||||
}
|
||||
|
||||
// nodeViews projects the merged model onto the printable view. Pure: no IO, no
|
||||
// globals — the whole verb's logic is testable on a dev host.
|
||||
func nodeViews(m *model.Model) []nodeView {
|
||||
out := make([]nodeView, 0, len(m.Nodes)) // never nil => never `null`
|
||||
for i := range m.Nodes {
|
||||
n := m.Nodes[i]
|
||||
v := nodeView{
|
||||
Name: n.Name,
|
||||
Enabled: n.Enabled,
|
||||
Sub: n.FromSub,
|
||||
Egress: n.Egress,
|
||||
Stale: n.Stale,
|
||||
Fingerprint: n.Fingerprint,
|
||||
Protocol: uriScheme(n.URI),
|
||||
}
|
||||
// The engine reads a node through exactly this call (generate/outbound.go,
|
||||
// chain.go, group.go); using it here is what makes "protocol" agree with
|
||||
// what will actually be dialled, and the error agree with what will be
|
||||
// skipped.
|
||||
if p, err := parse.ParseShareLink(n.URI); err == nil {
|
||||
if p.Protocol != "" {
|
||||
v.Protocol = p.Protocol
|
||||
}
|
||||
v.Server = p.Server
|
||||
v.Port = p.Port
|
||||
} else if n.URI != "" {
|
||||
v.ParseError = err.Error()
|
||||
}
|
||||
out = append(out, v)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// uriScheme returns the lower-cased scheme of a share link ("vless://…" ->
|
||||
// "vless"), or "" when there is none. It is the fallback protocol for a URI that
|
||||
// ParseShareLink refuses, so an unusable node is still described rather than
|
||||
// reported as a typeless blank.
|
||||
func uriScheme(uri string) string {
|
||||
i := strings.Index(uri, "://")
|
||||
if i <= 0 {
|
||||
return ""
|
||||
}
|
||||
return strings.ToLower(uri[:i])
|
||||
}
|
||||
@@ -0,0 +1,146 @@
|
||||
package main
|
||||
|
||||
import (
|
||||
"encoding/json"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// vlessURI is a syntactically complete share link (host, port, uuid, fragment)
|
||||
// so ParseShareLink really succeeds and the derived fields are exercised.
|
||||
const vlessURI = "vless://11111111-2222-3333-4444-555555555555@example.net:443?" +
|
||||
"encryption=none&security=tls&sni=example.net&type=ws&path=%2Fws#NL-vless-1"
|
||||
|
||||
// writeSubCache points the subscription cache at a temp dir and persists one
|
||||
// subscription's nodes there, the way `sub update` does on the router
|
||||
// (/etc/shater/subs/<sub>.json). Returns nothing: the point is the side effect
|
||||
// that model.MergeSubCaches will pick up.
|
||||
func writeSubCache(t *testing.T, sub string, nodes []model.Node) {
|
||||
t.Helper()
|
||||
t.Setenv("SHATER_SUBS_DIR", t.TempDir())
|
||||
if err := model.SaveSubCache(sub, nodes); err != nil {
|
||||
t.Fatalf("SaveSubCache(%q): %v", sub, err)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNodesJSONReportsCachedSubscriptionNodes is the regression for the bug this
|
||||
// verb had: `shaterd nodes` answered `[]` while the subscription cache held
|
||||
// hundreds of nodes. A NON-EMPTY cache must produce a NON-EMPTY list.
|
||||
func TestNodesJSONReportsCachedSubscriptionNodes(t *testing.T) {
|
||||
writeSubCache(t, "all-qomar", []model.Node{
|
||||
{Name: "NL-vless-1", Enabled: true, URI: vlessURI, FromSub: "all-qomar", Fingerprint: "66b33"},
|
||||
{Name: "NL-vless-2", Enabled: false, URI: vlessURI, FromSub: "all-qomar"},
|
||||
})
|
||||
|
||||
// The model source stands in for `uci export shater` (absent on a dev host);
|
||||
// the subscription half is the REAL model.MergeSubCaches path.
|
||||
restore := readModel
|
||||
readModel = func() (*model.Model, error) {
|
||||
m := &model.Model{Globals: model.DefaultGlobals()}
|
||||
model.MergeSubCaches(m)
|
||||
return m, nil
|
||||
}
|
||||
t.Cleanup(func() { readModel = restore })
|
||||
|
||||
b, err := nodesJSON()
|
||||
if err != nil {
|
||||
t.Fatalf("nodesJSON: %v", err)
|
||||
}
|
||||
var got []nodeView
|
||||
if err := json.Unmarshal(b, &got); err != nil {
|
||||
t.Fatalf("nodesJSON produced invalid JSON %q: %v", b, err)
|
||||
}
|
||||
if len(got) == 0 {
|
||||
t.Fatalf("nodes reported an EMPTY list while the subscription cache holds 2 nodes: %s", b)
|
||||
}
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("got %d nodes, want 2: %s", len(got), b)
|
||||
}
|
||||
byName := map[string]nodeView{}
|
||||
for _, v := range got {
|
||||
byName[v.Name] = v
|
||||
}
|
||||
first, ok := byName["NL-vless-1"]
|
||||
if !ok {
|
||||
t.Fatalf("cached node NL-vless-1 missing from %s", b)
|
||||
}
|
||||
if !first.Enabled {
|
||||
t.Errorf("NL-vless-1.enabled = false, want true")
|
||||
}
|
||||
if first.Sub != "all-qomar" {
|
||||
t.Errorf("NL-vless-1.sub = %q, want %q", first.Sub, "all-qomar")
|
||||
}
|
||||
if first.Protocol != "vless" {
|
||||
t.Errorf("NL-vless-1.protocol = %q, want %q", first.Protocol, "vless")
|
||||
}
|
||||
if first.Server != "example.net" || first.Port != 443 {
|
||||
t.Errorf("NL-vless-1 server:port = %s:%d, want example.net:443", first.Server, first.Port)
|
||||
}
|
||||
if first.ParseError != "" {
|
||||
t.Errorf("NL-vless-1.parse_error = %q, want empty", first.ParseError)
|
||||
}
|
||||
if byName["NL-vless-2"].Enabled {
|
||||
t.Errorf("NL-vless-2.enabled = true, want false (the model says disabled)")
|
||||
}
|
||||
// The credential-bearing share link must not be printed.
|
||||
if strings.Contains(string(b), "11111111-2222-3333-4444-555555555555") {
|
||||
t.Errorf("nodes output leaks the share-link uuid: %s", b)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNodeViewsShape pins the projection: manual vs subscription, an unparsable
|
||||
// URI still being LISTED (with the reason), and an empty model marshalling to
|
||||
// `[]` rather than `null`.
|
||||
func TestNodeViewsShape(t *testing.T) {
|
||||
m := &model.Model{Nodes: []model.Node{
|
||||
{Name: "manual-1", Enabled: true, URI: vlessURI, Egress: "wan2"},
|
||||
{Name: "broken", Enabled: true, URI: "nosuch://whatever", FromSub: "qomar", Stale: true},
|
||||
{Name: "no-uri", Enabled: false},
|
||||
}}
|
||||
got := nodeViews(m)
|
||||
if len(got) != 3 {
|
||||
t.Fatalf("nodeViews returned %d views, want 3", len(got))
|
||||
}
|
||||
if got[0].Sub != "" {
|
||||
t.Errorf("manual node sub = %q, want empty", got[0].Sub)
|
||||
}
|
||||
if got[0].Egress != "wan2" {
|
||||
t.Errorf("manual node egress = %q, want wan2", got[0].Egress)
|
||||
}
|
||||
if got[1].ParseError == "" {
|
||||
t.Errorf("unparsable node must carry the reason the engine will skip it")
|
||||
}
|
||||
if got[1].Protocol != "nosuch" {
|
||||
t.Errorf("unparsable node protocol = %q, want the bare scheme %q", got[1].Protocol, "nosuch")
|
||||
}
|
||||
if !got[1].Stale {
|
||||
t.Errorf("stale flag lost")
|
||||
}
|
||||
if got[2].Protocol != "" || got[2].ParseError != "" {
|
||||
t.Errorf("URI-less node: got protocol=%q parse_error=%q, want both empty", got[2].Protocol, got[2].ParseError)
|
||||
}
|
||||
|
||||
b, err := json.Marshal(nodeViews(&model.Model{}))
|
||||
if err != nil {
|
||||
t.Fatalf("marshal empty: %v", err)
|
||||
}
|
||||
if string(b) != "[]" {
|
||||
t.Errorf("empty model marshalled to %q, want []", b)
|
||||
}
|
||||
}
|
||||
|
||||
func TestURIScheme(t *testing.T) {
|
||||
for _, tc := range []struct{ in, want string }{
|
||||
{"vless://x@h:443", "vless"},
|
||||
{"VMESS://payload", "vmess"},
|
||||
{"", ""},
|
||||
{"not-a-uri", ""},
|
||||
{"://leading", ""},
|
||||
} {
|
||||
if got := uriScheme(tc.in); got != tc.want {
|
||||
t.Errorf("uriScheme(%q) = %q, want %q", tc.in, got, tc.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
package main
|
||||
|
||||
// The startup handshake that stops a restarting daemon from standing a new data
|
||||
// plane up on top of the previous one's teardown.
|
||||
//
|
||||
// Deliberately free of build tags: the logic is pure timing and is exercised by
|
||||
// the tests on any host, while the probe it polls (pidfile + kill(pid, 0)) is
|
||||
// linux-only and is installed by predecessor_linux.go.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
)
|
||||
|
||||
// predecessorBudget / predecessorPoll bound the wait for an outgoing daemon.
|
||||
//
|
||||
// The budget must comfortably exceed the slowest honest teardown (engine.Close
|
||||
// of a box with a few hundred outbounds plus the cache-file flush to flash, then
|
||||
// the nft/ip/sysctl removal) AND the init script's procd term_timeout, which is
|
||||
// the hard cap on how long a predecessor can live after its SIGTERM. Overshooting
|
||||
// costs nothing on a healthy box — the wait ends the instant the predecessor is
|
||||
// gone — while undershooting reintroduces the very overlap this exists to remove.
|
||||
//
|
||||
// Variables, not constants, so the tests can drive the loop without sleeping.
|
||||
var (
|
||||
predecessorBudget = 60 * time.Second
|
||||
predecessorPoll = 100 * time.Millisecond
|
||||
)
|
||||
|
||||
// aliveProbe is the seam waitForPredecessor polls. The portable default reports
|
||||
// "no predecessor", so a non-linux build (where there is no daemon at all) never
|
||||
// waits; predecessor_linux.go replaces it with the real pidfile probe.
|
||||
var aliveProbe = func() (int, bool) { return 0, false }
|
||||
|
||||
// waitForPredecessor blocks until no OTHER shaterd owns the pidfile, or until the
|
||||
// budget runs out.
|
||||
//
|
||||
// Returns (pid, true) only when a predecessor is STILL alive once the budget
|
||||
// expires — the genuine "two daemons" error the caller refuses on. A predecessor
|
||||
// that exits within the budget (the restart case) returns (0, false) and startup
|
||||
// continues, now guaranteed to be sequenced after its teardown, exactly as it is
|
||||
// after a manual `stop` + pause + `start`.
|
||||
func waitForPredecessor(budget, poll time.Duration, logger log.ContextLogger) (int, bool) {
|
||||
pid, ok := aliveProbe()
|
||||
if !ok || pid == os.Getpid() {
|
||||
return 0, false
|
||||
}
|
||||
if logger != nil {
|
||||
logger.Info("a previous shaterd (pid ", pid, ") is still shutting down — ",
|
||||
"waiting for its teardown to finish before applying a new data plane")
|
||||
}
|
||||
deadline := time.Now().Add(budget)
|
||||
for time.Now().Before(deadline) {
|
||||
time.Sleep(poll)
|
||||
pid, ok = aliveProbe()
|
||||
if !ok || pid == os.Getpid() {
|
||||
return 0, false
|
||||
}
|
||||
}
|
||||
return pid, true
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
//go:build linux
|
||||
|
||||
package main
|
||||
|
||||
// Bind the portable predecessor wait to the real single-owner probe. Split out
|
||||
// of main.go so waitForPredecessor itself stays build-tag-free and testable on
|
||||
// any developer host.
|
||||
|
||||
func init() { aliveProbe = daemonAlive }
|
||||
@@ -0,0 +1,106 @@
|
||||
package main
|
||||
|
||||
// B3 regression, daemon half.
|
||||
//
|
||||
// `/etc/init.d/shater restart` is `stop; start`, and procd's `stop` is
|
||||
// asynchronous: the ubus `service delete` returns as soon as SIGTERM has been
|
||||
// SENT, so `start` re-adds the instance while the outgoing `shaterd run` is
|
||||
// still executing its honest teardown. The startup guard used to exit(1) the
|
||||
// moment it saw a live predecessor and rely on procd's `respawn ... 5 ...` to
|
||||
// try again later — a blind retry that neither knows nor waits for the teardown
|
||||
// to finish, and that turns every restart into a logged crash plus a five-second
|
||||
// hole with no data plane.
|
||||
//
|
||||
// The contract these pin: startup BLOCKS until the predecessor is gone (so our
|
||||
// apply is strictly ordered after its teardown, exactly as it is after a manual
|
||||
// `stop` + pause + `start`), and only refuses when the predecessor outlives the
|
||||
// whole budget.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// scriptAlive installs an aliveProbe that reports a live predecessor for the
|
||||
// first n calls and "gone" afterwards, and restores the real probe.
|
||||
func scriptAlive(t *testing.T, pid, n int) *int {
|
||||
t.Helper()
|
||||
orig := aliveProbe
|
||||
t.Cleanup(func() { aliveProbe = orig })
|
||||
calls := 0
|
||||
aliveProbe = func() (int, bool) {
|
||||
calls++
|
||||
if calls <= n {
|
||||
return pid, true
|
||||
}
|
||||
return 0, false
|
||||
}
|
||||
return &calls
|
||||
}
|
||||
|
||||
// TestWaitForPredecessorWaitsForTeardown: a predecessor that is still tearing
|
||||
// the plane down must be WAITED for, not refused. Before the fix this returned
|
||||
// "still alive" on the first probe and the daemon exited 1.
|
||||
func TestWaitForPredecessorWaitsForTeardown(t *testing.T) {
|
||||
calls := scriptAlive(t, 4242, 3)
|
||||
|
||||
pid, stillAlive := waitForPredecessor(2*time.Second, time.Millisecond, nil)
|
||||
if stillAlive {
|
||||
t.Fatalf("a predecessor that exits within the budget must not be refused (pid %d)", pid)
|
||||
}
|
||||
if *calls < 4 {
|
||||
t.Fatalf("expected the guard to keep probing until the predecessor was gone, got %d probes", *calls)
|
||||
}
|
||||
}
|
||||
|
||||
// TestWaitForPredecessorRefusesAfterBudget: the single-owner invariant is kept —
|
||||
// a predecessor that never dies still ends in a refusal, it is just no longer
|
||||
// the FIRST answer.
|
||||
func TestWaitForPredecessorRefusesAfterBudget(t *testing.T) {
|
||||
orig := aliveProbe
|
||||
defer func() { aliveProbe = orig }()
|
||||
aliveProbe = func() (int, bool) { return 4242, true }
|
||||
|
||||
start := time.Now()
|
||||
pid, stillAlive := waitForPredecessor(30*time.Millisecond, time.Millisecond, nil)
|
||||
if !stillAlive || pid != 4242 {
|
||||
t.Fatalf("an immortal predecessor must still be refused, got pid=%d alive=%v", pid, stillAlive)
|
||||
}
|
||||
if time.Since(start) < 30*time.Millisecond {
|
||||
t.Fatalf("the guard must exhaust its budget before refusing")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWaitForPredecessorNoPredecessorIsFree: the boot path must pay nothing —
|
||||
// one probe, no sleep. A wait that cost a tick on every clean start would show up
|
||||
// as a slower boot for no reason.
|
||||
func TestWaitForPredecessorNoPredecessorIsFree(t *testing.T) {
|
||||
orig := aliveProbe
|
||||
defer func() { aliveProbe = orig }()
|
||||
calls := 0
|
||||
aliveProbe = func() (int, bool) { calls++; return 0, false }
|
||||
|
||||
start := time.Now()
|
||||
if _, stillAlive := waitForPredecessor(time.Minute, time.Second, nil); stillAlive {
|
||||
t.Fatalf("no predecessor must not be reported as alive")
|
||||
}
|
||||
if calls != 1 {
|
||||
t.Fatalf("expected exactly one probe when nothing is running, got %d", calls)
|
||||
}
|
||||
if time.Since(start) > 500*time.Millisecond {
|
||||
t.Fatalf("the no-predecessor path must not sleep")
|
||||
}
|
||||
}
|
||||
|
||||
// TestWaitForPredecessorIgnoresOwnPid: a pidfile naming THIS process (a crashed
|
||||
// predecessor whose pid we were handed, or a re-exec) is not a predecessor.
|
||||
func TestWaitForPredecessorIgnoresOwnPid(t *testing.T) {
|
||||
orig := aliveProbe
|
||||
defer func() { aliveProbe = orig }()
|
||||
aliveProbe = func() (int, bool) { return os.Getpid(), true }
|
||||
|
||||
if _, stillAlive := waitForPredecessor(time.Minute, time.Second, nil); stillAlive {
|
||||
t.Fatalf("our own pid must never count as a predecessor")
|
||||
}
|
||||
}
|
||||
@@ -14,34 +14,18 @@ package main
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// pickIfaceProfile returns the highest-Priority ENABLED profile whose MatchIface
|
||||
// matches the active uplink device AND whose schedule window contains now, or nil
|
||||
// when none match. A MatchIface entry matches when it equals activeDev literally
|
||||
// (already a device name) OR when it resolves to activeDev (it was a UCI interface
|
||||
// name, e.g. "wan" -> "eth1"). resolve maps a UCI iface name to its L3 device
|
||||
// (netplane.IfaceDevice in prod; a fake map in tests). Priority ties break by
|
||||
// lowest Name, matching generate's resolveActiveProfile determinism.
|
||||
//
|
||||
// THE SCHEDULE IS A SECOND CONDITION, AND'd with the interface — which is what
|
||||
// model.Profile has always documented ("only the specified ones are checked; all
|
||||
// must hold") and what this function used to ignore. An iface profile with a
|
||||
// 22:00-06:00 window was applied around the clock: the window was accepted,
|
||||
// stored, shown in the panel and inert, which is the same silent-pretending defect
|
||||
// as any other dead knob. A profile with no schedule is unaffected —
|
||||
// model.ProfileScheduleActive returns true for an empty window — so this only
|
||||
// changes configs that asked for a window in the first place.
|
||||
//
|
||||
// The evaluator lives in model (not here, and no longer only in generate) so the
|
||||
// watcher and the code generator cannot drift on what an overnight window or a
|
||||
// weekday token means. Its warnings are dropped here: this runs on a 25s ticker,
|
||||
// and re-logging a bad HH:MM every tick would bury the log — generate reports the
|
||||
// same config on every apply, which is where the operator will see it.
|
||||
func pickIfaceProfile(now time.Time, profiles []model.Profile, activeDev string, resolve func(string) string) *model.Profile {
|
||||
// matches the active uplink device, or nil when none match. A MatchIface entry
|
||||
// matches when it equals activeDev literally (already a device name) OR when it
|
||||
// resolves to activeDev (it was a UCI interface name, e.g. "wan" -> "eth1").
|
||||
// resolve maps a UCI iface name to its L3 device (netplane.IfaceDevice in prod;
|
||||
// a fake map in tests). Priority ties break by lowest Name, matching generate's
|
||||
// resolveActiveProfile determinism.
|
||||
func pickIfaceProfile(profiles []model.Profile, activeDev string, resolve func(string) string) *model.Profile {
|
||||
activeDev = strings.TrimSpace(activeDev)
|
||||
if activeDev == "" {
|
||||
return nil
|
||||
@@ -55,9 +39,6 @@ func pickIfaceProfile(now time.Time, profiles []model.Profile, activeDev string,
|
||||
if !ifaceMatches(p.MatchIface, activeDev, resolve) {
|
||||
continue
|
||||
}
|
||||
if active, _ := model.ProfileScheduleActive(now, *p); !active {
|
||||
continue
|
||||
}
|
||||
if best == nil || p.Priority > best.Priority ||
|
||||
(p.Priority == best.Priority && p.Name < best.Name) {
|
||||
best = p
|
||||
@@ -108,18 +89,13 @@ func currentIsIfaceProfile(profiles []model.Profile, current string) bool {
|
||||
// Rules:
|
||||
// - No ENABLED iface-driven profile exists => no-op (return current unchanged);
|
||||
// the watcher never touches config that has no iface profiles to manage.
|
||||
// - An iface-driven profile matches the uplink AND its schedule window is open
|
||||
// => pin its name (highest priority).
|
||||
// - An iface-driven profile matches the uplink => pin its name (highest
|
||||
// priority).
|
||||
// - No iface-driven profile matches:
|
||||
// - current pin is itself iface-driven (a stale pin whose uplink went away, OR
|
||||
// whose time window has just closed) => release it ("" -> generate falls back
|
||||
// to schedule/auto-select).
|
||||
// - current pin is itself iface-driven (a stale pin whose uplink went away)
|
||||
// => release it ("" -> generate falls back to auto-select).
|
||||
// - otherwise (empty, or a manual pin to a NON-iface profile) => leave alone.
|
||||
//
|
||||
// The release path is what makes an iface profile's schedule work at BOTH edges:
|
||||
// the watcher ticks every 25s, so the window closing looks exactly like the uplink
|
||||
// going away and the pin is dropped within a tick.
|
||||
func desiredActiveProfile(now time.Time, profiles []model.Profile, current, activeDev string, resolve func(string) string) string {
|
||||
func desiredActiveProfile(profiles []model.Profile, current, activeDev string, resolve func(string) string) string {
|
||||
hasIface := false
|
||||
for i := range profiles {
|
||||
if profiles[i].Enabled && len(profiles[i].MatchIface) > 0 {
|
||||
@@ -130,7 +106,7 @@ func desiredActiveProfile(now time.Time, profiles []model.Profile, current, acti
|
||||
if !hasIface {
|
||||
return current
|
||||
}
|
||||
if best := pickIfaceProfile(now, profiles, activeDev, resolve); best != nil {
|
||||
if best := pickIfaceProfile(profiles, activeDev, resolve); best != nil {
|
||||
return best.Name
|
||||
}
|
||||
if currentIsIfaceProfile(profiles, current) {
|
||||
|
||||
@@ -2,17 +2,10 @@ package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// anytime is the clock for the cases that carry NO schedule. An empty schedule
|
||||
// window is always open, so these assertions are about interface matching alone
|
||||
// and the instant is irrelevant — it is named rather than inlined so a reader does
|
||||
// not go looking for significance in the date.
|
||||
var anytime = time.Date(2026, 7, 21, 12, 0, 0, 0, time.UTC)
|
||||
|
||||
// fakeResolve maps UCI iface names to L3 devices (stands in for netplane.IfaceDevice).
|
||||
func fakeResolve(m map[string]string) func(string) string {
|
||||
return func(name string) string {
|
||||
@@ -35,7 +28,7 @@ func TestPickIfaceProfile(t *testing.T) {
|
||||
{Name: "lte", Enabled: true, Priority: 50, MatchIface: []string{"wwan0"}}, // literal device
|
||||
{Name: "lte-hi", Enabled: true, Priority: 90, MatchIface: []string{"wwan"}}, // -> wwan0 (same uplink, higher prio)
|
||||
{Name: "usb", Enabled: false, Priority: 99, MatchIface: []string{"usb0"}}, // disabled
|
||||
{Name: "sched", Enabled: true, Priority: 99, MatchIface: nil}, // not iface-driven
|
||||
{Name: "plain", Enabled: true, Priority: 99, MatchIface: nil}, // not iface-driven
|
||||
}
|
||||
|
||||
cases := []struct {
|
||||
@@ -51,7 +44,7 @@ func TestPickIfaceProfile(t *testing.T) {
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got := pickIfaceProfile(anytime, profiles, c.dev, resolve)
|
||||
got := pickIfaceProfile(profiles, c.dev, resolve)
|
||||
gotName := ""
|
||||
if got != nil {
|
||||
gotName = got.Name
|
||||
@@ -69,7 +62,7 @@ func TestPickIfaceProfileTieBreakByName(t *testing.T) {
|
||||
{Name: "beta", Enabled: true, Priority: 10, MatchIface: []string{"eth1"}},
|
||||
{Name: "alpha", Enabled: true, Priority: 10, MatchIface: []string{"eth1"}},
|
||||
}
|
||||
got := pickIfaceProfile(anytime, profiles, "eth1", resolve)
|
||||
got := pickIfaceProfile(profiles, "eth1", resolve)
|
||||
if got == nil || got.Name != "alpha" {
|
||||
t.Fatalf("tie should break to lowest name 'alpha', got %v", got)
|
||||
}
|
||||
@@ -108,7 +101,7 @@ func TestDesiredActiveProfile(t *testing.T) {
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got := desiredActiveProfile(anytime, c.profiles, c.current, c.dev, resolve)
|
||||
got := desiredActiveProfile(c.profiles, c.current, c.dev, resolve)
|
||||
if got != c.want {
|
||||
t.Fatalf("desiredActiveProfile(current=%q, dev=%q) = %q, want %q", c.current, c.dev, got, c.want)
|
||||
}
|
||||
@@ -137,92 +130,3 @@ func TestParseDefaultRouteDev(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// --- schedule as a second condition ------------------------------------------
|
||||
|
||||
// mondayNoon / mondayNight are inside and outside a 22:00-06:00 overnight window.
|
||||
var (
|
||||
mondayNoon = time.Date(2026, 7, 20, 12, 0, 0, 0, time.UTC)
|
||||
mondayNight = time.Date(2026, 7, 20, 23, 30, 0, 0, time.UTC)
|
||||
)
|
||||
|
||||
// TestIfaceProfileHonoursSchedule is the regression: an iface-driven profile with
|
||||
// a time window used to be pinned around the clock, because the watcher matched on
|
||||
// the uplink and never looked at the schedule. The window was accepted, stored and
|
||||
// displayed while restricting nothing.
|
||||
func TestIfaceProfileHonoursSchedule(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{{
|
||||
Name: "lte-night", Enabled: true, Priority: 10,
|
||||
MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00",
|
||||
}}
|
||||
|
||||
if got := pickIfaceProfile(mondayNight, profiles, "wwan0", resolve); got == nil {
|
||||
t.Fatal("inside the window the profile must be selected")
|
||||
}
|
||||
if got := pickIfaceProfile(mondayNoon, profiles, "wwan0", resolve); got != nil {
|
||||
t.Fatalf("outside the window the profile must NOT be selected, got %q", got.Name)
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileWithoutScheduleIsUnaffected: the change must only touch configs
|
||||
// that asked for a window. An empty schedule is always open.
|
||||
func TestIfaceProfileWithoutScheduleIsUnaffected(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{{Name: "lte", Enabled: true, MatchIface: []string{"wwan0"}}}
|
||||
for _, at := range []time.Time{mondayNoon, mondayNight} {
|
||||
if got := pickIfaceProfile(at, profiles, "wwan0", resolve); got == nil {
|
||||
t.Fatalf("a schedule-less iface profile must always be selectable (at %v)", at)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileScheduleClosingReleasesThePin: the window closing must behave
|
||||
// exactly like the uplink going away, or the profile would stay pinned until the
|
||||
// next reboot — which is how a "night only" profile silently becomes permanent.
|
||||
func TestIfaceProfileScheduleClosingReleasesThePin(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{{
|
||||
Name: "lte-night", Enabled: true, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00",
|
||||
}}
|
||||
if got := desiredActiveProfile(mondayNight, profiles, "", "wwan0", resolve); got != "lte-night" {
|
||||
t.Fatalf("inside the window the pin must be taken, got %q", got)
|
||||
}
|
||||
if got := desiredActiveProfile(mondayNoon, profiles, "lte-night", "wwan0", resolve); got != "" {
|
||||
t.Fatalf("outside the window the stale pin must be released, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileScheduleDoesNotClobberAManualPin: a user's manual pin to a
|
||||
// NON-iface profile is still none of the watcher's business, window or not.
|
||||
func TestIfaceProfileScheduleDoesNotClobberAManualPin(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{
|
||||
{Name: "lte-night", Enabled: true, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00"},
|
||||
{Name: "manual", Enabled: true},
|
||||
}
|
||||
if got := desiredActiveProfile(mondayNoon, profiles, "manual", "wwan0", resolve); got != "manual" {
|
||||
t.Fatalf("a manual pin to a non-iface profile must be left alone, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileScheduleTieBreak: with two matching iface profiles, only the one
|
||||
// whose window is OPEN may win — priority must not resurrect a closed profile.
|
||||
func TestIfaceProfileScheduleTieBreak(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{
|
||||
{Name: "night-hi", Enabled: true, Priority: 90, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00"},
|
||||
{Name: "day-lo", Enabled: true, Priority: 10, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "06:00", SchedEnd: "22:00"},
|
||||
}
|
||||
if got := pickIfaceProfile(mondayNoon, profiles, "wwan0", resolve); got == nil || got.Name != "day-lo" {
|
||||
t.Fatalf("at noon the open low-priority profile must win, got %v", got)
|
||||
}
|
||||
if got := pickIfaceProfile(mondayNight, profiles, "wwan0", resolve); got == nil || got.Name != "night-hi" {
|
||||
t.Fatalf("at night the open high-priority profile must win, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ package devices
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
@@ -46,12 +47,13 @@ func findRows(rows []Discovered, mac string) []Discovered {
|
||||
}
|
||||
|
||||
// TestDiscoverDualStackSameMAC is the audit's "one MAC, a v4 and a v6 address"
|
||||
// combination. Discovery is keyed by IP, so a dual-stacked client legitimately
|
||||
// produces TWO rows — this test pins the CONTRACT that matters to the panel and to
|
||||
// per-device stats: both rows carry the SAME MAC, and if that MAC is a configured
|
||||
// device then BOTH rows are marked configured with the same name/blockCount. A row
|
||||
// that silently lost its configured status would show up in the UI as an unmanaged
|
||||
// device that quietly escapes its parental-control rules.
|
||||
// combination. Discovery merges by MAC, so a dual-stacked client is ONE row —
|
||||
// this test pins the CONTRACT that matters to the panel and to per-device
|
||||
// stats: the row carries BOTH addresses in ips (primary first), and if that MAC
|
||||
// is a configured device the merged row is marked configured with its
|
||||
// name/blockCount. A device that split into rows (or lost its configured
|
||||
// status) would show up in the UI as an unmanaged duplicate that quietly
|
||||
// escapes its parental-control rules.
|
||||
func TestDiscoverDualStackSameMAC(t *testing.T) {
|
||||
const mac = "aa:bb:cc:dd:ee:01"
|
||||
writeNetwork(t, lanOnlyNetwork)
|
||||
@@ -71,26 +73,34 @@ func TestDiscoverDualStackSameMAC(t *testing.T) {
|
||||
}}
|
||||
rows := Discover(configured)
|
||||
|
||||
// The configured device must NOT additionally appear as a phantom offline
|
||||
// row: it was seen, so exactly one row carries that MAC.
|
||||
got := findRows(rows, mac)
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("dual-stack host produced %d rows, want 2 (one per address): %+v", len(got), rows)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("dual-stack host produced %d rows, want 1 merged row: %+v", len(got), rows)
|
||||
}
|
||||
for _, r := range got {
|
||||
if !r.Configured {
|
||||
t.Errorf("row %s (mac %s) is not marked configured — it would escape its device rules", r.IP, r.MAC)
|
||||
}
|
||||
if r.Name != "Phone" {
|
||||
t.Errorf("row %s: Name = %q, want Phone", r.IP, r.Name)
|
||||
}
|
||||
if r.BlockCount != 2 {
|
||||
t.Errorf("row %s: BlockCount = %d, want 2", r.IP, r.BlockCount)
|
||||
}
|
||||
r := got[0]
|
||||
if r.IP != "192.168.1.50" {
|
||||
t.Fatalf("primary must be the leased v4 address, got %q", r.IP)
|
||||
}
|
||||
|
||||
// The configured device must NOT additionally appear as a phantom offline row:
|
||||
// it was seen, so exactly the two live rows carry that MAC.
|
||||
if n := len(findRows(rows, mac)); n != 2 {
|
||||
t.Fatalf("configured-but-unseen fallback added a phantom row: %d rows for mac %s", n, mac)
|
||||
want := []string{"192.168.1.50", "fe80::a8bb:ccff:fedd:ee01"}
|
||||
if !reflect.DeepEqual(r.IPs, want) {
|
||||
t.Fatalf("ips = %+v, want %+v (primary first)", r.IPs, want)
|
||||
}
|
||||
if r.State != "online" || !r.Online {
|
||||
t.Fatalf("REACHABLE v4 must win the merged state, got %+v", r)
|
||||
}
|
||||
if !r.Configured {
|
||||
t.Errorf("merged row (mac %s) is not marked configured — it would escape its device rules", r.MAC)
|
||||
}
|
||||
if r.Name != "Phone" {
|
||||
t.Errorf("Name = %q, want Phone", r.Name)
|
||||
}
|
||||
if r.BlockCount != 2 {
|
||||
t.Errorf("BlockCount = %d, want 2", r.BlockCount)
|
||||
}
|
||||
if r.Network != "lan" || r.Iface != "br-lan" {
|
||||
t.Errorf("merged row not labelled from its primary address: %+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,7 +190,8 @@ func TestDiscoverIsDeterministic(t *testing.T) {
|
||||
t.Fatalf("run %d returned %d rows, first run returned %d", i, len(got), len(first))
|
||||
}
|
||||
for j := range got {
|
||||
if got[j] != first[j] {
|
||||
// reflect.DeepEqual: Discovered carries a slice (IPs) so == is out.
|
||||
if !reflect.DeepEqual(got[j], first[j]) {
|
||||
t.Fatalf("run %d row %d differs from the first run:\n got %+v\nwant %+v", i, j, got[j], first[j])
|
||||
}
|
||||
}
|
||||
|
||||
+104
-21
@@ -98,7 +98,11 @@ func MACToIP(path string) map[string]string {
|
||||
// Discovered is one row of GET /api/devices: a host seen on the LAN (via lease
|
||||
// and/or neighbour table) cross-referenced with the configured devices.
|
||||
type Discovered struct {
|
||||
IP string `json:"ip"`
|
||||
IP string `json:"ip"` // primary address (most recent lease, else "most online")
|
||||
// IPs is every address currently known for this device (IPv4+IPv6, multiple
|
||||
// leases), primary first. Never nil; empty only for a configured-but-unseen
|
||||
// device whose config entry carries no IP.
|
||||
IPs []string `json:"ips"`
|
||||
MAC string `json:"mac"`
|
||||
Hostname string `json:"hostname"`
|
||||
Online bool `json:"online"` // reachable or recently-seen (state != offline)
|
||||
@@ -123,14 +127,18 @@ type neigh struct {
|
||||
State string // online|idle|offline
|
||||
}
|
||||
|
||||
// Discover merges the DHCP lease table with `ip neigh show`, cross-references the
|
||||
// configured devices, and returns one row per discovered host (union of leases +
|
||||
// neighbours), plus any configured device that was not otherwise seen (as an
|
||||
// offline row) so the panel can always render the full parental-control list.
|
||||
// Discover merges the DHCP lease table with `ip neigh show`, folds every address
|
||||
// sharing a MAC into ONE row per physical device (a dual-stacked or multi-leased
|
||||
// client is still one device), cross-references the configured devices, and
|
||||
// appends any configured device that was not otherwise seen (as an offline row)
|
||||
// so the panel can always render the full parental-control list.
|
||||
// Never returns nil.
|
||||
func Discover(configured []model.Device) []Discovered {
|
||||
// Index by IP, seeding from the lease table (ip/mac/hostname).
|
||||
// Phase 1: aggregate per-IP, seeding from the lease table (ip/mac/hostname).
|
||||
// The MAC-level merge happens only after every signal is in, so a MAC that
|
||||
// arrives late (from the neighbour table) still groups its earlier lease IPs.
|
||||
byIP := map[string]*Discovered{}
|
||||
leaseSeq := map[string]int{} // IP -> index of its LAST lease line; absent = neigh-only
|
||||
order := []string{}
|
||||
add := func(ip string) *Discovered {
|
||||
if d, ok := byIP[ip]; ok {
|
||||
@@ -141,7 +149,7 @@ func Discover(configured []model.Device) []Discovered {
|
||||
order = append(order, ip)
|
||||
return d
|
||||
}
|
||||
for _, l := range ParseLeases(LeasesPath) {
|
||||
for seq, l := range ParseLeases(LeasesPath) {
|
||||
d := add(l.IP)
|
||||
if d.MAC == "" {
|
||||
d.MAC = l.MAC
|
||||
@@ -149,6 +157,7 @@ func Discover(configured []model.Device) []Discovered {
|
||||
if d.Hostname == "" {
|
||||
d.Hostname = l.Hostname
|
||||
}
|
||||
leaseSeq[l.IP] = seq // later lease lines are more recent
|
||||
}
|
||||
|
||||
// Derive the LAN device set (and WAN devices) from the network config so the
|
||||
@@ -171,28 +180,98 @@ func Discover(configured []model.Device) []Discovered {
|
||||
if d.Iface == "" {
|
||||
d.Iface = n.Dev
|
||||
}
|
||||
// Prefer the "most online" state if a host somehow appears twice.
|
||||
// Prefer the "most online" state if an address somehow appears twice.
|
||||
if stateRank(n.State) > stateRank(d.State) {
|
||||
d.State = n.State
|
||||
}
|
||||
}
|
||||
|
||||
// Cross-reference the configured devices (match by MAC first, else IP). Track
|
||||
// which configured entries matched so unseen ones can be appended as offline.
|
||||
matched := make([]bool, len(configured))
|
||||
for _, d := range byIP {
|
||||
if i, ok := matchConfigured(configured, d.MAC, d.IP); ok {
|
||||
markConfigured(d, configured[i])
|
||||
matched[i] = true
|
||||
// Phase 2: group addresses by lower-cased MAC — one group per physical
|
||||
// device. Addresses with NO known MAC stay one-per-IP: with no hardware
|
||||
// identity there is nothing safe to merge on. Groups keep first-seen order
|
||||
// (lease-file order, then neighbour order) so output is deterministic.
|
||||
type group struct {
|
||||
primary *Discovered // the address the row's ip/iface/network come from
|
||||
members []*Discovered
|
||||
}
|
||||
groups := make([]*group, 0, len(order))
|
||||
byMAC := map[string]*group{}
|
||||
// morePrimary reports whether a should displace cur as a group's primary:
|
||||
// the most recent lease wins (mirrors MACToIP, so the panel and the generate
|
||||
// stage agree on a device's "current" IP), then the "more online" address;
|
||||
// otherwise the first-seen address keeps the slot.
|
||||
morePrimary := func(a, cur *Discovered) bool {
|
||||
as, aok := leaseSeq[a.IP]
|
||||
cs, cok := leaseSeq[cur.IP]
|
||||
if aok != cok {
|
||||
return aok
|
||||
}
|
||||
if aok && as != cs {
|
||||
return as > cs
|
||||
}
|
||||
return stateRank(a.State) > stateRank(cur.State)
|
||||
}
|
||||
for _, ip := range order {
|
||||
d := byIP[ip]
|
||||
mac := strings.ToLower(d.MAC)
|
||||
if mac == "" {
|
||||
groups = append(groups, &group{primary: d, members: []*Discovered{d}})
|
||||
continue
|
||||
}
|
||||
g, ok := byMAC[mac]
|
||||
if !ok {
|
||||
g = &group{primary: d, members: []*Discovered{d}}
|
||||
byMAC[mac] = g
|
||||
groups = append(groups, g)
|
||||
continue
|
||||
}
|
||||
g.members = append(g.members, d)
|
||||
if morePrimary(d, g.primary) {
|
||||
g.primary = d
|
||||
}
|
||||
}
|
||||
|
||||
out := make([]Discovered, 0, len(order)+len(configured))
|
||||
for _, ip := range order {
|
||||
d := byIP[ip]
|
||||
d.Online = d.State != "offline"
|
||||
labelNetwork(d, nets)
|
||||
out = append(out, *d)
|
||||
// Phase 3: flatten each group into one row and cross-reference the
|
||||
// configured devices (MAC first, else primary IP). Track which configured
|
||||
// entries matched so unseen ones can be appended as offline rows.
|
||||
matched := make([]bool, len(configured))
|
||||
out := make([]Discovered, 0, len(groups)+len(configured))
|
||||
for _, g := range groups {
|
||||
row := *g.primary
|
||||
row.IPs = make([]string, 0, len(g.members))
|
||||
row.IPs = append(row.IPs, g.primary.IP)
|
||||
for _, m := range g.members {
|
||||
if m == g.primary {
|
||||
continue
|
||||
}
|
||||
row.IPs = append(row.IPs, m.IP)
|
||||
// Best state among the device's addresses wins the merged row.
|
||||
if stateRank(m.State) > stateRank(row.State) {
|
||||
row.State = m.State
|
||||
}
|
||||
}
|
||||
// Hostname: first non-empty in first-seen order (the primary may be a
|
||||
// hostname-less neighbour row while an older lease knew the name).
|
||||
for _, m := range g.members {
|
||||
if m.Hostname != "" {
|
||||
row.Hostname = m.Hostname
|
||||
break
|
||||
}
|
||||
}
|
||||
// Iface follows the primary; fall back to any member that knows it.
|
||||
for _, m := range g.members {
|
||||
if row.Iface != "" {
|
||||
break
|
||||
}
|
||||
row.Iface = m.Iface
|
||||
}
|
||||
row.Online = row.State != "offline"
|
||||
labelNetwork(&row, nets)
|
||||
if i, ok := matchConfigured(configured, row.MAC, row.IP); ok {
|
||||
markConfigured(&row, configured[i])
|
||||
matched[i] = true
|
||||
}
|
||||
out = append(out, row)
|
||||
}
|
||||
// Append configured-but-unseen devices as offline rows (identity from config).
|
||||
for i, cd := range configured {
|
||||
@@ -201,8 +280,12 @@ func Discover(configured []model.Device) []Discovered {
|
||||
}
|
||||
row := Discovered{
|
||||
IP: strings.TrimSpace(cd.IP), MAC: strings.ToLower(strings.TrimSpace(cd.MAC)),
|
||||
IPs: []string{},
|
||||
State: "offline", Online: false,
|
||||
}
|
||||
if row.IP != "" {
|
||||
row.IPs = append(row.IPs, row.IP)
|
||||
}
|
||||
markConfigured(&row, cd)
|
||||
labelNetwork(&row, nets)
|
||||
out = append(out, row)
|
||||
|
||||
@@ -102,6 +102,9 @@ func TestDiscoverMerge(t *testing.T) {
|
||||
if c.MAC != "77:88:99:aa:bb:cc" || c.State != "online" || c.Configured {
|
||||
t.Fatalf("neigh-only host wrong: %+v", c)
|
||||
}
|
||||
if len(c.IPs) != 1 || c.IPs[0] != "192.168.1.99" {
|
||||
t.Fatalf("single-address row must carry ips=[its ip], got %+v", c.IPs)
|
||||
}
|
||||
|
||||
// The away phone (configured, unseen) must appear as an offline row.
|
||||
var away *Discovered
|
||||
@@ -118,6 +121,46 @@ func TestDiscoverMerge(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestDiscoverMergesLeasesSameMAC proves one physical device with several
|
||||
// addresses (two leases on the same MAC, e.g. after a re-lease) yields ONE row
|
||||
// whose ips carries every address, with the most recent lease as the primary.
|
||||
func TestDiscoverMergesLeasesSameMAC(t *testing.T) {
|
||||
const mac = "aa:bb:cc:dd:ee:ff"
|
||||
leases := "" +
|
||||
"1700000000 " + mac + " 192.168.1.50 kidpc 01:aa\n" +
|
||||
"1700000100 " + mac + " 192.168.1.77 kidpc 01:aa\n"
|
||||
LeasesPath = writeLeases(t, leases)
|
||||
|
||||
// Only the old address is in the neighbour table (REACHABLE): the merged
|
||||
// row must still be online even though the PRIMARY (newest lease) is not.
|
||||
defer fakeNeigh(t, "192.168.1.50 dev br-lan lladdr "+mac+" REACHABLE\n")()
|
||||
|
||||
rows := Discover(nil)
|
||||
|
||||
var got []Discovered
|
||||
for _, r := range rows {
|
||||
if r.MAC == mac {
|
||||
got = append(got, r)
|
||||
}
|
||||
}
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("same-MAC leases must merge into one row, got %d: %+v", len(got), rows)
|
||||
}
|
||||
d := got[0]
|
||||
if d.IP != "192.168.1.77" {
|
||||
t.Fatalf("primary must be the most recent lease IP, got %q", d.IP)
|
||||
}
|
||||
if len(d.IPs) != 2 || d.IPs[0] != "192.168.1.77" || d.IPs[1] != "192.168.1.50" {
|
||||
t.Fatalf("ips must list every address primary-first, got %+v", d.IPs)
|
||||
}
|
||||
if d.State != "online" || !d.Online {
|
||||
t.Fatalf("best state among addresses must win the merge, got %+v", d)
|
||||
}
|
||||
if d.Hostname != "kidpc" {
|
||||
t.Fatalf("hostname lost in merge: %+v", d)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDiscoverDropsWANNeigh proves the neighbour table is filtered to LAN
|
||||
// devices and each survivor is labelled with its OpenWrt network: a WAN-side
|
||||
// slirp row (dev eth1, 10.0.2.2) is dropped while the LAN-side row (dev br-lan,
|
||||
|
||||
+16
-27
@@ -71,33 +71,21 @@ type Engine struct {
|
||||
lastGood option.Options // last known-good options (the config running BEFORE current)
|
||||
hasLastGood bool // false until a second successful Apply gives us a predecessor
|
||||
|
||||
// Manual "Test all nodes" probe-all state (see probeall.go). These are guarded by
|
||||
// their own atomics — INDEPENDENT of mu — so a status poll never blocks behind an
|
||||
// Apply, and a run may proceed concurrently with reads. testRunning is the
|
||||
// singleton guard (only one run at a time); testDone/testTotal report progress.
|
||||
testRunning atomic.Bool
|
||||
testDone atomic.Int64
|
||||
testTotal atomic.Int64
|
||||
// log is the engine's own logger. Today its one consumer is the observatory's
|
||||
// verdict-flip trail (observatory.go probeOneInto) — the single diagnostic
|
||||
// record of a node going alive<->dead. Set once in New, never nil there; code
|
||||
// paths reached from a hand-built Engine{} must nil-check it.
|
||||
log log.ContextLogger
|
||||
|
||||
// dead is our OWN overlay of "this outbound was probed and did not answer",
|
||||
// tag -> time of that failed probe (see health.go for why this is not written into
|
||||
// the engine's urltest history). It lives on the Engine, not on a Box, so — like
|
||||
// the pre-registered history store — it survives an Apply swap instead of turning
|
||||
// every known-dead node back into "untested" on every apply. Guarded by its own
|
||||
// leaf mutex, INDEPENDENT of mu, so a health read never blocks behind an Apply.
|
||||
deadMu sync.Mutex
|
||||
dead map[string]time.Time
|
||||
// Observatory state (see observatory.go): the background prober that keeps
|
||||
// the health board fresh for everything the routing rules reach. Its own leaf
|
||||
// mutex, independent of mu, so a tick never blocks an Apply and vice versa.
|
||||
obsMu sync.Mutex
|
||||
obs observatoryState
|
||||
|
||||
// Scheduled sweep state (see sweep.go): the background observatory that keeps node
|
||||
// health fresh so every strategy has current data to select on. Its own leaf mutex,
|
||||
// independent of mu, so a tick never blocks an Apply and vice versa.
|
||||
sweepMu sync.Mutex
|
||||
sweep sweepState
|
||||
|
||||
// Per-group test state (see grouptest.go). Same shape and same reasoning as the
|
||||
// probe-all state above — independent of mu so a progress poll never blocks
|
||||
// behind an Apply — plus a short leaf lock for the results slice, which is a
|
||||
// value the atomics cannot carry.
|
||||
// Per-group test state (see grouptest.go). Guarded by its own atomics and a
|
||||
// short leaf lock (for the results slice, a value the atomics cannot carry) —
|
||||
// independent of mu so a progress poll never blocks behind an Apply.
|
||||
groupTestRunning atomic.Bool
|
||||
groupTestDone atomic.Int64
|
||||
groupTestTotal atomic.Int64
|
||||
@@ -111,7 +99,8 @@ type Engine struct {
|
||||
|
||||
// New builds the engine context (with the shater-owned slim protocol/dns/
|
||||
// service registries wired via registry.Context) and returns a stopped Engine.
|
||||
// An optional logger may be supplied for the deprecated-feature manager;
|
||||
// An optional logger may be supplied for the deprecated-feature manager and the
|
||||
// observatory's verdict-flip trail;
|
||||
// log.StdLogger() is used by default. The variadic mirrors a "logFactory"-style
|
||||
// optional argument.
|
||||
func New(logger ...log.ContextLogger) *Engine {
|
||||
@@ -151,7 +140,7 @@ func New(logger ...log.ContextLogger) *Engine {
|
||||
// URLTestHistory() below. The pointer is stable across Apply swaps, so health
|
||||
// history survives config changes instead of being reset on every apply.
|
||||
ctx = service.ContextWithPtr(ctx, urltest.NewHistoryStorage())
|
||||
return &Engine{ctx: ctx}
|
||||
return &Engine{ctx: ctx, log: l}
|
||||
}
|
||||
|
||||
// Apply installs opts as the running configuration.
|
||||
|
||||
+121
-39
@@ -25,18 +25,18 @@ import (
|
||||
//
|
||||
// # Where the truth already is (no new probing)
|
||||
//
|
||||
// Nothing here dials anything. Every number below is a projection of what the engine
|
||||
// has ALREADY collected, read through one HealthView (health.go) — the shared
|
||||
// urltest.HistoryStorage for measurements plus our own overlay of failure verdicts,
|
||||
// both process-global and stable across Apply swaps:
|
||||
// Nothing here dials anything. Every number below is a projection of the shared
|
||||
// health board (common/urltest), read through one HealthView (health.go) —
|
||||
// process-global and stable across Apply swaps:
|
||||
//
|
||||
// - a urltest-strategy group probes its OWN member outbounds on its interval and
|
||||
// writes the result under the member's tag (protocol/group/urltest.go testNodes →
|
||||
// StoreURLTestHistory(realTag)). For an egress-bound group those member tags ARE
|
||||
// the "group-…" copies, so the per-copy truth is already in the store — we simply
|
||||
// never looked at it, because we only ever projected base node tags;
|
||||
// - the manual "Test all nodes" run (probeall.go) probes every tag including those
|
||||
// copies, and is the ONLY thing that can record a FAILURE.
|
||||
// writes the result under the member's tag (protocol/group/urltest.go testNodes).
|
||||
// For an egress-bound group those member tags ARE the "group-…" copies, so the
|
||||
// per-copy truth is already in the store;
|
||||
// - the observatory (observatory.go) probes everything the routing rules reach —
|
||||
// members, egress copies, chain exits — on the global probe interval;
|
||||
// - a failure, whoever finds it (group checker, observatory, failed user dial),
|
||||
// is recorded with MarkFailed and never by deletion.
|
||||
//
|
||||
// So the entire feature is a read. The cost of one GroupHealth call is
|
||||
// len(groups) x len(members) map lookups — cheap enough to serve on a few-second panel
|
||||
@@ -44,21 +44,20 @@ import (
|
||||
//
|
||||
// # "dead" vs "untested" (the honesty requirement)
|
||||
//
|
||||
// A urltest group DELETES a member's history entry when its probe fails
|
||||
// (urltest.go:512 DeleteURLTestHistory). So from the group's own probing, "probed and
|
||||
// failed" and "never probed" are the SAME observation: no entry. We must not paper
|
||||
// over that — a group whose members were never probed must not look healthy. The one
|
||||
// thing that CAN distinguish them is TestAllNodes, whose failure verdicts live in the
|
||||
// overlay HealthView consults. decodeHealth (health.go) is the single place those
|
||||
// rules are written down; this file only counts what it returns.
|
||||
// The board keeps successes AND failures per tag, and a verdict is computed on
|
||||
// read against a TTL: dead is always a POSITIVE finding (a probe or dial ran and
|
||||
// failed, recently), untested is "nothing fresh enough is known" — including a
|
||||
// once-alive record that aged past the TTL. decodeHealth (health.go) is the
|
||||
// single place those rules are written down; this file only counts what it
|
||||
// returns.
|
||||
//
|
||||
// # Why so many members are legitimately "untested"
|
||||
// # Why "untested" still legitimately exists
|
||||
//
|
||||
// A urltest group only probes while it is BEING USED: Touch() starts the ticker and an
|
||||
// idle group stops it (urltest.go Touch/performUpdateCheck). A selector group never
|
||||
// probes at all — it dials one fixed member (selector.go). On a 376-node subscription
|
||||
// most members therefore have no history and never will. That is a true statement about
|
||||
// what the router knows, and it is what the panel must show.
|
||||
// The observatory probes only what the routing rules can reach (plan §5.C). A
|
||||
// group no enabled rule routes through is deliberately never probed — its
|
||||
// members stay untested and the group is reported Used=false, which the panel
|
||||
// renders as "unused" rather than as a health problem. Members of a USED group
|
||||
// fill in within seconds of an apply (immediate first observatory pass).
|
||||
|
||||
// GroupMemberHealth is one member of one group, as that group sees it.
|
||||
//
|
||||
@@ -102,18 +101,17 @@ type GroupMemberHealth struct {
|
||||
// fallback 8 / 8 alive tested 8 of 24
|
||||
// stealth 2 / 2 alive
|
||||
//
|
||||
// Untested is deliberately NOT a third equal column, and must NEVER be folded into
|
||||
// Dead. A urltest group probes its members LAZILY — only while it is being used — so
|
||||
// on a freshly booted router with a 376-node subscription the history is nearly empty
|
||||
// by design. Counting no-entry as dead would render "3 alive / 373 dead" at exactly
|
||||
// the moment nothing is wrong. A reading that alarms without cause teaches the
|
||||
// operator to distrust every reading, which is worse than no reading.
|
||||
// Untested is deliberately NOT a third equal column, and must NEVER be folded
|
||||
// into Dead. An UNUSED group (Used=false) is never probed at all, and a used
|
||||
// group's numbers may lag a few seconds behind an apply (the observatory's first
|
||||
// pass) or age out on the health TTL. Counting no-fresh-data as dead would raise
|
||||
// an alarm at exactly the moment nothing is wrong, and a reading that alarms
|
||||
// without cause teaches the operator to distrust every reading.
|
||||
//
|
||||
// Dead is only ever a POSITIVE finding: a probe ran and failed, recorded by the manual
|
||||
// "Test all nodes" run. Absence of a measurement proves nothing — the group deletes the
|
||||
// entry when its own probe fails, so "never probed" and "probed and failed" collapse
|
||||
// into the same absence, and only that manual run can tell them apart. Anything with
|
||||
// neither a measurement nor a verdict is Untested, full stop.
|
||||
// Dead is only ever a POSITIVE finding: a probe or dial ran and FAILED recently,
|
||||
// recorded on the shared health board by the observatory, the group's own
|
||||
// checker, or a failed user dial. Absence of fresh data proves nothing — it is
|
||||
// Untested, full stop.
|
||||
type GroupHealth struct {
|
||||
Group string `json:"group"` // the group's outbound tag == its configured name
|
||||
Type string `json:"type"` // "selector" | "urltest" (engine group type)
|
||||
@@ -122,6 +120,13 @@ type GroupHealth struct {
|
||||
// The panel should say so: these numbers are not comparable with the global
|
||||
// per-node numbers, and deliberately so.
|
||||
Bound bool `json:"bound"`
|
||||
// Used is false when the group is reachable from NO enabled routing rule
|
||||
// (nor Final, nor a DNS detour) — it is outside the observatory's plan, so
|
||||
// nothing probes it and its members stay untested by design. The panel
|
||||
// renders such a group "unused" instead of showing health counters. Reported
|
||||
// true when the observatory is disabled or not yet configured: no badge is
|
||||
// better than a wrong one.
|
||||
Used bool `json:"used"`
|
||||
// Selected is the NODE NAME the group currently routes through (adapter
|
||||
// OutboundGroup.Now(), mapped back through the copy tag). "" when the group has
|
||||
// not selected anything yet.
|
||||
@@ -176,8 +181,10 @@ func (e *Engine) GroupHealth(withMembers bool) []GroupHealth {
|
||||
}
|
||||
// ONE health view for the whole projection: every member of every group is decoded
|
||||
// against the same instant, so a 3-group / 376-member config cannot report counters
|
||||
// that disagree with each other because the store moved underneath them.
|
||||
// that disagree with each other because the store moved underneath them. The
|
||||
// used-set is the one the observatory published for the running plan.
|
||||
view := e.HealthView()
|
||||
used := e.observatoryUsed()
|
||||
for _, ob := range om.Outbounds() {
|
||||
if !isGroupOutbound(ob.Type(), ob.Tag()) {
|
||||
continue
|
||||
@@ -188,7 +195,7 @@ func (e *Engine) GroupHealth(withMembers bool) []GroupHealth {
|
||||
// asked for its members; there is nothing honest to report about it.
|
||||
continue
|
||||
}
|
||||
out = append(out, groupHealthOf(g, ob.Type(), view, withMembers))
|
||||
out = append(out, groupHealthOf(g, ob.Type(), view, used, withMembers))
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -213,9 +220,11 @@ func (e *Engine) GroupHealthOne(name string) (GroupHealth, bool) {
|
||||
|
||||
// groupHealthOf builds one group's health row from the shared history. Pure apart
|
||||
// from the history reads, so it is unit-testable against a hand-built store.
|
||||
func groupHealthOf(g adapter.OutboundGroup, typ string, view HealthView, withMembers bool) GroupHealth {
|
||||
// used is the observatory's published used-set; nil means "unknown", reported as
|
||||
// Used=true (see GroupHealth.Used).
|
||||
func groupHealthOf(g adapter.OutboundGroup, typ string, view HealthView, used map[string]bool, withMembers bool) GroupHealth {
|
||||
name := g.Tag()
|
||||
gh := GroupHealth{Group: name, Type: typ, FreshestSeconds: -1}
|
||||
gh := GroupHealth{Group: name, Type: typ, FreshestSeconds: -1, Used: used == nil || used[name]}
|
||||
|
||||
members := g.All()
|
||||
gh.Total = len(members)
|
||||
@@ -287,7 +296,7 @@ func groupHealthOf(g adapter.OutboundGroup, typ string, view HealthView, withMem
|
||||
// whose name happens to contain "-m<digits>-" cannot confuse the split. The residual
|
||||
// edge case is a real node literally NAMED "group-<thisgroup>-m0-x" inside that same
|
||||
// group, which would be read as a copy of node "x": the "group-"/"chain-"/"egress-"
|
||||
// prefixes are documented reserved tag namespaces (see shouldProbeOutbound), and the
|
||||
// prefixes are documented reserved tag namespaces (see probeablePlanTag), and the
|
||||
// generator already refuses copies that collide with a real node.
|
||||
func parseGroupCopyTag(group, tag string) (member string, ok bool) {
|
||||
prefix := "group-" + group + "-m"
|
||||
@@ -310,3 +319,76 @@ func parseGroupCopyTag(group, tag string) (member string, ok bool) {
|
||||
}
|
||||
return member, true
|
||||
}
|
||||
|
||||
// ChainHealth is one configured chain's reachability, mirroring GroupHealth.Used
|
||||
// for the chain card (plan §5.E): a chain no enabled rule routes through is never
|
||||
// probed — the observatory walks only reachable paths — and the panel renders it
|
||||
// "unused" rather than as a health problem. A chain has no membership counters: it
|
||||
// is a fixed path, and its end-to-end health is the exit test's job, not a roll-up.
|
||||
type ChainHealth struct {
|
||||
// Name is the chain's model name (config chain "chain:<name>"), what the Targets
|
||||
// page lists and what a rule targets.
|
||||
Name string `json:"name"`
|
||||
// Used is false when the chain is reachable from NO enabled routing rule (nor
|
||||
// Final, nor a DNS detour) — it is outside the observatory's plan, so its exit
|
||||
// is never probed and its end-to-end health stays untested by design. The panel
|
||||
// renders such a chain "unused" instead of an exit-test readout. Reported true
|
||||
// when the observatory is disabled or not yet configured: no badge is better
|
||||
// than a wrong one (the same rule as GroupHealth.Used).
|
||||
Used bool `json:"used"`
|
||||
}
|
||||
|
||||
// ChainHealth reports the reachability (used/unused) of every named chain, one row
|
||||
// per name, in the order given. It is the chain analogue of GroupHealth.Used: a
|
||||
// chain the observatory's used-set does not cover is reported Used=false so the
|
||||
// panel can mark it "unused" instead of running an exit test against a path nothing
|
||||
// routes through.
|
||||
//
|
||||
// names come from the desired-state model, NOT the running box: a chain no rule
|
||||
// references is never materialised (generate/chain.go resolveChain is lazy), so it
|
||||
// is invisible to a box-only enumeration — yet the panel lists it from the config
|
||||
// and must be able to badge it. The engine supplies the only fact a box read can
|
||||
// add here, the observatory's published used-set. Pure apart from that read; nil
|
||||
// names or a stopped engine (nil used-set) yield an empty/used-everything result.
|
||||
func (e *Engine) ChainHealth(names []string) []ChainHealth {
|
||||
out := make([]ChainHealth, 0, len(names))
|
||||
used := e.observatoryUsed()
|
||||
usedChains := usedChainNames(used)
|
||||
for _, name := range names {
|
||||
name = strings.TrimSpace(name)
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
out = append(out, ChainHealth{Name: name, Used: used == nil || usedChains[name]})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// usedChainNames recovers the set of chain NAMES the observatory's used-set covers.
|
||||
//
|
||||
// The used-set is keyed by outbound TAG (the generator's schema, materialised in
|
||||
// opts); a chain's exit wrapper is tagged "chain-<name>-h<digits>" (generate/chain.go
|
||||
// buildHopWrapper), so parseChainExitTag (grouptest.go) inverts that exactly. Member
|
||||
// copies ("chain-<name>-h<i>-<member>") and non-chain tags parse false and are
|
||||
// skipped. nil used-set ⇒ nil (the caller treats nil as "everything used" — no badge
|
||||
// is better than a wrong one, the same convention as GroupHealth.Used).
|
||||
//
|
||||
// A single-hop chain WITHOUT an egress entry never gets a wrapper (buildChain
|
||||
// resolves it straight to the underlying node/group tag), so no chain- tag exists
|
||||
// for it and it cannot be recovered here — it reads Used=false even when a rule
|
||||
// targets it. That mirrors the exit test (chainTargetsFrom), which likewise cannot
|
||||
// discover such a chain and reports it missing when named; a 1-hop chain is an alias
|
||||
// for the hop it names, and its reachability is that hop's, reported on the hop's
|
||||
// own card.
|
||||
func usedChainNames(used map[string]bool) map[string]bool {
|
||||
if used == nil {
|
||||
return nil
|
||||
}
|
||||
out := make(map[string]bool, len(used))
|
||||
for tag := range used {
|
||||
if name, ok := parseChainExitTag(tag); ok {
|
||||
out[name] = true
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -91,16 +91,17 @@ func TestParseGroupCopyTag(t *testing.T) {
|
||||
func TestGroupHealthUnbound(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{Time: now.Add(-time.Minute), Delay: 100})
|
||||
// n2 is DEAD: a failed probe leaves nothing in the engine's history and is recorded
|
||||
// only in our overlay (see health.go — an invented history entry would be read as
|
||||
// "alive" by the round_robin pool planner).
|
||||
dead := map[string]time.Time{"n2": now.Add(-10 * time.Second)}
|
||||
// n3: nothing at all — never probed, or its group probe failed and the engine
|
||||
// deleted the entry. Indistinguishable, therefore: untested.
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{LastOK: now.Add(-time.Minute), Delay: 100})
|
||||
// n2 is DEAD: a failed probe/dial marks LastFail on the board record — nothing
|
||||
// is deleted, so a dead member stays distinguishable from a never-measured one.
|
||||
hist.StoreURLTestHistory("n2", &adapter.URLTestHistory{LastFail: now.Add(-10 * time.Second)})
|
||||
// n3: nothing at all — never probed (outside every used path). Untested.
|
||||
|
||||
g := &fakeGroup{tag: "auto", kind: C.TypeURLTest, all: []string{"n1", "n2", "n3"}, now: "n1"}
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, dead, now), true)
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, healthTTLFloor, now), nil, true)
|
||||
if !gh.Used {
|
||||
t.Error("a nil used-set must report Used=true (no badge is better than a wrong one)")
|
||||
}
|
||||
|
||||
if gh.Bound {
|
||||
t.Error("unbound group reported bound=true")
|
||||
@@ -139,16 +140,17 @@ func TestGroupHealthBound(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
// Base tags say everything is fine (the nodes answer on the plain WAN)...
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{Time: now, Delay: 50})
|
||||
hist.StoreURLTestHistory("n2", &adapter.URLTestHistory{Time: now, Delay: 60})
|
||||
hist.StoreURLTestHistory("n3", &adapter.URLTestHistory{Time: now, Delay: 70})
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{LastOK: now, Delay: 50})
|
||||
hist.StoreURLTestHistory("n2", &adapter.URLTestHistory{LastOK: now, Delay: 60})
|
||||
hist.StoreURLTestHistory("n3", &adapter.URLTestHistory{LastOK: now, Delay: 70})
|
||||
// ...but THROUGH the group's egress only n1 answers, n2 is a confirmed failure and
|
||||
// n3 was never measured. Note the sparse indices: generate/group.go skips members
|
||||
// whose copy cannot be built, so the index is not a dense 0..n-1 and must not be
|
||||
// used to look members up positionally.
|
||||
hist.StoreURLTestHistory(copyTag("vpn", 0, "n1"),
|
||||
&adapter.URLTestHistory{Time: now.Add(-2 * time.Second), Delay: 480})
|
||||
dead := map[string]time.Time{copyTag("vpn", 4, "n2"): now.Add(-time.Second)}
|
||||
&adapter.URLTestHistory{LastOK: now.Add(-2 * time.Second), Delay: 480})
|
||||
hist.StoreURLTestHistory(copyTag("vpn", 4, "n2"),
|
||||
&adapter.URLTestHistory{LastFail: now.Add(-time.Second)})
|
||||
|
||||
g := &fakeGroup{
|
||||
tag: "vpn",
|
||||
@@ -160,7 +162,12 @@ func TestGroupHealthBound(t *testing.T) {
|
||||
},
|
||||
now: copyTag("vpn", 0, "n1"),
|
||||
}
|
||||
gh := groupHealthOf(g, C.TypeSelector, newHealthView(hist, dead, now), true)
|
||||
// The used-set names this group: Used=true, and an unrelated name changes nothing.
|
||||
used := map[string]bool{"vpn": true}
|
||||
gh := groupHealthOf(g, C.TypeSelector, newHealthView(hist, healthTTLFloor, now), used, true)
|
||||
if !gh.Used {
|
||||
t.Error("group named in the used-set reported Used=false")
|
||||
}
|
||||
|
||||
if !gh.Bound {
|
||||
t.Error("egress-bound group reported bound=false")
|
||||
@@ -189,7 +196,7 @@ func TestGroupHealthBound(t *testing.T) {
|
||||
func TestGroupHealthSummaryOmitsMembers(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
g := &fakeGroup{tag: "auto", kind: C.TypeURLTest, all: []string{"n1", "n2"}}
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, nil, time.Now()), false)
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, 0, time.Now()), nil, false)
|
||||
if gh.Members != nil {
|
||||
t.Errorf("summary carried %d member rows, want none", len(gh.Members))
|
||||
}
|
||||
@@ -199,6 +206,21 @@ func TestGroupHealthSummaryOmitsMembers(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupHealthUnused: a group the observatory's used-set does NOT name is
|
||||
// reported Used=false — the panel renders it "unused" instead of health counters
|
||||
// (its members are deliberately never probed and stay untested by design).
|
||||
func TestGroupHealthUnused(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
g := &fakeGroup{tag: "idle", kind: C.TypeSelector, all: []string{"n1", "n2"}}
|
||||
used := map[string]bool{"someOtherGroup": true}
|
||||
gh := groupHealthOf(g, C.TypeSelector, newHealthView(hist, 0, time.Now()), used, true)
|
||||
if gh.Used {
|
||||
t.Fatal("group absent from the used-set reported Used=true")
|
||||
}
|
||||
// The counters are still honest: everything untested, nothing invented.
|
||||
assertCounters(t, gh, 2, 0, 0, 2)
|
||||
}
|
||||
|
||||
// TestGroupHealthNilEngine: the read is nil-safe end to end and always yields a
|
||||
// non-nil slice (the panel maps over it unconditionally).
|
||||
func TestGroupHealthNilEngine(t *testing.T) {
|
||||
@@ -234,3 +256,61 @@ func assertCounters(t *testing.T, gh GroupHealth, total, alive, dead, untested i
|
||||
gh.Alive+gh.Dead+gh.Untested, gh.Total)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// TestUsedChainNames pins the chain-name recovery from the observatory's tag-keyed
|
||||
// used-set: only exit-wrapper tags "chain-<name>-h<digits>" yield a name, and every
|
||||
// other tag the set can hold (group tags, per-group egress copies, chain member
|
||||
// copies, plain node tags) is left out. nil ⇒ nil (the caller's "everything used").
|
||||
func TestUsedChainNames(t *testing.T) {
|
||||
if got := usedChainNames(nil); got != nil {
|
||||
t.Fatalf("usedChainNames(nil) = %v, want nil (everything used sentinel)", got)
|
||||
}
|
||||
used := map[string]bool{
|
||||
"chain-relay-h2": true, // a 2-hop chain "relay"'s exit wrapper
|
||||
"chain-relay-h1": true, // its group-hop wrapper (used, not an exit tag — still parses to "relay")
|
||||
"chain-a-h2-h1": true, // chain literally named "a-h2" (parseChainExitTag takes the last -h)
|
||||
"chain-relay-h1-nodeA": false, // a member copy — must NOT parse as an exit, so no false "relay"/"nodeA"
|
||||
"auto": true, // a group tag — not a chain
|
||||
"group-vpn-m0-n1": true, // a per-group egress copy — not a chain
|
||||
"n1": true, // a plain node tag — not a chain
|
||||
"chain-c": true, // no hop suffix — not an exit tag
|
||||
"chain--h1": true, // empty name — not an exit tag
|
||||
}
|
||||
got := usedChainNames(used)
|
||||
// Only the three real exit/group wrapper tags parse to a chain name. A group-hop
|
||||
// wrapper ("chain-relay-h1") is NOT an exit, but parseChainExitTag recovers the
|
||||
// same name from it (it only checks the -h<digits> suffix, not topological
|
||||
// exits-ness), so "relay" is covered either way — which is what the badge needs.
|
||||
want := map[string]bool{"relay": true, "a-h2": true}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("usedChainNames = %v, want %v", got, want)
|
||||
}
|
||||
for name := range want {
|
||||
if !got[name] {
|
||||
t.Errorf("usedChainNames missing chain %q (got %v)", name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainHealthNilUsedSet: a stopped engine (no observatory configured) yields a
|
||||
// nil used-set, and ChainHealth reports every named chain Used=true — no badge is
|
||||
// better than a wrong one, the same convention as GroupHealth.Used. Names are
|
||||
// echoed back in order, blanks dropped.
|
||||
func TestChainHealthNilUsedSet(t *testing.T) {
|
||||
e := New()
|
||||
got := e.ChainHealth([]string{"relay", "", " unused ", "single"})
|
||||
want := []ChainHealth{
|
||||
{Name: "relay", Used: true},
|
||||
{Name: "unused", Used: true},
|
||||
{Name: "single", Used: true},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("ChainHealth = %+v, want %+v (nil used-set ⇒ all used, blanks dropped)", got, want)
|
||||
}
|
||||
for i, w := range want {
|
||||
if got[i] != w {
|
||||
t.Errorf("ChainHealth[%d] = %+v, want %+v", i, got[i], w)
|
||||
}
|
||||
}
|
||||
}
|
||||
+244
-56
@@ -16,27 +16,29 @@ import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// Group test — "what am I actually going out through, and how fast" (F2).
|
||||
// Exit test — "what am I actually going out through, and how fast" (F2, plan §5.E).
|
||||
//
|
||||
// # Why this is not just another node probe
|
||||
//
|
||||
// TestAllNodes (probeall.go) answers "which nodes are alive". It cannot answer the
|
||||
// question an operator actually asks after switching a group: *through which
|
||||
// address am I leaving the country right now*. A group is an indirection — a
|
||||
// selector or urltest over many members — so the delay of the group and the public
|
||||
// address it exits from are properties of the CURRENT selection, not of any node
|
||||
// the panel can point at.
|
||||
// The observatory (observatory.go) answers "which nodes are alive". It cannot
|
||||
// answer the question an operator actually asks after switching a group: *through
|
||||
// which address am I leaving the country right now*. A group is an indirection —
|
||||
// a selector or urltest over many members — so the delay of the group and the
|
||||
// public address it exits from are properties of the CURRENT selection, not of
|
||||
// any node the panel can point at. A CHAIN is the same question over a longer
|
||||
// path: its exit wrapper "chain-<name>-hN" tunnels through every hop, so dialling
|
||||
// it measures the whole L1..Ln path end to end.
|
||||
//
|
||||
// So one test per group measures two things:
|
||||
// So one test per target (group or chain) measures two things:
|
||||
//
|
||||
// delay_ms — reusing urltest.URLTest, the SAME primitive probeall.go uses for a
|
||||
// node. There is deliberately no second latency mechanism here: the
|
||||
// group outbound is just an outbound, and probing it exercises exactly
|
||||
// the path traffic will take (selector -> selected member -> server).
|
||||
// exit_ip — a real HTTP request THROUGH the group to a service that echoes the
|
||||
// client address back. Nothing else can produce this number: the router
|
||||
// cannot know its own public address, and the proxy protocol does not
|
||||
// report it.
|
||||
// delay_ms — reusing urltest.URLTest, the SAME primitive the observatory uses
|
||||
// for a node. There is deliberately no second latency mechanism
|
||||
// here: the target is just an outbound, and probing it exercises
|
||||
// exactly the path traffic will take.
|
||||
// exit_ip — a real HTTP request THROUGH the target to a service that echoes
|
||||
// the client address back. Nothing else can produce this number: the
|
||||
// router cannot know its own public address, and the proxy protocol
|
||||
// does not report it.
|
||||
//
|
||||
// # The direct-egress trap
|
||||
//
|
||||
@@ -66,12 +68,17 @@ const (
|
||||
exitBodyLimit = 4 << 10
|
||||
)
|
||||
|
||||
// GroupTestResult is one group's test outcome. The JSON tags are the panel
|
||||
// contract — see the shater API docs for /api/groups/test.
|
||||
// GroupTestResult is one target's test outcome — a group's or a chain's (Group
|
||||
// then carries the chain's model name). The JSON tags are the panel contract —
|
||||
// see the shater API docs for /api/groups/test.
|
||||
//
|
||||
// Selected is the group's current pick (OutboundGroup.Now()); for a chain it is
|
||||
// the node NAME the chain's last group hop currently selects, "" when the chain
|
||||
// has no group hop (a fixed path selects nothing).
|
||||
//
|
||||
// OK reports whether the LATENCY measurement succeeded, which is the test's primary
|
||||
// question. A failed exit-address lookup deliberately does NOT clear it: knowing the
|
||||
// group is up and fast is useful on its own, and a probe service being unreachable
|
||||
// target is up and fast is useful on its own, and a probe service being unreachable
|
||||
// says nothing about the tunnel. In that case OK stays true and ExitIP is empty —
|
||||
// "not determined", never a guess and never somebody else's address.
|
||||
type GroupTestResult struct {
|
||||
@@ -88,7 +95,8 @@ type GroupTestResult struct {
|
||||
// isGroupOutbound reports whether an outbound is a user-facing GROUP worth testing:
|
||||
// a selector/urltest that is not one of the generator's internal copies.
|
||||
//
|
||||
// The prefix exclusions mirror shouldProbeOutbound (probeall.go): "chain-<name>-h<i>"
|
||||
// The prefix exclusions mirror the reserved tag namespaces (see probeablePlanTag):
|
||||
// "chain-<name>-h<i>"
|
||||
// is a per-chain hop wrapper and "group-<name>-m<i>-<member>" a per-group egress
|
||||
// copy — both are implementation detail of a target the operator DID configure, and
|
||||
// testing them would report tags that appear nowhere in the UI.
|
||||
@@ -101,19 +109,167 @@ func isGroupOutbound(typ, tag string) bool {
|
||||
return !strings.HasPrefix(tag, "chain-") && !strings.HasPrefix(tag, "group-")
|
||||
}
|
||||
|
||||
// TestGroups launches a one-shot background test of the named groups (empty/nil =
|
||||
// every group in the running box), returning started=false when a run is already in
|
||||
// flight.
|
||||
// groupTestTarget is one thing a run measures: a user-facing GROUP (name == the
|
||||
// group outbound's tag) or a CHAIN (name == the chain's model name, dialled via
|
||||
// its exit wrapper). sel reports the target's current selection at dial time,
|
||||
// nil when the target selects nothing.
|
||||
type groupTestTarget struct {
|
||||
name string
|
||||
ob adapter.Outbound
|
||||
sel func() string
|
||||
}
|
||||
|
||||
// chainTargetsFrom discovers each materialised chain in the running
|
||||
// outbound/endpoint set and returns one target per chain, dialled via its EXIT
|
||||
// wrapper.
|
||||
//
|
||||
// It is a SINGLETON on the same pattern as TestAllNodes: a second request while a
|
||||
// run is in flight is refused rather than queued or run in parallel, because these
|
||||
// runs open real tunnelled connections and a panel that double-fires a button must
|
||||
// not multiply the load on the uplink.
|
||||
// Discovery is topological, not name parsing: among the "chain-" tagged
|
||||
// outbounds, a chain's exit is the one no other chain outbound depends on (every
|
||||
// other hop wrapper and member copy is somebody's Detour/member dependency). The
|
||||
// chain NAME is then recovered from the exit tag "chain-<name>-h<digits>"; a
|
||||
// name crafted to collide with another chain's hop namespace is the same
|
||||
// documented edge case parseGroupCopyTag accepts.
|
||||
//
|
||||
// A single-hop chain is NOT discoverable: the generator resolves it straight to
|
||||
// the underlying node/group tag with no wrapper (generate/chain.go buildChain),
|
||||
// so nothing chain-tagged exists in the box for it. Such a chain is reported
|
||||
// missing when named — its alias (the node/group itself) is the thing to test.
|
||||
func chainTargetsFrom(pool []adapter.Outbound) []groupTestTarget {
|
||||
byTag := map[string]adapter.Outbound{}
|
||||
for _, ob := range pool {
|
||||
if strings.HasPrefix(ob.Tag(), "chain-") {
|
||||
byTag[ob.Tag()] = ob
|
||||
}
|
||||
}
|
||||
if len(byTag) == 0 {
|
||||
return nil
|
||||
}
|
||||
referenced := map[string]bool{}
|
||||
for _, ob := range byTag {
|
||||
for _, dep := range ob.Dependencies() {
|
||||
if _, ok := byTag[dep]; ok {
|
||||
referenced[dep] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
var out []groupTestTarget
|
||||
for tag, ob := range byTag {
|
||||
if referenced[tag] {
|
||||
continue
|
||||
}
|
||||
name, ok := parseChainExitTag(tag)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
t := groupTestTarget{name: name, ob: ob}
|
||||
if hop := chainLastGroupHop(byTag, name); hop != nil {
|
||||
t.sel = func() string { return chainMemberName(hop.Now(), name) }
|
||||
}
|
||||
out = append(out, t)
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].name < out[j].name })
|
||||
return out
|
||||
}
|
||||
|
||||
// parseChainExitTag recovers the chain name from an exit wrapper tag
|
||||
// "chain-<name>-h<digits>". Member copies ("…-h<i>-<member>") and non-chain tags
|
||||
// parse false.
|
||||
func parseChainExitTag(tag string) (string, bool) {
|
||||
rest, ok := strings.CutPrefix(tag, "chain-")
|
||||
if !ok {
|
||||
return "", false
|
||||
}
|
||||
i := strings.LastIndex(rest, "-h")
|
||||
if i <= 0 {
|
||||
return "", false
|
||||
}
|
||||
if _, ok := parseAllDigits(rest[i+2:]); !ok {
|
||||
return "", false
|
||||
}
|
||||
return rest[:i], true
|
||||
}
|
||||
|
||||
// chainLastGroupHop finds the chain's LAST group hop — the wrapper selector
|
||||
// "chain-<name>-h<i>" with the largest hop index that is a group outbound. That
|
||||
// hop's Now() is the only selection a chain has (plan §5.E); nil when the chain
|
||||
// is a fixed node path.
|
||||
func chainLastGroupHop(byTag map[string]adapter.Outbound, name string) adapter.OutboundGroup {
|
||||
prefix := "chain-" + name + "-h"
|
||||
best := -1
|
||||
var bestG adapter.OutboundGroup
|
||||
for tag, ob := range byTag {
|
||||
rest, ok := strings.CutPrefix(tag, prefix)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
idx, ok := parseAllDigits(rest)
|
||||
if !ok {
|
||||
continue // a member copy, or another chain sharing the prefix
|
||||
}
|
||||
g, isGroup := ob.(adapter.OutboundGroup)
|
||||
if !isGroup {
|
||||
continue
|
||||
}
|
||||
if idx > best {
|
||||
best, bestG = idx, g
|
||||
}
|
||||
}
|
||||
return bestG
|
||||
}
|
||||
|
||||
// chainMemberName maps a chain member copy tag "chain-<name>-h<i>-<member>" back
|
||||
// to the member's node name (the name is carried in the tag itself, exactly as
|
||||
// with per-group egress copies). Anything else is returned verbatim — a name the
|
||||
// panel can at least display.
|
||||
func chainMemberName(tag, chain string) string {
|
||||
rest, ok := strings.CutPrefix(tag, "chain-"+chain+"-h")
|
||||
if !ok {
|
||||
return tag
|
||||
}
|
||||
j := strings.IndexByte(rest, '-')
|
||||
if j <= 0 {
|
||||
return tag
|
||||
}
|
||||
if _, ok := parseAllDigits(rest[:j]); !ok {
|
||||
return tag
|
||||
}
|
||||
if member := rest[j+1:]; member != "" {
|
||||
return member
|
||||
}
|
||||
return tag
|
||||
}
|
||||
|
||||
// parseAllDigits parses a non-empty all-digit string.
|
||||
func parseAllDigits(s string) (int, bool) {
|
||||
if s == "" {
|
||||
return 0, false
|
||||
}
|
||||
n := 0
|
||||
for i := range len(s) {
|
||||
c := s[i]
|
||||
if c < '0' || c > '9' {
|
||||
return 0, false
|
||||
}
|
||||
n = n*10 + int(c-'0')
|
||||
}
|
||||
return n, true
|
||||
}
|
||||
|
||||
// TestGroups launches a one-shot background test of the named targets — groups
|
||||
// and chains (empty/nil = every group and every chain in the running box),
|
||||
// returning started=false when a run is already in flight.
|
||||
//
|
||||
// It is a SINGLETON: a second request while a run is in flight is refused rather
|
||||
// than queued or run in parallel, because these runs open real tunnelled
|
||||
// connections and a panel that double-fires a button must not multiply the load
|
||||
// on the uplink. The observatory checks the same guard and skips its tick while
|
||||
// a run is in flight (observatory.go), so a manual test never competes with
|
||||
// background probing for the uplink.
|
||||
//
|
||||
// probeURL is the latency-probe URL; "" falls back to urltest's gstatic default.
|
||||
//
|
||||
// Apply-swap safety: the target outbounds are snapshotted up front, so a config swap
|
||||
// mid-run cannot change what is being tested. A group torn down mid-run simply fails
|
||||
// mid-run cannot change what is being tested. A target torn down mid-run simply fails
|
||||
// its probe and is reported not-ok.
|
||||
func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
if !e.groupTestRunning.CompareAndSwap(false, true) {
|
||||
@@ -133,13 +289,14 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
defer e.groupTestRunning.Store(false)
|
||||
|
||||
results := make([]GroupTestResult, len(targets)+len(missing))
|
||||
// A group the caller named that the running box does not have is a RESULT,
|
||||
// not a silent omission: the operator asked about it and deserves to be told
|
||||
// it is not there (typo, not applied yet, or dropped for having no members).
|
||||
// A name the caller asked for that the running box has neither as a group
|
||||
// nor as a chain is a RESULT, not a silent omission: the operator asked
|
||||
// about it and deserves to be told it is not there (typo, not applied yet,
|
||||
// or dropped for having no usable members/hops).
|
||||
for i, name := range missing {
|
||||
results[len(targets)+i] = GroupTestResult{
|
||||
Group: name,
|
||||
Error: "no such group in the running engine",
|
||||
Error: "no such group or chain in the running engine",
|
||||
TestedUnix: time.Now().Unix(),
|
||||
}
|
||||
}
|
||||
@@ -148,27 +305,30 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
|
||||
sem := make(chan struct{}, groupTestConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for i, ob := range targets {
|
||||
for i, tgt := range targets {
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(i int, ob adapter.Outbound) {
|
||||
go func(i int, tgt groupTestTarget) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
res := e.testOneGroup(ob, probeURL)
|
||||
res := e.testOneTarget(tgt, probeURL)
|
||||
e.storeGroupTestResult(i, res)
|
||||
e.groupTestDone.Add(1)
|
||||
}(i, ob)
|
||||
}(i, tgt)
|
||||
}
|
||||
wg.Wait()
|
||||
}()
|
||||
return true
|
||||
}
|
||||
|
||||
// groupTargets resolves the requested group names against the running box,
|
||||
// returning the outbounds to test and the names that do not exist. An empty/nil
|
||||
// names list means "every group"; a stopped engine yields no targets (and every
|
||||
// explicitly named group as missing, so the caller is told why).
|
||||
func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missing []string) {
|
||||
// groupTargets resolves the requested names against the running box — groups
|
||||
// first, then chains — returning the targets to test and the names that exist as
|
||||
// neither. An empty/nil names list means "every group and every chain"; a
|
||||
// stopped engine yields no targets (and every explicitly named target as
|
||||
// missing, so the caller is told why). A group and a chain sharing one name
|
||||
// cannot collide: the generator's chain tags live in a reserved namespace, and
|
||||
// the group match wins deterministically.
|
||||
func (e *Engine) groupTargets(names []string) (targets []groupTestTarget, missing []string) {
|
||||
var want map[string]bool
|
||||
var order []string
|
||||
for _, n := range names {
|
||||
@@ -188,7 +348,13 @@ func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missi
|
||||
found := map[string]bool{}
|
||||
if inst := e.Instance(); inst != nil {
|
||||
if om := inst.Outbound(); om != nil {
|
||||
// The pool for chain discovery must include endpoints: a chain hop
|
||||
// rebuilt from a wireguard/AWG node is an ENDPOINT copy, invisible in
|
||||
// Outbounds() — an exit materialised that way would otherwise vanish
|
||||
// from the run.
|
||||
var pool []adapter.Outbound
|
||||
for _, ob := range om.Outbounds() {
|
||||
pool = append(pool, ob)
|
||||
if !isGroupOutbound(ob.Type(), ob.Tag()) {
|
||||
continue
|
||||
}
|
||||
@@ -196,7 +362,26 @@ func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missi
|
||||
continue
|
||||
}
|
||||
found[ob.Tag()] = true
|
||||
targets = append(targets, ob)
|
||||
tgt := groupTestTarget{name: ob.Tag(), ob: ob}
|
||||
if g, ok := ob.(adapter.OutboundGroup); ok {
|
||||
tgt.sel = g.Now
|
||||
}
|
||||
targets = append(targets, tgt)
|
||||
}
|
||||
if em := inst.Endpoint(); em != nil {
|
||||
for _, ep := range em.Endpoints() {
|
||||
pool = append(pool, ep)
|
||||
}
|
||||
}
|
||||
for _, ct := range chainTargetsFrom(pool) {
|
||||
if want != nil && !want[ct.name] {
|
||||
continue
|
||||
}
|
||||
if found[ct.name] {
|
||||
continue
|
||||
}
|
||||
found[ct.name] = true
|
||||
targets = append(targets, ct)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -208,16 +393,18 @@ func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missi
|
||||
return targets, missing
|
||||
}
|
||||
|
||||
// testOneGroup measures one group: which member it currently selects, the latency
|
||||
// through it, and the public address it exits from.
|
||||
func (e *Engine) testOneGroup(ob adapter.Outbound, probeURL string) GroupTestResult {
|
||||
res := GroupTestResult{Group: ob.Tag(), TestedUnix: time.Now().Unix()}
|
||||
if g, ok := ob.(adapter.OutboundGroup); ok {
|
||||
res.Selected = g.Now()
|
||||
// testOneTarget measures one target: what it currently selects, the latency
|
||||
// through it, and the public address it exits from. For a chain the dialled
|
||||
// outbound is the exit wrapper, so the delay and the exit address are end-to-end
|
||||
// properties of the whole L1..Ln path.
|
||||
func (e *Engine) testOneTarget(t groupTestTarget, probeURL string) GroupTestResult {
|
||||
res := GroupTestResult{Group: t.name, TestedUnix: time.Now().Unix()}
|
||||
if t.sel != nil {
|
||||
res.Selected = t.sel()
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), groupTestDelayTimeout)
|
||||
delay, err := urltest.URLTest(ctx, probeURL, ob)
|
||||
delay, err := urltest.URLTest(ctx, probeURL, t.ob)
|
||||
cancel()
|
||||
if err != nil {
|
||||
res.Error = err.Error()
|
||||
@@ -227,7 +414,7 @@ func (e *Engine) testOneGroup(ob adapter.Outbound, probeURL string) GroupTestRes
|
||||
res.DelayMs = int(delay)
|
||||
|
||||
// The exit address is best-effort by design: see GroupTestResult.OK.
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(ob)
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(t.ob)
|
||||
return res
|
||||
}
|
||||
|
||||
@@ -387,13 +574,14 @@ func (e *Engine) GroupTestStatus() (running bool, done, total int, scope []strin
|
||||
e.groupTestResults()
|
||||
}
|
||||
|
||||
// scopeOf is the set of group names a run covers: everything it will produce a result
|
||||
// for, whether that result is a measurement or a "no such group". Sorted so the panel
|
||||
// sees a stable order and two polls of the same run never differ.
|
||||
func scopeOf(targets []adapter.Outbound, missing []string) []string {
|
||||
// scopeOf is the set of target names a run covers — group and chain alike:
|
||||
// everything it will produce a result for, whether that result is a measurement
|
||||
// or a "no such group or chain". Sorted so the panel sees a stable order and two
|
||||
// polls of the same run never differ.
|
||||
func scopeOf(targets []groupTestTarget, missing []string) []string {
|
||||
out := make([]string, 0, len(targets)+len(missing))
|
||||
for _, ob := range targets {
|
||||
out = append(out, ob.Tag())
|
||||
for _, t := range targets {
|
||||
out = append(out, t.name)
|
||||
}
|
||||
out = append(out, missing...)
|
||||
sort.Strings(out)
|
||||
|
||||
@@ -1,13 +1,17 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
)
|
||||
|
||||
// TestIsGroupOutbound pins which outbounds the group test considers a testable
|
||||
@@ -34,6 +38,172 @@ func TestIsGroupOutbound(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// depOutbound is a failingOutbound with declared dependencies, for the chain
|
||||
// discovery tests (the generator's hop copies depend on the previous hop's tag).
|
||||
type depOutbound struct {
|
||||
failingOutbound
|
||||
deps []string
|
||||
}
|
||||
|
||||
func (d *depOutbound) Dependencies() []string { return d.deps }
|
||||
|
||||
// depGroup is a fakeGroup with declared dependencies — a chain's group-hop
|
||||
// wrapper selector.
|
||||
type depGroup struct {
|
||||
fakeGroup
|
||||
deps []string
|
||||
}
|
||||
|
||||
func (d *depGroup) Dependencies() []string { return d.deps }
|
||||
|
||||
// TestParseChainExitTag pins the exit-tag format "chain-<name>-h<digits>",
|
||||
// including a chain whose NAME itself ends in "-h<digits>" (the last "-h" wins)
|
||||
// and the member-copy shape that must NOT parse as an exit.
|
||||
func TestParseChainExitTag(t *testing.T) {
|
||||
cases := []struct {
|
||||
tag string
|
||||
name string
|
||||
ok bool
|
||||
}{
|
||||
{"chain-c-h2", "c", true},
|
||||
{"chain-my chain-h10", "my chain", true},
|
||||
{"chain-a-h2-h1", "a-h2", true}, // chain literally named "a-h2"
|
||||
{"chain-c-h2-member", "", false}, // a member copy, not an exit
|
||||
{"chain-c", "", false}, // no hop suffix
|
||||
{"auto", "", false}, // not a chain tag at all
|
||||
{"chain--h1", "", false}, // empty name
|
||||
{"group-vpn-m0-chain-x-h1", "", false}, // wrong namespace
|
||||
}
|
||||
for _, c := range cases {
|
||||
name, ok := parseChainExitTag(c.tag)
|
||||
if name != c.name || ok != c.ok {
|
||||
t.Errorf("parseChainExitTag(%q) = (%q,%v), want (%q,%v)", c.tag, name, ok, c.name, c.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainTargetsFrom: discovery is topological — the exit is the chain outbound
|
||||
// nobody else in the chain namespace depends on. A chain with a group hop reports
|
||||
// that hop's current pick mapped back to the NODE NAME; a fixed chain selects
|
||||
// nothing; non-chain outbounds are ignored.
|
||||
func TestChainTargetsFrom(t *testing.T) {
|
||||
pool := []adapter.Outbound{
|
||||
// chain "c": exit h2 (plain copy) -> group hop h1 over two member copies.
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h2"}, []string{"chain-c-h1"}},
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h1", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h1-nodeA", "chain-c-h1-nodeB"},
|
||||
now: "chain-c-h1-nodeA",
|
||||
}, []string{"chain-c-h1-nodeA", "chain-c-h1-nodeB"}},
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h1-nodeA"}, nil},
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h1-nodeB"}, nil},
|
||||
// chain "fix": two plain node hops, no group — selects nothing.
|
||||
&depOutbound{failingOutbound{tag: "chain-fix-h2"}, []string{"chain-fix-h1"}},
|
||||
&depOutbound{failingOutbound{tag: "chain-fix-h1"}, nil},
|
||||
// Not a chain: must be ignored entirely.
|
||||
&depOutbound{failingOutbound{tag: "auto"}, []string{"n1"}},
|
||||
}
|
||||
targets := chainTargetsFrom(pool)
|
||||
if len(targets) != 2 {
|
||||
t.Fatalf("got %d chain targets, want 2: %+v", len(targets), targets)
|
||||
}
|
||||
// Sorted by name: "c" then "fix".
|
||||
c, fix := targets[0], targets[1]
|
||||
if c.name != "c" || c.ob.Tag() != "chain-c-h2" {
|
||||
t.Fatalf("chain c = (%q, %q), want dialled via its exit chain-c-h2", c.name, c.ob.Tag())
|
||||
}
|
||||
if c.sel == nil {
|
||||
t.Fatal("chain c has a group hop; sel must report its pick")
|
||||
}
|
||||
if got := c.sel(); got != "nodeA" {
|
||||
t.Errorf("chain c selected = %q, want the NODE NAME nodeA (not the copy tag)", got)
|
||||
}
|
||||
if fix.name != "fix" || fix.ob.Tag() != "chain-fix-h2" {
|
||||
t.Fatalf("chain fix = (%q, %q), want exit chain-fix-h2", fix.name, fix.ob.Tag())
|
||||
}
|
||||
if fix.sel != nil {
|
||||
t.Error("a fixed chain selects nothing; sel must be nil")
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainMemberName pins the copy-tag → node-name mapping the panel displays.
|
||||
func TestChainMemberName(t *testing.T) {
|
||||
cases := []struct {
|
||||
tag, chain, want string
|
||||
}{
|
||||
{"chain-c-h1-🇩🇪 Frankfurt-01", "c", "🇩🇪 Frankfurt-01"},
|
||||
{"chain-c-h1-007", "c", "007"}, // an all-digit node name survives
|
||||
{"chain-c-h1", "c", "chain-c-h1"},
|
||||
{"unrelated", "c", "unrelated"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := chainMemberName(c.tag, c.chain); got != c.want {
|
||||
t.Errorf("chainMemberName(%q,%q) = %q, want %q", c.tag, c.chain, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// dialableOutbound routes every dial to a fixed local address — a stand-in for a
|
||||
// chain exit whose whole path is up. urltest.URLTest dials the probe URL's host
|
||||
// through the outbound, so pointing every dial at a local HTTP server makes the
|
||||
// latency probe succeed without any network.
|
||||
type dialableOutbound struct {
|
||||
failingOutbound
|
||||
addr string
|
||||
deps []string
|
||||
}
|
||||
|
||||
func (d *dialableOutbound) Dependencies() []string { return d.deps }
|
||||
func (d *dialableOutbound) DialContext(ctx context.Context, network string, _ M.Socksaddr) (net.Conn, error) {
|
||||
return (&net.Dialer{}).DialContext(ctx, network, d.addr)
|
||||
}
|
||||
|
||||
// TestChainExitTestMeasuresEndToEnd is the §6-S4 acceptance path for chains: the
|
||||
// exit test dials the chain's EXIT TAG, returns a measured delay, carries the
|
||||
// chain's model name (not the wrapper tag) as the result's Group, and reports the
|
||||
// last group hop's pick as Selected. The exit address is measured through the
|
||||
// same outbound (unreachable from a test => empty, never a guess).
|
||||
func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
probe := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
defer probe.Close()
|
||||
|
||||
exit := &dialableOutbound{
|
||||
failingOutbound: failingOutbound{tag: "chain-x-h2"},
|
||||
addr: probe.Listener.Addr().String(),
|
||||
deps: []string{"chain-x-h1"},
|
||||
}
|
||||
pool := []adapter.Outbound{
|
||||
exit,
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-x-h1", kind: C.TypeSelector,
|
||||
all: []string{"chain-x-h1-relay"},
|
||||
now: "chain-x-h1-relay",
|
||||
}, []string{"chain-x-h1-relay"}},
|
||||
&depOutbound{failingOutbound{tag: "chain-x-h1-relay"}, nil},
|
||||
}
|
||||
targets := chainTargetsFrom(pool)
|
||||
if len(targets) != 1 || targets[0].ob.Tag() != "chain-x-h2" {
|
||||
t.Fatalf("targets = %+v, want the chain dialled via its exit tag", targets)
|
||||
}
|
||||
|
||||
e := New()
|
||||
res := e.testOneTarget(targets[0], probe.URL)
|
||||
if res.Group != "x" {
|
||||
t.Errorf("Group = %q, want the chain's model name x", res.Group)
|
||||
}
|
||||
if !res.OK || res.Error != "" {
|
||||
t.Fatalf("result = %+v, want a successful measurement through the exit tag (a local roundtrip may legitimately read 0ms)", res)
|
||||
}
|
||||
if res.Selected != "relay" {
|
||||
t.Errorf("Selected = %q, want the last group hop's pick relay", res.Selected)
|
||||
}
|
||||
if res.ExitIP != "" {
|
||||
t.Errorf("ExitIP = %q, want empty — the exit services are unreachable here and must never be guessed", res.ExitIP)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupTestSingleton is the "no parallel runs" invariant: while a run holds the
|
||||
// guard, a second TestGroups must be refused rather than starting a concurrent run
|
||||
// (each of these opens real tunnelled connections).
|
||||
|
||||
+74
-160
@@ -7,155 +7,121 @@ import (
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
)
|
||||
|
||||
// Node health = the engine's MEASUREMENTS + our own FAILURE verdicts, kept apart.
|
||||
// Node health = a read of the shared HEALTH BOARD (common/urltest), classified
|
||||
// against a TTL.
|
||||
//
|
||||
// # Why the two are separate stores
|
||||
// # One store, one truth
|
||||
//
|
||||
// The engine's urltest.HistoryStorage is the engine's own truth, and it holds exactly
|
||||
// one kind of fact: "this outbound answered a probe in N ms". A failed probe is not
|
||||
// represented there at all — protocol/group/urltest.go DELETES the entry (urltest.go
|
||||
// :512) rather than recording the failure. So "probed and dead" and "never probed"
|
||||
// look identical in it, which is precisely the distinction the panel needs.
|
||||
// The board's entry shape is {LastOK, Delay, LastFail}: a success is stored by
|
||||
// StoreURLTestHistory (preserving the last failure), a failure by MarkFailed
|
||||
// (preserving the last success for display), and NOTHING is deleted on failure —
|
||||
// deletion stays reserved for nodes removed from the config. Whoever learns
|
||||
// about a death — a group's own checker, the observatory, a failed user dial —
|
||||
// marks it in the same store the strategies select on, so the panel and the
|
||||
// selection can never disagree about who is dead. The private "dead overlay"
|
||||
// this file used to keep is gone with the problem it worked around.
|
||||
//
|
||||
// The obvious shortcut is to write our own failure INTO that store under a sentinel
|
||||
// delay. We did, and it was wrong. The engine does not treat the store as opaque data
|
||||
// it merely hands back: the round_robin balancer's pool planner reads it as a liveness
|
||||
// oracle, and its test is presence, not value —
|
||||
// # Verdicts are computed on read, against a TTL
|
||||
//
|
||||
// protocol/group/urltest.go, rebuildPool (~:709) and seedPool (~:748):
|
||||
// if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
// results[tag] = candidate{tag: tag, delay: history.Delay, alive: true}
|
||||
// A record is only as good as its age. The TTL is max(3 x global probe interval,
|
||||
// 10 minutes) — three missed refresh periods, floored so a router with a short
|
||||
// interval does not flap everything to "untested" during a brief probe outage:
|
||||
//
|
||||
// so ANY entry, whatever its delay, is a live candidate. Our `failover` strategy is
|
||||
// built on round_robin with a one-slot pool and PoolTolerance 0, whose planner fills a
|
||||
// hole with the first "live" member in order (generate/group.go failoverBalancer). A
|
||||
// sentinel-marked dead node would therefore walk straight into the slot and hold it —
|
||||
// the group would pin itself to a node we had just proved dead, which is the exact
|
||||
// failure failover exists to prevent. Bounded by one probe interval, but that interval
|
||||
// is 30s for failover, and the trigger is an operator pressing "test all".
|
||||
// alive — LastOK newer than LastFail and younger than TTL;
|
||||
// dead — LastFail newer than LastOK and younger than TTL;
|
||||
// untested — nothing fresh enough either way.
|
||||
//
|
||||
// A value chosen so one consumer reads it correctly (0xFFFF is the SLOWEST delay, so
|
||||
// least_test's minimum-delay Select never prefers it) is not safe for a consumer that
|
||||
// only asks whether the key exists. Rather than audit every present and future reader
|
||||
// of the engine's store, we keep our verdict in our own overlay: the engine's history
|
||||
// then contains only real measurements in every mode, and we do not depend on how any
|
||||
// version of the engine interprets what is in it. It also disposes of the least_test
|
||||
// caveat for free — a node with no entry was never a candidate to begin with.
|
||||
//
|
||||
// # Invalidation: timestamps, not subscriptions
|
||||
//
|
||||
// A "dead" verdict must evaporate the moment the node answers again — including when
|
||||
// the answer comes from the GROUP's own probing, which writes to the engine's store
|
||||
// and knows nothing about us. HealthView.State therefore honours our mark only while
|
||||
// it is NEWER than the stored measurement (decodeHealth): any success recorded after
|
||||
// we marked a tag dead wins, whoever recorded it.
|
||||
//
|
||||
// The alternative was urltest.HistoryStorage.AddUpdateHook. It is worse on both counts:
|
||||
// the hook emits a bare struct{} with no tag and no indication of whether it fired for
|
||||
// a store or a delete, so we would have to rescan everything on every DNS-adjacent
|
||||
// write and still could not tell a success from a deletion — and a deletion carries no
|
||||
// timestamp to compare against anyway. Comparing timestamps is strictly more precise,
|
||||
// needs no subscription, no goroutine and no teardown.
|
||||
// decodeHealth below MUST agree with urltest.HistoryStorage.VerdictAt — it is
|
||||
// the same classification plus the delay/age the panel renders. The strategies
|
||||
// consume VerdictAt directly; both read the same record with the same rules.
|
||||
//
|
||||
// # Lifetime
|
||||
//
|
||||
// The overlay lives on the Engine, not on a Box, so it survives an Apply swap exactly
|
||||
// as the pre-registered history store does (see engine.New). Without that, every apply
|
||||
// would silently turn every "dead" back into "untested". It is bounded by pruning to
|
||||
// the set of outbounds the last full run actually probed (pruneProbeFailed).
|
||||
// The board is pre-registered on the engine context (see engine.New), so it
|
||||
// survives an Apply swap: a config change never turns a known-dead node back
|
||||
// into "untested".
|
||||
|
||||
// Health states. A closed set: these three strings are the JSON values the panel
|
||||
// switches on (see GroupMemberHealth.State / NodeHealthStat.State).
|
||||
const (
|
||||
// HealthAlive: a probe SUCCEEDED and no newer failure contradicts it.
|
||||
// HealthAlive: the newest fresh observation is a SUCCESS.
|
||||
HealthAlive = "alive"
|
||||
// HealthDead: a probe RAN and FAILED, and no newer success contradicts it. Always
|
||||
// a positive finding — never inferred from missing data.
|
||||
// HealthDead: the newest fresh observation is a FAILURE — recorded by the
|
||||
// observatory, a group's own checker, or a failed user dial. Always a
|
||||
// positive finding, never inferred from missing data.
|
||||
HealthDead = "dead"
|
||||
// HealthUntested: nothing is known. Either nothing ever probed this outbound, or
|
||||
// a group probed it, failed, and deleted the entry (the engine records failures by
|
||||
// deletion). The two are indistinguishable from the engine's store alone, so this
|
||||
// is reported as "no data" and NEVER as dead — and never as healthy.
|
||||
// HealthUntested: nothing fresh enough is known. Either nothing ever probed
|
||||
// this outbound (it is reachable from no enabled rule — see GroupHealth.Used)
|
||||
// or every observation is older than the TTL. NEVER to be rendered as
|
||||
// healthy, and never as dead either.
|
||||
HealthUntested = "untested"
|
||||
)
|
||||
|
||||
// HealthView is a consistent, cheap-to-query snapshot of node health: the engine's
|
||||
// measurement store plus a copy of our failure verdicts, frozen at one instant.
|
||||
// HealthView is a consistent, cheap-to-query snapshot of node health: the shared
|
||||
// health board frozen with one clock and one TTL.
|
||||
//
|
||||
// Take one view and query it for many tags. That is what makes the per-group
|
||||
// projection affordable (a 376-member group is 376 State calls against one view) and
|
||||
// what keeps every consumer — group health and the per-node stats — on the SAME
|
||||
// decoder, so the "no data means untested, never dead" rule exists in exactly one
|
||||
// place.
|
||||
// projection affordable (a 376-member group is 376 State calls against one view)
|
||||
// and what keeps every consumer — group health and the per-node stats — on the
|
||||
// SAME decoder, so the "no fresh data means untested, never dead" rule exists in
|
||||
// exactly one place.
|
||||
//
|
||||
// The zero value is usable and reports everything as HealthUntested.
|
||||
type HealthView struct {
|
||||
hist *urltest.HistoryStorage
|
||||
dead map[string]time.Time
|
||||
ttl time.Duration
|
||||
now time.Time
|
||||
}
|
||||
|
||||
// HealthView returns a snapshot of current node health. Safe on a stopped engine (it
|
||||
// reads the process-global history store, which outlives any Box) and safe for
|
||||
// HealthView returns a snapshot of current node health. Safe on a stopped engine
|
||||
// (it reads the process-global board, which outlives any Box) and safe for
|
||||
// concurrent use; the returned view is an immutable value the caller owns.
|
||||
func (e *Engine) HealthView() HealthView {
|
||||
v := HealthView{hist: e.URLTestHistory(), now: time.Now()}
|
||||
e.deadMu.Lock()
|
||||
if len(e.dead) > 0 {
|
||||
v.dead = make(map[string]time.Time, len(e.dead))
|
||||
for tag, at := range e.dead {
|
||||
v.dead[tag] = at
|
||||
}
|
||||
}
|
||||
e.deadMu.Unlock()
|
||||
return v
|
||||
return HealthView{hist: e.URLTestHistory(), ttl: e.healthTTL(), now: time.Now()}
|
||||
}
|
||||
|
||||
// State reports one outbound tag's health as of this view: the state, the last
|
||||
// SUCCESSFUL probe's RTT in ms (0 unless alive), and how many seconds ago the reported
|
||||
// observation was made (-1 when there is none).
|
||||
// SUCCESSFUL probe's RTT in ms (0 unless alive), and how many seconds ago the
|
||||
// reported observation was made (-1 when there is none).
|
||||
//
|
||||
// tag is the tag that was actually PROBED — a node's own tag for an ordinary member,
|
||||
// and the per-group egress copy tag for a member of an egress-bound group. Those are
|
||||
// two different network paths and therefore two different, independently valid states
|
||||
// for the same node.
|
||||
// tag is the tag that was actually PROBED — a node's own tag for an ordinary
|
||||
// member, and the per-group egress copy tag for a member of an egress-bound
|
||||
// group. Those are two different network paths and therefore two different,
|
||||
// independently valid states for the same node.
|
||||
func (v HealthView) State(tag string) (state string, delayMs int, ageSeconds int64) {
|
||||
// LoadURLTestHistory is nil-receiver safe, so a view taken with no history store
|
||||
// (bare Engine{}, engine never started) degrades to "untested", not a panic.
|
||||
return decodeHealth(v.hist.LoadURLTestHistory(tag), v.dead[tag], v.now)
|
||||
// LoadURLTestHistory is nil-receiver safe, so a view taken with no history
|
||||
// store (bare Engine{}, engine never started) degrades to "untested".
|
||||
return decodeHealth(v.hist.LoadURLTestHistory(tag), v.ttl, v.now)
|
||||
}
|
||||
|
||||
// decodeHealth is the SINGLE point where a stored measurement and our failure verdict
|
||||
// become one of the three states. Everything that reports node health — per-group
|
||||
// (grouphealth.go) and per-node (shater/stats) — goes through here, so the rules below
|
||||
// cannot drift between the two surfaces.
|
||||
// decodeHealth is the SINGLE point where a board record becomes one of the three
|
||||
// states. Everything that reports node health — per-group (grouphealth.go) and
|
||||
// per-node (shater/stats) — goes through here, so the rules cannot drift between
|
||||
// the two surfaces. It classifies exactly as urltest.HistoryStorage.VerdictAt
|
||||
// does (the strategies' view of the same record), adding only the delay and age
|
||||
// the panel renders.
|
||||
//
|
||||
// deadAt is the zero Time when we hold no failure verdict for the tag.
|
||||
//
|
||||
// success present, not older than our verdict -> alive (a later success always
|
||||
// wins, whoever recorded it:
|
||||
// our own run OR the group's)
|
||||
// verdict present and newer (or no success) -> dead (a probe ran and failed)
|
||||
// neither -> untested
|
||||
//
|
||||
// An entry with a ZERO timestamp cannot be ordered against a dated verdict, so a
|
||||
// verdict wins over it; that combination only arises from a hand-built history entry.
|
||||
func decodeHealth(h *adapter.URLTestHistory, deadAt, now time.Time) (state string, delayMs int, ageSeconds int64) {
|
||||
if h != nil && !deadAt.After(h.Time) {
|
||||
// An entry is only ever written after a probe SUCCEEDED (the engine stores on
|
||||
// success and deletes on failure), so its presence is proof of reachability
|
||||
// regardless of the magnitude of the delay.
|
||||
return HealthAlive, int(h.Delay), ageSince(h.Time, now)
|
||||
// A ttl <= 0 (zero HealthView in tests) falls back to the TTL floor.
|
||||
func decodeHealth(h *adapter.URLTestHistory, ttl time.Duration, now time.Time) (state string, delayMs int, ageSeconds int64) {
|
||||
if h == nil {
|
||||
return HealthUntested, 0, -1
|
||||
}
|
||||
if !deadAt.IsZero() {
|
||||
return HealthDead, 0, ageSince(deadAt, now)
|
||||
if ttl <= 0 {
|
||||
ttl = healthTTLFloor
|
||||
}
|
||||
switch {
|
||||
case h.LastOK.After(h.LastFail) && now.Sub(h.LastOK) < ttl:
|
||||
return HealthAlive, int(h.Delay), ageSince(h.LastOK, now)
|
||||
case h.LastFail.After(h.LastOK) && now.Sub(h.LastFail) < ttl:
|
||||
return HealthDead, 0, ageSince(h.LastFail, now)
|
||||
default:
|
||||
return HealthUntested, 0, -1
|
||||
}
|
||||
return HealthUntested, 0, -1
|
||||
}
|
||||
|
||||
// ageSince is whole seconds from t to now, clamped at 0 (a clock step must not produce
|
||||
// a negative age), or -1 when t is unset — the same "-1 means unknown" convention the
|
||||
// JSON carries.
|
||||
// ageSince is whole seconds from t to now, clamped at 0 (a clock step must not
|
||||
// produce a negative age), or -1 when t is unset — the same "-1 means unknown"
|
||||
// convention the JSON carries.
|
||||
func ageSince(t, now time.Time) int64 {
|
||||
if t.IsZero() {
|
||||
return -1
|
||||
@@ -166,55 +132,3 @@ func ageSince(t, now time.Time) int64 {
|
||||
}
|
||||
return age
|
||||
}
|
||||
|
||||
// MarkProbeFailed records that a probe of tag RAN and FAILED, at this instant.
|
||||
//
|
||||
// It is the only writer of a dead verdict, and it deliberately writes nowhere near the
|
||||
// engine's own history store (see the file header). It is the public counterpart of
|
||||
// URLTestHistory(), through which a caller records a SUCCESS: successes are the
|
||||
// engine's own truth and go in the engine's store, failures are ours and go here.
|
||||
//
|
||||
// There is deliberately no public "unmark": a verdict is retired by a later SUCCESS,
|
||||
// wherever that success is recorded (decodeHealth compares timestamps), so a caller
|
||||
// cannot leave a node stuck dead by forgetting to clear it.
|
||||
func (e *Engine) MarkProbeFailed(tag string) {
|
||||
e.deadMu.Lock()
|
||||
if e.dead == nil {
|
||||
e.dead = make(map[string]time.Time)
|
||||
}
|
||||
e.dead[tag] = time.Now()
|
||||
e.deadMu.Unlock()
|
||||
}
|
||||
|
||||
// clearProbeFailed drops any dead verdict for tag, called when a probe SUCCEEDS.
|
||||
//
|
||||
// Strictly speaking this is redundant — decodeHealth already ignores a verdict older
|
||||
// than the success we just stored — but dropping it keeps the map from accumulating
|
||||
// entries for tags that have long since recovered, and keeps the map's contents
|
||||
// meaning what its name says.
|
||||
func (e *Engine) clearProbeFailed(tag string) {
|
||||
e.deadMu.Lock()
|
||||
delete(e.dead, tag)
|
||||
e.deadMu.Unlock()
|
||||
}
|
||||
|
||||
// pruneProbeFailed drops verdicts for tags outside keep — the set a full probe run
|
||||
// just covered. That is what bounds the overlay: a node removed from the config, or a
|
||||
// per-group copy that vanished when a group lost its egress binding, stops being
|
||||
// probed and would otherwise keep its verdict forever.
|
||||
//
|
||||
// keep must be a COMPLETE probe target set; an empty/nil keep is ignored rather than
|
||||
// treated as "nothing is live", so a run that found no targets at all (stopped engine)
|
||||
// cannot wipe verdicts collected while it was up.
|
||||
func (e *Engine) pruneProbeFailed(keep map[string]bool) {
|
||||
if len(keep) == 0 {
|
||||
return
|
||||
}
|
||||
e.deadMu.Lock()
|
||||
for tag := range e.dead {
|
||||
if !keep[tag] {
|
||||
delete(e.dead, tag)
|
||||
}
|
||||
}
|
||||
e.deadMu.Unlock()
|
||||
}
|
||||
|
||||
@@ -31,64 +31,75 @@ func (f *failingOutbound) ListenPacket(context.Context, M.Socksaddr) (net.Packet
|
||||
|
||||
var _ adapter.Outbound = (*failingOutbound)(nil)
|
||||
|
||||
// TestDecodeHealth pins the single decoder: the three states, and above all the two
|
||||
// rules that make the feature honest — no data is UNTESTED (never dead), and a
|
||||
// success recorded AFTER our failure verdict wins (that is how a dead mark is
|
||||
// invalidated, including by the group's own probing, which knows nothing about us).
|
||||
// TestDecodeHealth pins the single decoder: the three states, and above all the
|
||||
// rules that make the feature honest — no fresh data is UNTESTED (never dead), a
|
||||
// success recorded AFTER a failure wins (that is how a dead mark is invalidated,
|
||||
// including by the group's own probing), and every observation ages out on the
|
||||
// TTL. The classification must agree with urltest.HistoryStorage.VerdictAt — the
|
||||
// strategies' view of the same record.
|
||||
func TestDecodeHealth(t *testing.T) {
|
||||
now := time.Now()
|
||||
at := func(d time.Duration) time.Time { return now.Add(d) }
|
||||
const ttl = 10 * time.Minute
|
||||
cases := []struct {
|
||||
name string
|
||||
h *adapter.URLTestHistory
|
||||
deadAt time.Time
|
||||
wantState string
|
||||
wantDelay int
|
||||
wantAge int64
|
||||
}{
|
||||
{
|
||||
name: "nothing known is untested, never dead",
|
||||
// The whole point: a group records a failed probe by DELETING the entry, so
|
||||
// absence cannot be read as a failure.
|
||||
// No record at all: the node is outside every used path, or was never
|
||||
// probed. Absence cannot be read as a failure.
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
{
|
||||
name: "measurement only is alive",
|
||||
h: &adapter.URLTestHistory{Time: at(-30 * time.Second), Delay: 142},
|
||||
name: "fresh success is alive",
|
||||
h: &adapter.URLTestHistory{LastOK: at(-30 * time.Second), Delay: 142},
|
||||
wantState: HealthAlive, wantDelay: 142, wantAge: 30,
|
||||
},
|
||||
{
|
||||
name: "verdict only is dead",
|
||||
deadAt: at(-5 * time.Second),
|
||||
name: "fresh failure is dead",
|
||||
h: &adapter.URLTestHistory{LastFail: at(-5 * time.Second)},
|
||||
wantState: HealthDead, wantAge: 5,
|
||||
},
|
||||
{
|
||||
name: "verdict NEWER than the measurement wins: dead",
|
||||
h: &adapter.URLTestHistory{Time: at(-time.Minute), Delay: 90},
|
||||
deadAt: at(-10 * time.Second),
|
||||
name: "failure NEWER than the success wins: dead (the last success's " +
|
||||
"delay stays stored for display but the state is the failure's)",
|
||||
h: &adapter.URLTestHistory{LastOK: at(-time.Minute), Delay: 90, LastFail: at(-10 * time.Second)},
|
||||
wantState: HealthDead, wantAge: 10,
|
||||
},
|
||||
{
|
||||
name: "measurement NEWER than the verdict wins: alive (this is how a dead " +
|
||||
name: "success NEWER than the failure wins: alive (this is how a dead " +
|
||||
"mark is invalidated, e.g. by the group's own successful probe)",
|
||||
h: &adapter.URLTestHistory{Time: at(-10 * time.Second), Delay: 77},
|
||||
deadAt: at(-time.Minute),
|
||||
h: &adapter.URLTestHistory{LastOK: at(-10 * time.Second), Delay: 77, LastFail: at(-time.Minute)},
|
||||
wantState: HealthAlive, wantDelay: 77, wantAge: 10,
|
||||
},
|
||||
{
|
||||
name: "a stored zero delay is still a SUCCESS, so alive",
|
||||
h: &adapter.URLTestHistory{Time: at(-time.Second), Delay: 0},
|
||||
h: &adapter.URLTestHistory{LastOK: at(-time.Second), Delay: 0},
|
||||
wantState: HealthAlive, wantDelay: 0, wantAge: 1,
|
||||
},
|
||||
{
|
||||
name: "measurement with no timestamp cannot outrank a dated verdict",
|
||||
name: "a success older than the TTL is untested — a measurement is only " +
|
||||
"as good as its age, and 'alive forever' is the defect the TTL removes",
|
||||
h: &adapter.URLTestHistory{LastOK: at(-ttl - time.Minute), Delay: 50},
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
{
|
||||
name: "a failure older than the TTL is untested, not dead forever",
|
||||
h: &adapter.URLTestHistory{LastFail: at(-ttl - time.Minute)},
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
{
|
||||
name: "a record with no usable timestamp is untested",
|
||||
h: &adapter.URLTestHistory{Delay: 5},
|
||||
deadAt: at(-time.Second),
|
||||
wantState: HealthDead, wantAge: 1,
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
state, delay, age := decodeHealth(c.h, c.deadAt, now)
|
||||
state, delay, age := decodeHealth(c.h, ttl, now)
|
||||
if state != c.wantState || delay != c.wantDelay || age != c.wantAge {
|
||||
t.Errorf("%s:\n got (%q,%d,%d)\n want (%q,%d,%d)",
|
||||
c.name, state, delay, age, c.wantState, c.wantDelay, c.wantAge)
|
||||
@@ -96,22 +107,12 @@ func TestDecodeHealth(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestProbeFailureNeverEntersEngineHistory is the DoD tripwire for the round_robin /
|
||||
// failover pool hazard, and the reason the dead verdict is not a sentinel delay.
|
||||
//
|
||||
// The pool planner in protocol/group/urltest.go decides liveness by the PRESENCE of a
|
||||
// history entry, not by its value:
|
||||
//
|
||||
// rebuildPool (~:709) / seedPool (~:748):
|
||||
// if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
// results[tag] = candidate{tag: tag, delay: history.Delay, alive: true}
|
||||
//
|
||||
// so ANY entry we invent for a dead node makes it a live candidate — and `failover`
|
||||
// (round_robin, one-slot pool, PoolTolerance 0) would fill its single slot with a node
|
||||
// we had just proved dead and hold it for a whole probe interval. The test therefore
|
||||
// asserts the invariant that makes that impossible: after a FAILED probe the engine's
|
||||
// history has no entry for the tag at all, while our own overlay reports it dead.
|
||||
func TestProbeFailureNeverEntersEngineHistory(t *testing.T) {
|
||||
// TestProbeFailureMarksBoardDead is the observatory→selection coupling point: a
|
||||
// failed probe MUST become a same-instant dead verdict on the shared board —
|
||||
// urltest.Verdict is what the balancer slots and Select() read (protocol/group),
|
||||
// so this is exactly what makes the node unselectable in the same tick. The entry
|
||||
// must NOT be deleted (a marked tag stays distinguishable from "never measured").
|
||||
func TestProbeFailureMarksBoardDead(t *testing.T) {
|
||||
e := New()
|
||||
hist := e.URLTestHistory()
|
||||
if hist == nil {
|
||||
@@ -120,40 +121,43 @@ func TestProbeFailureNeverEntersEngineHistory(t *testing.T) {
|
||||
const deadTag = "dead-node"
|
||||
const liveTag = "live-node"
|
||||
// A live member, recorded the way a real successful probe is.
|
||||
hist.StoreURLTestHistory(liveTag, &adapter.URLTestHistory{Time: time.Now(), Delay: 120})
|
||||
hist.StoreURLTestHistory(liveTag, &adapter.URLTestHistory{LastOK: time.Now(), Delay: 120})
|
||||
|
||||
e.probeOne(&failingOutbound{tag: deadTag}, "", hist)
|
||||
|
||||
if h := hist.LoadURLTestHistory(deadTag); h != nil {
|
||||
t.Fatalf("a FAILED probe wrote %+v into the engine's history; it must write "+
|
||||
"nothing there — any entry is read as 'alive' by the round_robin pool planner", h)
|
||||
// The board holds a real record (not a deletion) whose verdict is dead.
|
||||
if h := hist.LoadURLTestHistory(deadTag); h == nil || h.LastFail.IsZero() {
|
||||
t.Fatalf("a FAILED probe must MarkFailed on the board, got %+v", h)
|
||||
}
|
||||
ttl := e.healthTTL()
|
||||
if v := hist.Verdict(deadTag, ttl); v != urltest.VerdictDead {
|
||||
t.Fatalf("board verdict after failed probe = %v, want dead — this is what selection reads", v)
|
||||
}
|
||||
if v := hist.Verdict(liveTag, ttl); v != urltest.VerdictAlive {
|
||||
t.Fatalf("live node verdict = %v, want alive (unaffected)", v)
|
||||
}
|
||||
// And the panel's view agrees — one truth, two readers.
|
||||
if state, _, _ := e.HealthView().State(deadTag); state != HealthDead {
|
||||
t.Fatalf("failed probe: state = %q, want %q from our own overlay", state, HealthDead)
|
||||
}
|
||||
|
||||
// Now replay the planner's own rule over the engine's store: the dead node must not
|
||||
// be a candidate, the live one must be.
|
||||
for tag, wantCandidate := range map[string]bool{deadTag: false, liveTag: true} {
|
||||
got := hist.LoadURLTestHistory(tag) != nil // == the planner's `history != nil` test
|
||||
if got != wantCandidate {
|
||||
t.Errorf("pool candidacy of %q = %v, want %v", tag, got, wantCandidate)
|
||||
}
|
||||
t.Fatalf("failed probe: state = %q, want %q", state, HealthDead)
|
||||
}
|
||||
}
|
||||
|
||||
// A SUCCESSFUL probe stores a real measurement in the engine's history (the groups'
|
||||
// own selection legitimately benefits from it) and drops any stale dead verdict.
|
||||
func TestProbeSuccessClearsDeadVerdict(t *testing.T) {
|
||||
// A SUCCESSFUL probe stores a real measurement, which outranks the previous
|
||||
// failure on the same board record.
|
||||
func TestProbeSuccessOutranksDeadVerdict(t *testing.T) {
|
||||
e := New()
|
||||
hist := e.URLTestHistory()
|
||||
const tag = "flaky-node"
|
||||
e.MarkProbeFailed(tag)
|
||||
hist.MarkFailed(tag)
|
||||
if state, _, _ := e.HealthView().State(tag); state != HealthDead {
|
||||
t.Fatalf("after MarkProbeFailed: state = %q, want %q", state, HealthDead)
|
||||
t.Fatalf("after MarkFailed: state = %q, want %q", state, HealthDead)
|
||||
}
|
||||
// Simulate what probeOne does on success.
|
||||
e.URLTestHistory().StoreURLTestHistory(tag, &adapter.URLTestHistory{Time: time.Now(), Delay: 33})
|
||||
e.clearProbeFailed(tag)
|
||||
// Simulate what probeOneInto does on success — strictly LATER than the
|
||||
// failure: with the wall clock, LastOK could land on the SAME tick as
|
||||
// LastFail (Windows clock granularity) and read as untested (ties are
|
||||
// deliberately not alive — the ordering is strict).
|
||||
failedAt := hist.LoadURLTestHistory(tag).LastFail
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: failedAt.Add(time.Second), Delay: 33})
|
||||
|
||||
state, delay, _ := e.HealthView().State(tag)
|
||||
if state != HealthAlive || delay != 33 {
|
||||
@@ -161,14 +165,13 @@ func TestProbeSuccessClearsDeadVerdict(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The dead overlay must survive an Apply swap exactly as the pre-registered history
|
||||
// store does — otherwise every apply would silently turn every known-dead node back
|
||||
// into "untested". It hangs off the Engine, not off a Box, so tearing the box down
|
||||
// (the destructive half of a swap) must not touch it.
|
||||
// A dead verdict must survive an Apply swap: the board is pre-registered on the
|
||||
// engine context, not owned by a Box, so tearing the box down (the destructive
|
||||
// half of a swap) must not touch it.
|
||||
func TestDeadVerdictSurvivesEngineTeardown(t *testing.T) {
|
||||
e := New()
|
||||
const tag = "node-a"
|
||||
e.MarkProbeFailed(tag)
|
||||
e.URLTestHistory().MarkFailed(tag)
|
||||
if err := e.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
@@ -180,32 +183,26 @@ func TestDeadVerdictSurvivesEngineTeardown(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// pruneProbeFailed bounds the overlay to what a full run actually probed, and must
|
||||
// refuse an empty keep-set so a run against a stopped engine cannot wipe verdicts
|
||||
// collected while it was up.
|
||||
func TestPruneProbeFailed(t *testing.T) {
|
||||
// healthTTL is max(3 x global probe interval, the 10-minute floor) — plan §5.A.
|
||||
// Every board consumer must classify with this same TTL, so the formula is pinned.
|
||||
func TestHealthTTLFormula(t *testing.T) {
|
||||
e := New()
|
||||
for _, tag := range []string{"keep-me", "gone-node", "group-vpn-m0-gone"} {
|
||||
e.MarkProbeFailed(tag)
|
||||
|
||||
// Unconfigured: the engine default interval (3m) gives 9m, floored to 10m.
|
||||
if got := e.healthTTL(); got != healthTTLFloor {
|
||||
t.Fatalf("default TTL = %v, want the %v floor", got, healthTTLFloor)
|
||||
}
|
||||
|
||||
e.pruneProbeFailed(nil)
|
||||
e.pruneProbeFailed(map[string]bool{})
|
||||
for _, tag := range []string{"keep-me", "gone-node"} {
|
||||
if state, _, _ := e.HealthView().State(tag); state != HealthDead {
|
||||
t.Fatalf("an empty keep-set pruned %q; it must be ignored", tag)
|
||||
}
|
||||
// A long interval dominates the floor: 3 x 5m = 15m.
|
||||
e.ConfigureObservatory(ObservatoryConfig{ProbeInterval: 5 * time.Minute})
|
||||
if got, want := e.healthTTL(), 15*time.Minute; got != want {
|
||||
t.Fatalf("TTL for 5m interval = %v, want %v", got, want)
|
||||
}
|
||||
|
||||
e.pruneProbeFailed(map[string]bool{"keep-me": true, "never-probed": true})
|
||||
if state, _, _ := e.HealthView().State("keep-me"); state != HealthDead {
|
||||
t.Errorf("keep-me was pruned but is still a probe target")
|
||||
}
|
||||
for _, tag := range []string{"gone-node", "group-vpn-m0-gone"} {
|
||||
if state, _, _ := e.HealthView().State(tag); state != HealthUntested {
|
||||
t.Errorf("%s: state = %q after prune, want %q (it is no longer a probe target)",
|
||||
tag, state, HealthUntested)
|
||||
}
|
||||
// A short interval is floored: 3 x 30s is far below 10m.
|
||||
e.ConfigureObservatory(ObservatoryConfig{ProbeInterval: 30 * time.Second})
|
||||
if got := e.healthTTL(); got != healthTTLFloor {
|
||||
t.Fatalf("TTL for 30s interval = %v, want the %v floor", got, healthTTLFloor)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,6 +221,6 @@ func TestHealthViewZeroValue(t *testing.T) {
|
||||
|
||||
// newHealthView is the test constructor for a view over a hand-built store, used by
|
||||
// the group-health tests.
|
||||
func newHealthView(hist *urltest.HistoryStorage, dead map[string]time.Time, now time.Time) HealthView {
|
||||
return HealthView{hist: hist, dead: dead, now: now}
|
||||
func newHealthView(hist *urltest.HistoryStorage, ttl time.Duration, now time.Time) HealthView {
|
||||
return HealthView{hist: hist, ttl: ttl, now: now}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,415 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// The observatory — the background prober that keeps the health board fresh for
|
||||
// everything the routing rules can reach (plan §5.C; replaces the full-population
|
||||
// sweep and the manual "Test all nodes" run).
|
||||
//
|
||||
// # What it does, and what it deliberately does not
|
||||
//
|
||||
// One central walker probes the reachability plan (BuildObservatoryPlan) with the
|
||||
// global probe URL and writes every outcome into the shared health board
|
||||
// (common/urltest): a success through StoreURLTestHistory, a failure through
|
||||
// MarkFailed. The strategies read the same board, so a death found here makes the
|
||||
// node unselectable immediately — no private overlay, no panel-only truth.
|
||||
//
|
||||
// It never replaces a group's own probing. An ACTIVE urltest group is its own
|
||||
// fast failure detector (failover switches on its 30s probe tick), and the
|
||||
// freshness gate below skips any tag whose newest observation is younger than the
|
||||
// global probe interval — which is precisely the set of tags an active group is
|
||||
// already keeping current. The observatory's budget lands on what nobody else
|
||||
// probes: idle groups' members, selector candidates, chain exits.
|
||||
//
|
||||
// # Schedule and cost
|
||||
//
|
||||
// The plan is walked by a cursor across small ticks — observatoryTick apart,
|
||||
// observatoryBatch measurements started per tick, observatoryConcurrency in
|
||||
// flight, probeTimeout per probe. Worst case per tick is 24/12 x 5s = 10s, one
|
||||
// period, so a fully dead population saturates the schedule but never stacks (a
|
||||
// busy tick makes the next one a no-op). The population is only what the rules
|
||||
// reach, so a full pass is minutes even on a subscription-sized config, and the
|
||||
// first pass starts IMMEDIATELY after an apply (cold boot: used paths validated
|
||||
// in seconds).
|
||||
//
|
||||
// # Reconfiguration must not restart the walk
|
||||
//
|
||||
// ConfigureObservatory runs on every successful apply, and cron reconciles every
|
||||
// minute — almost always with an identical config. Rebuilding the plan and
|
||||
// zeroing the cursor there would keep the walk from ever completing a pass (the
|
||||
// release-blocking sweep bug). So the cursor is kept whenever the rebuilt plan is
|
||||
// measurement-for-measurement identical to the old one (plansEqual); only a
|
||||
// genuinely different plan restarts it, which is correct — the old cursor indexes
|
||||
// a list that no longer exists.
|
||||
|
||||
const (
|
||||
// observatoryTick is the schedule period. Internal constant, no knob: the
|
||||
// per-tag refresh rate is governed by the global probe interval, not by this.
|
||||
observatoryTick = 10 * time.Second
|
||||
// observatoryBatch is the number of measurements started per tick.
|
||||
observatoryBatch = 24
|
||||
// observatoryConcurrency is the number of measurements in flight at once.
|
||||
observatoryConcurrency = 12
|
||||
// probeTimeout bounds a single probe (connect + HTTP HEAD).
|
||||
probeTimeout = 5 * time.Second
|
||||
// healthTTLFloor is the minimum verdict TTL. The TTL is
|
||||
// max(3 x global probe interval, this floor) — plan §5.A: three missed
|
||||
// refresh periods demote a stale success to "untested" rather than letting
|
||||
// an old measurement read as alive forever.
|
||||
healthTTLFloor = 10 * time.Minute
|
||||
)
|
||||
|
||||
// ObservatoryConfig is what ConfigureObservatory needs from an apply. The zero
|
||||
// value is DISABLED.
|
||||
type ObservatoryConfig struct {
|
||||
Enabled bool
|
||||
// Options is the JUST-APPLIED generated config the reachability plan is
|
||||
// derived from (the generator owns the tag schema; opts is where it is
|
||||
// materialised).
|
||||
Options option.Options
|
||||
// ProbeURL is the global probe URL; "" lets urltest use its gstatic default.
|
||||
ProbeURL string
|
||||
// ProbeInterval is the global probe interval — the freshness gate and the
|
||||
// verdict-TTL base. <=0 falls back to the engine default (3 minutes).
|
||||
ProbeInterval time.Duration
|
||||
}
|
||||
|
||||
// observatoryState is the observatory's mutable state, guarded by obsMu.
|
||||
type observatoryState struct {
|
||||
enabled bool
|
||||
interval time.Duration // effective global probe interval
|
||||
// stop/done/nudge belong to the running loop; nil when no loop is running.
|
||||
// nudge wakes the loop early after a plan change (immediate first pass).
|
||||
stop chan struct{}
|
||||
done chan struct{}
|
||||
nudge chan struct{}
|
||||
// plan is the cycle being walked, cursor the position in it. The plan is
|
||||
// static between applies — it derives from the applied options only.
|
||||
plan []ProbeJob
|
||||
used map[string]bool
|
||||
cursor int
|
||||
cycles uint64
|
||||
// busy guards against a slow tick overlapping the next one.
|
||||
busy bool
|
||||
}
|
||||
|
||||
// ConfigureObservatory installs (or re-tunes, or stops) the observatory. It is
|
||||
// idempotent and called from applyLocked on every SUCCESSFUL apply — the only
|
||||
// moment the facts it depends on (the applied options, the global probe
|
||||
// settings, the GroupHealth master switch) can change. See the file header for
|
||||
// why an identical plan keeps the cursor.
|
||||
func (e *Engine) ConfigureObservatory(cfg ObservatoryConfig) {
|
||||
interval := cfg.ProbeInterval
|
||||
if interval <= 0 {
|
||||
interval = C.DefaultURLTestInterval
|
||||
}
|
||||
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
e.obs.enabled = cfg.Enabled
|
||||
e.obs.interval = interval
|
||||
|
||||
if !cfg.Enabled {
|
||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
||||
if e.obs.stop != nil {
|
||||
close(e.obs.stop)
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
jobs, used := BuildObservatoryPlan(cfg.Options, cfg.ProbeURL)
|
||||
e.obs.used = used
|
||||
fresh := e.obs.plan == nil || !plansEqual(jobs, e.obs.plan)
|
||||
if fresh {
|
||||
e.obs.plan = jobs
|
||||
e.obs.cursor = 0
|
||||
}
|
||||
|
||||
if e.obs.stop == nil {
|
||||
stop, done := make(chan struct{}), make(chan struct{})
|
||||
nudge := make(chan struct{}, 1)
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = stop, done, nudge
|
||||
go e.observatoryLoop(stop, done, nudge)
|
||||
} else if fresh {
|
||||
// The plan really changed: start the new walk now rather than on the
|
||||
// next tick (the immediate-first-pass contract after an apply).
|
||||
select {
|
||||
case e.obs.nudge <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// StopObservatory stops the observatory and waits for an in-flight tick to
|
||||
// finish. Idempotent; safe when nothing runs (teardown path).
|
||||
//
|
||||
// The wait is deliberate and bounded by one batch (worst case
|
||||
// Batch/Concurrency x probeTimeout = 10s): teardown closes the box immediately
|
||||
// afterwards, and probes still in flight would then fail their dials and be
|
||||
// recorded as failure verdicts about nodes that were never actually unreachable.
|
||||
func (e *Engine) StopObservatory() {
|
||||
e.obsMu.Lock()
|
||||
stop, done := e.obs.stop, e.obs.done
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
||||
e.obs.enabled = false
|
||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
||||
if stop != nil {
|
||||
close(stop)
|
||||
}
|
||||
e.obsMu.Unlock()
|
||||
if done != nil {
|
||||
<-done
|
||||
}
|
||||
}
|
||||
|
||||
// ObservatoryStatus reports whether the observatory is on, how many measurements
|
||||
// the current plan holds, and how many full passes have completed. Cheap; used by
|
||||
// tests and available for diagnostics.
|
||||
func (e *Engine) ObservatoryStatus() (enabled bool, planned int, cycles uint64) {
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
return e.obs.enabled, len(e.obs.plan), e.obs.cycles
|
||||
}
|
||||
|
||||
// observatoryUsed returns the published used-set: every tag reachable from the
|
||||
// applied routing rules, as computed for the current plan. nil until an enabled
|
||||
// ConfigureObservatory ran (callers treat nil as "everything used" — no badge is
|
||||
// better than a wrong one). The map is replaced wholesale on reconfiguration and
|
||||
// never mutated in place, so the returned reference is safe to read.
|
||||
func (e *Engine) observatoryUsed() map[string]bool {
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
return e.obs.used
|
||||
}
|
||||
|
||||
// healthTTL is how long an observation stays authoritative:
|
||||
// max(3 x global probe interval, healthTTLFloor). It MUST be computed with the
|
||||
// same formula every consumer of the board uses (plan §5.A), because a verdict
|
||||
// is a (record, TTL) pair — see HealthView and common/urltest Verdict.
|
||||
func (e *Engine) healthTTL() time.Duration {
|
||||
e.obsMu.Lock()
|
||||
interval := e.obs.interval
|
||||
e.obsMu.Unlock()
|
||||
if interval <= 0 {
|
||||
interval = C.DefaultURLTestInterval
|
||||
}
|
||||
ttl := 3 * interval
|
||||
if ttl < healthTTLFloor {
|
||||
ttl = healthTTLFloor
|
||||
}
|
||||
return ttl
|
||||
}
|
||||
|
||||
// observatoryLoop is the ticker. It owns no state of its own — everything lives
|
||||
// under obsMu — so a re-tune from ConfigureObservatory takes effect on the next
|
||||
// tick. The first tick runs immediately (cold start: an applied config's used
|
||||
// paths get their first verdicts in seconds, not after a full tick period).
|
||||
//
|
||||
// Contract with the fork's native group checker (protocol/group CheckOutbounds):
|
||||
// this loop is the BACKGROUND half of the pair — each tick the freshness gate
|
||||
// (observatoryShouldProbe) skips tags an active group is already measuring
|
||||
// itself, so nothing is probed twice and the group's standby cadence (failover's
|
||||
// 30s interval) is never suppressed. Both halves write to the same health board.
|
||||
func (e *Engine) observatoryLoop(stop <-chan struct{}, done chan<- struct{}, nudge <-chan struct{}) {
|
||||
defer close(done)
|
||||
e.observatoryTickOnce()
|
||||
ticker := time.NewTicker(observatoryTick)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-stop:
|
||||
return
|
||||
case <-ticker.C:
|
||||
e.observatoryTickOnce()
|
||||
case <-nudge:
|
||||
e.observatoryTickOnce()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// observatoryTickOnce runs one batch. It is deliberately conservative about when
|
||||
// it does nothing:
|
||||
//
|
||||
// - a manual exit test (grouptest.go) is in flight => SKIP. That run is the
|
||||
// human's explicit request; the observatory defers rather than competing for
|
||||
// the uplink. It checks the guard but never CLAIMS it, so a pressed Test
|
||||
// button can never be answered "already running" because of background work;
|
||||
// - the previous tick has not finished => SKIP, so slow probes never stack;
|
||||
// - the engine is stopped => jobs no longer resolve and are skipped (probeJob).
|
||||
func (e *Engine) observatoryTickOnce() {
|
||||
if e.groupTestRunning.Load() {
|
||||
return
|
||||
}
|
||||
|
||||
e.obsMu.Lock()
|
||||
if e.obs.busy || !e.obs.enabled || len(e.obs.plan) == 0 {
|
||||
e.obsMu.Unlock()
|
||||
return
|
||||
}
|
||||
if e.obs.cursor >= len(e.obs.plan) {
|
||||
// Cycle complete: count it and start the next pass from the top. This is
|
||||
// the ONE place the cursor is deliberately rewound.
|
||||
e.obs.cycles++
|
||||
e.obs.cursor = 0
|
||||
}
|
||||
start := e.obs.cursor
|
||||
end := start + observatoryBatch
|
||||
if end > len(e.obs.plan) {
|
||||
end = len(e.obs.plan)
|
||||
}
|
||||
batch := append([]ProbeJob(nil), e.obs.plan[start:end]...)
|
||||
e.obs.cursor = end
|
||||
interval := e.obs.interval
|
||||
e.obs.busy = true
|
||||
e.obsMu.Unlock()
|
||||
|
||||
defer func() {
|
||||
e.obsMu.Lock()
|
||||
e.obs.busy = false
|
||||
e.obsMu.Unlock()
|
||||
}()
|
||||
|
||||
hist := e.URLTestHistory()
|
||||
if hist == nil {
|
||||
return
|
||||
}
|
||||
now := time.Now()
|
||||
sem := make(chan struct{}, observatoryConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for _, j := range batch {
|
||||
if !observatoryShouldProbe(hist, j, interval, now) {
|
||||
continue
|
||||
}
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(j ProbeJob) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
e.probeJob(j)
|
||||
}(j)
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// observatoryShouldProbe is the freshness gate: a job runs when ANY tag it
|
||||
// covers has no observation at all, or its newest observation — success OR
|
||||
// failure — is older than the global probe interval.
|
||||
//
|
||||
// This is what keeps the observatory off the tags an ACTIVE group measures
|
||||
// itself: the group probes at the same global interval (failover at 30s), so its
|
||||
// tags are always fresher than the gate and are skipped — no duplicate probes,
|
||||
// and the group's own detection cadence is never suppressed.
|
||||
//
|
||||
// "any", not "all", on purpose — the job writes to every tag it covers, so a
|
||||
// single stale tag is reason enough.
|
||||
func observatoryShouldProbe(hist *urltest.HistoryStorage, j ProbeJob, interval time.Duration, now time.Time) bool {
|
||||
for _, tag := range j.Store {
|
||||
h := hist.LoadURLTestHistory(tag)
|
||||
if h == nil {
|
||||
return true
|
||||
}
|
||||
newest := h.LastOK
|
||||
if h.LastFail.After(newest) {
|
||||
newest = h.LastFail
|
||||
}
|
||||
if now.Sub(newest) >= interval {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// probeJob resolves a job's dial tag against the CURRENT box and probes it,
|
||||
// recording the outcome under every tag the job covers.
|
||||
//
|
||||
// Resolving at dial time (rather than holding the adapter.Outbound captured when
|
||||
// the plan was built) is what makes a plan safe across an Apply swap: an
|
||||
// outbound the new config dropped simply no longer resolves, and the job is
|
||||
// SKIPPED. Recording it dead would be a fabricated verdict about a node that was
|
||||
// never dialled — and one that outlives the config change.
|
||||
//
|
||||
// Outbound(tag) falls through to the endpoint manager, so wireguard/AmneziaWG
|
||||
// nodes and their copies resolve here exactly like plain outbounds.
|
||||
func (e *Engine) probeJob(j ProbeJob) {
|
||||
inst := e.Instance()
|
||||
if inst == nil {
|
||||
return
|
||||
}
|
||||
om := inst.Outbound()
|
||||
if om == nil {
|
||||
return
|
||||
}
|
||||
ob, ok := om.Outbound(j.Dial)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
hist := e.URLTestHistory()
|
||||
if hist == nil {
|
||||
return
|
||||
}
|
||||
e.probeOneInto(ob, j.URL, j.Store, hist)
|
||||
}
|
||||
|
||||
// probeOneInto probes one outbound and records the result in the shared health
|
||||
// board, for every tag the measurement covers:
|
||||
//
|
||||
// success — StoreURLTestHistory: a real measurement, exactly as the groups'
|
||||
// own probing records one; selection legitimately benefits from it.
|
||||
// failure — MarkFailed: LastFail is stamped while any previous success is
|
||||
// preserved for display. The strategies read the same board, so the
|
||||
// node becomes unselectable immediately (plan §5.A/§5.B).
|
||||
//
|
||||
// store may name MORE tags than the one dialled. That is not extrapolation: the
|
||||
// planner only groups tags whose dial path is identical, plus the base-tag alias
|
||||
// of an egress copy (see probeplan.go), so the single measurement is literally
|
||||
// the answer for each of them.
|
||||
//
|
||||
// Every alive<->dead verdict FLIP is logged (info) — the single diagnostic trail
|
||||
// for localising a dead chain hop and for spotting a flapping node (plan §5.C).
|
||||
//
|
||||
// The delay is clamped to >=1ms because 0 is the engine's "unset" value in its
|
||||
// own selection arithmetic (protocol/group/urltest.go Select treats 0 as
|
||||
// no-value), so a genuinely sub-millisecond node must not report it.
|
||||
func (e *Engine) probeOneInto(ob adapter.Outbound, probeURL string, store []string, hist *urltest.HistoryStorage) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), probeTimeout)
|
||||
defer cancel()
|
||||
|
||||
delay, err := urltest.URLTest(ctx, probeURL, ob)
|
||||
ttl := e.healthTTL()
|
||||
if err != nil {
|
||||
for _, tag := range store {
|
||||
if e.log != nil && hist.Verdict(tag, ttl) == urltest.VerdictAlive {
|
||||
e.log.Info("observatory: ", tag, ": alive -> dead (probe: ", err, ")")
|
||||
}
|
||||
hist.MarkFailed(tag)
|
||||
}
|
||||
return
|
||||
}
|
||||
if delay == 0 {
|
||||
delay = 1
|
||||
}
|
||||
now := time.Now()
|
||||
for _, tag := range store {
|
||||
if e.log != nil && hist.Verdict(tag, ttl) == urltest.VerdictDead {
|
||||
e.log.Info("observatory: ", tag, ": dead -> alive (probe)")
|
||||
}
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: now, Delay: delay})
|
||||
}
|
||||
}
|
||||
|
||||
// probeOne probes a single outbound and records it under its own tag only. It is
|
||||
// probeOneInto's one-tag form, kept for callers (and tests) that have an
|
||||
// outbound rather than a plan.
|
||||
func (e *Engine) probeOne(ob adapter.Outbound, probeURL string, hist *urltest.HistoryStorage) {
|
||||
e.probeOneInto(ob, probeURL, []string{ob.Tag()}, hist)
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user