Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
84d2766592 | ||
|
|
33f940c2fb | ||
|
|
a57717dabb | ||
|
|
129e31fbdc | ||
|
|
4f0618515e | ||
|
|
fd698162c9 | ||
|
|
ffa78d67bc | ||
|
|
61e51495c3 | ||
|
|
bfe71cd1dd | ||
|
|
9b3644becb | ||
|
|
0a8bdbaf46 | ||
|
|
ebe2e7807b | ||
|
|
c1e1b17a61 | ||
|
|
782770f306 | ||
|
|
8980a25a59 |
@@ -70,6 +70,26 @@
|
||||
# - apt .deb archives for the apk lane's debian:bookworm host-deps
|
||||
# (.cache/apt) — key = hash of ci/sdk-build-apk.sh (the apt list is in it).
|
||||
# - usign binary (.cache/tools) — static helper, fixed key.
|
||||
# - SDK feeds/ git checkouts (.cache/feeds) — the single biggest recurring
|
||||
# cost: `scripts/feeds update -a` cloned base+packages+luci+routing+
|
||||
# telephony EVERY run (~7 min/job; github.com is ~1 MB/s from this
|
||||
# runner — run 51 evidence). The feeds dir is symlinked into the SDK
|
||||
# container from the workspace cache; `feeds update` on an existing clone
|
||||
# is a fast fetch+checkout of the pinned revs. Correctness-safe: update
|
||||
# always checks out feeds.conf's pins, and ci/sdk-build*.sh wipes the
|
||||
# cache + re-clones fresh if update ever fails on a cached checkout.
|
||||
# Key = lane + SDK release (shared across the two arch jobs of a lane —
|
||||
# same release pins identical feed revs; the sequential runner means the
|
||||
# second arch restores what the first saved). restore-keys lets an SDK
|
||||
# version bump start from the old clones (git fetch delta, not re-clone).
|
||||
# Act_runner facts this design leans on (verified in run 51 logs):
|
||||
# - the cache backend works: restores/saves confirmed, hashFiles() works;
|
||||
# - docker images (openwrt/sdk, debian:bookworm, runner-images) live on the
|
||||
# PERSISTENT host daemon — "Image is up to date" each run, no re-download;
|
||||
# - each actions/cache SAVE is followed by an exact 3-minute act_runner
|
||||
# stall (node process lingers; hit→no-save→no stall). Steady state saves
|
||||
# nothing, so adding cache entries is fine, but keys that change every
|
||||
# run (e.g. github.sha) would cost +3 min/entry/run — do NOT do that.
|
||||
|
||||
name: release
|
||||
|
||||
@@ -150,6 +170,21 @@ jobs:
|
||||
restore-keys: |
|
||||
dl-
|
||||
|
||||
# feeds git checkouts (see header): both 24.10.4 arch jobs share one entry
|
||||
# (same release = same feeds.conf.default pins), so derive the release
|
||||
# from the matrix sdk tag (x86_64-24.10.4 -> 24.10.4).
|
||||
- name: Compute feeds cache key
|
||||
id: feedskey
|
||||
run: echo "ver=$(echo '${{ matrix.sdk }}' | sed 's/.*-//')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache SDK feeds checkouts
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: .cache/feeds
|
||||
key: feeds-opkg-${{ steps.feedskey.outputs.ver }}
|
||||
restore-keys: |
|
||||
feeds-opkg-
|
||||
|
||||
- name: Cache CI tools (usign)
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
@@ -281,7 +316,9 @@ jobs:
|
||||
# mirror) — so even with a dead cache server this never wedges the run.
|
||||
- name: Compute SDK cache key
|
||||
id: sdkkey
|
||||
run: echo "tarball=$(basename '${{ matrix.sdk_url }}')" >> "$GITHUB_OUTPUT"
|
||||
run: |
|
||||
echo "tarball=$(basename '${{ matrix.sdk_url }}')" >> "$GITHUB_OUTPUT"
|
||||
echo "relver=$(echo '${{ matrix.sdk_url }}' | sed -n 's#.*/releases/\([^/]*\)/.*#\1#p')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Cache ImmortalWrt SDK tarball
|
||||
uses: actions/cache@v3.3.2
|
||||
@@ -289,6 +326,16 @@ jobs:
|
||||
path: .cache/sdk
|
||||
key: sdk-${{ steps.sdkkey.outputs.tarball }}
|
||||
|
||||
# feeds git checkouts (see header) — one entry shared by both apk arch
|
||||
# jobs of one ImmortalWrt release (identical feeds.conf.default pins).
|
||||
- name: Cache SDK feeds checkouts
|
||||
uses: actions/cache@v3.3.2
|
||||
with:
|
||||
path: .cache/feeds
|
||||
key: feeds-apk-${{ steps.sdkkey.outputs.relver }}
|
||||
restore-keys: |
|
||||
feeds-apk-
|
||||
|
||||
- name: Install UPX
|
||||
run: sudo apt-get update -qq && sudo apt-get install -y -qq upx-ucl
|
||||
|
||||
@@ -440,6 +487,10 @@ jobs:
|
||||
release-apk:
|
||||
name: release apk
|
||||
needs: build-apk
|
||||
# Publish whatever arch feeds succeeded — do NOT block the aarch64 release
|
||||
# when an unrelated arch (e.g. x86_64) fails. download-artifact only fetches
|
||||
# artifacts that exist, and the publish loop skips missing apkfeed-* dirs.
|
||||
if: ${{ !cancelled() }}
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
- name: Checkout
|
||||
|
||||
@@ -38,6 +38,7 @@ nul
|
||||
|
||||
# -- upstream sing-box-lx -----------------------------------------
|
||||
/.idea/
|
||||
.idea/
|
||||
/vendor/
|
||||
/*.json
|
||||
/*.srs
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
## Правила делегирования
|
||||
|
||||
1. ЛЮБАЯ реализация (код, тесты, конфиги, рефакторинг, отладка) выполняется
|
||||
субагентами через инструмент Agent с `model: "opus"`. Сам ты правишь файлы
|
||||
субагентами через инструмент Agent с `model: "fable"`. Сам ты правишь файлы
|
||||
только в одном случае: тривиальная правка в 1–2 строки, где постановка
|
||||
задачи дороже самой правки.
|
||||
|
||||
@@ -27,7 +27,7 @@
|
||||
названия скиллов прямо в текст задания.
|
||||
|
||||
4. Независимые задачи запускай ПАРАЛЛЕЛЬНО — несколько вызовов Agent в одном
|
||||
сообщении, каждый с `model: "opus"`. Зависимые — последовательно, передавая
|
||||
сообщении, каждый с `model: "fable"`. Зависимые — последовательно, передавая
|
||||
в следующее ТЗ результаты предыдущего.
|
||||
|
||||
5. Приёмка: результат каждого субагента ты проверяешь сам (читаешь diff
|
||||
|
||||
@@ -157,7 +157,9 @@ if need == 0: стоп // все слоты живы, не
|
||||
|
||||
## Ошибка дайла
|
||||
|
||||
Слоты/пул **не трогаем**. По одной ошибке дайла причина неизвестна (мёртвая нода / упавший сайт назначения / своя сеть пропала — неразличимы). Состав пула меняет **только дотест** (честный health-check через `url`). Соединение просто фейлится для этого запроса.
|
||||
Слоты/пул **не трогаем** — инвариант сохранён: по одной ошибке дайла причина неизвестна (мёртвая нода / упавший сайт назначения / своя сеть пропала — неразличимы), состав пула меняет **только дотест** (честный health-check через `url`).
|
||||
|
||||
> **Обновление (план «живой пул», §5.B / health board):** поведение «соединение просто фейлится» заменено. Дайл-провал теперь пишет `MarkFailed` в health board и дайл ретраится следующим кандидатом (≤3 членов, в пределах дедлайна коннекта) — коннект юзера выживает, пока в группе есть живая нода. Слоты при этом по-прежнему физически не двигаются: мёртвый жилец остаётся в своём слоте, но вердикт board (`dead`, TTL) делает его невыбираемым на следующем pick — селекция уходит с мёртвого слота, не дожидаясь дотеста. Sticky, replace-in-slot и never-shrink не затронуты.
|
||||
|
||||
---
|
||||
|
||||
@@ -248,7 +250,7 @@ PoolSlot {
|
||||
1. `balancer`-объект; `mode` снаружи как индикатор. Липкость — плоское `balancer.sticky_hash []string` (не вложенный объект — механизм один, выбирать нечего). Дефолт (омит или `[]`)→`["process","domain"]` (липкость из коробки); выключение через sentinel **`["none"]`**. *(rc.15: изначально планировали `[]`→выкл через nil-vs-`[]`, но `badjson.UnmarshallExcludedContext` ре-маршалит структуру и схлопывает `[]`→nil — различить на проде нельзя, подтверждено живым прогоном. `["none"]` переживает round-trip.)*
|
||||
2. `pool_tolerance` — **наше** поле; апстрим `tolerance` затирается на 50, для round_robin не используется (варн).
|
||||
3. `pool < 0` → ошибка; `pool==0`/опущено → дефолт 3; `round_robin` без `balancer` → дефолты.
|
||||
4. Дайл-ошибка слоты не трогает (причина неизвестна).
|
||||
4. Дайл-ошибка слоты не трогает (причина неизвестна). *(План «живой пул» §5.B: дайл-провал теперь `MarkFailed` в board + ретрай следующим кандидатом; слоты по-прежнему не двигаются — см. «Ошибка дайла».)*
|
||||
5. Пул не пустеет — замена только при наличии живого кандидата.
|
||||
6. **Слоты фиксированы; ноды текут сквозь; delay-ранг — отдельное вычисление для выявления худшей, не перестановка слотов.**
|
||||
7. **sticky slot-hash: `slot[hash(key) % pool]` (`pool` фиксирован) → строгий ноль реконнектов для живого узла, 0 памяти.** Единственный механизм. Заменяет ОБА механизма rc.11. Модуль достаточен, т.к. число слотов не меняется — преимущество jump-hash/rendezvous (плавный ресайз) здесь не нужно.
|
||||
@@ -273,7 +275,7 @@ PoolSlot {
|
||||
- **Слоты фиксированы:** вылет ноды из середины не двигает другие слоты; победитель занимает слот вытесненного.
|
||||
- **slot-hash:** один ключ → стабильный слот при неизменном `pool`; **живой узел в своём слоте держит ВСЕ свои ключи** при замене жильцов в других слотах (ноль реконнектов); замена в слоте `k` трогает только ключи `→ k`; ключ `""` → фиксированный слот.
|
||||
- **Замена 1:1:** нет живых → мёртвая держит слот; `min(pool, nodes)`.
|
||||
- **Дайл-ошибка** не меняет состав пула.
|
||||
- **Дайл-ошибка** не меняет состав пула (но с планом «живой пул» §5.B демотит слот через вердикт board и ретраится — см. «Ошибка дайла»).
|
||||
- **Валидация:** `pool<1`, `balancer`+wrong-mode, неизвестный `sticky_hash`-компонент → ошибки; `tolerance`+round_robin → варн.
|
||||
- **Дефолт sticky_hash:** `nil` (поле опущено) → `["process","domain"]` (липкость есть); `[]` → липкости нет (round_robin по counter); различение nil-vs-`[]` после unmarshal.
|
||||
- **`least_test` дефолт** — без изменений.
|
||||
|
||||
+10
-2
@@ -19,11 +19,19 @@ type ClashServer interface {
|
||||
AddModeUpdateHook(hook *observable.Subscriber[struct{}])
|
||||
}
|
||||
|
||||
// lx:begin health-board
|
||||
// Health board (plan §5.A): a history entry now records both the last success and
|
||||
// the last failure instead of being deleted on failure. `Time` is renamed to
|
||||
// `LastOK`; its JSON tag stays "time" so the Clash API history payload is
|
||||
// unchanged, and `LastFail` is omitted when zero for the same reason.
|
||||
type URLTestHistory struct {
|
||||
Time time.Time `json:"time"`
|
||||
Delay uint16 `json:"delay"`
|
||||
LastOK time.Time `json:"time"`
|
||||
Delay uint16 `json:"delay"`
|
||||
LastFail time.Time `json:"last_fail,omitzero"`
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
|
||||
type V2RayServer interface {
|
||||
LifecycleService
|
||||
StatsService() ConnectionTracker
|
||||
|
||||
+10
-1
@@ -68,6 +68,15 @@ CACHE="$REPO/.cache"
|
||||
mkdir -p "$CACHE/sdk" "$CACHE/dl" "$CACHE/apt"
|
||||
chmod -R a+rwX "$CACHE/dl" "$CACHE/apt" 2>/dev/null || true
|
||||
|
||||
# feeds/ git checkouts (actions/cache key: feeds-apk-<release>) — symlinked
|
||||
# over the SDK's feeds dir inside the container (ci/sdk-build-apk.sh) so
|
||||
# `scripts/feeds update -a` fetches deltas instead of re-cloning the
|
||||
# ImmortalWrt feeds every run. Top-level chmod only: contents are created and
|
||||
# owned by the container's uid-1000 build user (restore preserves ownership).
|
||||
FEEDS_CACHE="$CACHE/feeds/apk"
|
||||
mkdir -p "$FEEDS_CACHE"
|
||||
chmod a+rwX "$CACHE" "$CACHE/feeds" "$FEEDS_CACHE" 2>/dev/null || true
|
||||
|
||||
# Fetch the SDK tarball ON THE RUNNER (restored cache -> own Gitea release-asset
|
||||
# mirror -> upstream with stall-kill + retries) instead of the old bare
|
||||
# `wget` inside the container, which hung whole runs when
|
||||
@@ -84,7 +93,7 @@ docker pull -q debian:bookworm
|
||||
docker run --rm --volumes-from "$(hostname)" \
|
||||
-e ARCH="$ARCH" -e REPO="$REPO" -e OUT="$OUT" -e SDK_URL="$SDK_URL" \
|
||||
-e SDK_TAR="$SDK_TAR" -e DL_DIR="$CACHE/dl" -e APT_CACHE="$CACHE/apt" \
|
||||
-e KEY_APK="${KEY_APK:-}" \
|
||||
-e FEEDS_CACHE="$FEEDS_CACHE" -e KEY_APK="${KEY_APK:-}" \
|
||||
debian:bookworm bash "$REPO/ci/sdk-build-apk.sh"
|
||||
|
||||
# --- 2) sanity: the per-arch apk repo dir must be complete -------------------
|
||||
|
||||
@@ -57,6 +57,17 @@ DL_DIR="$REPO/.cache/dl"
|
||||
mkdir -p "$DL_DIR"
|
||||
chmod -R a+rwX "$DL_DIR" 2>/dev/null || true
|
||||
|
||||
# --- 0.6) persistent feeds/ git checkouts -------------------------------------
|
||||
# Workspace dir restored/saved by actions/cache (key: feeds-opkg-<release>) and
|
||||
# symlinked over the SDK's feeds/ inside the container (ci/sdk-build.sh), so
|
||||
# `scripts/feeds update -a` fetches deltas instead of re-cloning base+packages+
|
||||
# luci from scratch (~7 min/run on this runner's slow github.com link).
|
||||
# Top-level chmod only: the contents are created by the container's uid-1000
|
||||
# build user and restored with the same ownership (tar-as-root preserves it).
|
||||
FEEDS_CACHE="$REPO/.cache/feeds/opkg"
|
||||
mkdir -p "$FEEDS_CACHE"
|
||||
chmod a+rwX "$REPO/.cache" "$REPO/.cache/feeds" "$FEEDS_CACHE" 2>/dev/null || true
|
||||
|
||||
# --- 1) SDK package build (4 packages) in the arch-matched SDK image ----------
|
||||
# We drive the `openwrt/sdk` docker image directly (not openwrt/gh-action-sdk):
|
||||
# on a self-hosted Gitea act_runner the marketplace action fetch can be
|
||||
@@ -69,6 +80,7 @@ echo "[feed] SDK build arch=$ARCH image=openwrt/sdk:$SDK_TAG"
|
||||
docker pull "openwrt/sdk:$SDK_TAG"
|
||||
docker run --rm --volumes-from "$(hostname)" \
|
||||
-e ARCH="$ARCH" -e REPO="$REPO" -e OUT="$OUT" -e DL_DIR="$DL_DIR" \
|
||||
-e FEEDS_CACHE="$FEEDS_CACHE" \
|
||||
"openwrt/sdk:$SDK_TAG" \
|
||||
sh "$REPO/ci/sdk-build.sh"
|
||||
|
||||
|
||||
+18
-2
@@ -112,8 +112,24 @@ cd "$SDKDIR"
|
||||
cp -f feeds.conf.default feeds.conf
|
||||
grep -q '^src-link shater ' feeds.conf || echo "src-link shater $REPO/openwrt" >> feeds.conf
|
||||
|
||||
# Persistent feeds checkouts: $FEEDS_CACHE (workspace dir, actions/cache-
|
||||
# persisted, visible via --volumes-from) replaces the fresh SDK's empty feeds/
|
||||
# dir, so `feeds update` git-fetches deltas instead of re-cloning the
|
||||
# ImmortalWrt feeds every run. Update always checks out feeds.conf's pinned
|
||||
# revisions; on any failure with cached checkouts the cache is wiped and the
|
||||
# update retried with fresh clones — a stale cache can never wedge the build.
|
||||
if [ -n "${FEEDS_CACHE:-}" ] && mkdir -p "$FEEDS_CACHE" 2>/dev/null; then
|
||||
rm -rf feeds
|
||||
ln -s "$FEEDS_CACHE" feeds
|
||||
echo "[apk-sdk] feeds/ -> $FEEDS_CACHE (persistent cache)"
|
||||
fi
|
||||
echo "[apk-sdk] feeds update -a"
|
||||
./scripts/feeds update -a
|
||||
if ! ./scripts/feeds update -a; then
|
||||
[ -L feeds ] || { echo "[apk-sdk] ERROR: feeds update failed"; exit 8; }
|
||||
echo "[apk-sdk] WARNING: feeds update failed on cached checkouts — wiping cache, cloning fresh"
|
||||
find "$FEEDS_CACHE" -mindepth 1 -maxdepth 1 -exec rm -rf {} + 2>/dev/null || true
|
||||
./scripts/feeds update -a
|
||||
fi
|
||||
echo "[apk-sdk] feeds install (prefer shater feed)"
|
||||
./scripts/feeds install -p shater shaterd shater-core byedpi luci-app-shater
|
||||
|
||||
@@ -188,7 +204,7 @@ INNER
|
||||
chmod 0644 /home/build/inner.sh
|
||||
|
||||
su build -s /bin/bash -c \
|
||||
"ARCH='$ARCH' REPO='$REPO' OUT='$OUT' SDKDIR='$SDKDIR' KEYFILE='${KEYFILE:-}' DL_DIR='${DL_DIR:-}' bash /home/build/inner.sh"
|
||||
"ARCH='$ARCH' REPO='$REPO' OUT='$OUT' SDKDIR='$SDKDIR' KEYFILE='${KEYFILE:-}' DL_DIR='${DL_DIR:-}' FEEDS_CACHE='${FEEDS_CACHE:-}' bash /home/build/inner.sh"
|
||||
|
||||
chmod -R a+rwX "$OUT" 2>/dev/null || true
|
||||
echo "[apk-sdk] OK arch=$ARCH — apk feed dir:"
|
||||
|
||||
+19
-1
@@ -50,8 +50,26 @@ grep -q '^src-link shater ' feeds.conf || echo "src-link shater $REPO/openwrt" >
|
||||
# luci, packages, routing, telephony). We need `luci` for feeds/luci/luci.mk and
|
||||
# `base`/`packages` for the runtime deps (kmod-nft-tproxy, kmod-nft-socket,
|
||||
# ip-full, rpcd, luci-base) to resolve.
|
||||
#
|
||||
# Persistent feeds checkouts: $FEEDS_CACHE (a workspace dir the runner restores
|
||||
# via actions/cache, shared into this container via --volumes-from) replaces
|
||||
# the SDK's ephemeral feeds/ dir, so `feeds update` git-fetches deltas instead
|
||||
# of re-cloning base+packages+luci every run (~7 min on the runner's slow
|
||||
# github.com link). Correctness-safe: update always checks out feeds.conf's
|
||||
# pinned revisions; if it ever fails on a cached checkout (e.g. a force-pushed
|
||||
# upstream), the cache is wiped and the update retried with fresh clones.
|
||||
if [ -n "${FEEDS_CACHE:-}" ] && mkdir -p "$FEEDS_CACHE" 2>/dev/null; then
|
||||
rm -rf feeds
|
||||
ln -s "$FEEDS_CACHE" feeds
|
||||
echo "[sdk] feeds/ -> $FEEDS_CACHE (persistent cache)"
|
||||
fi
|
||||
echo "[sdk] feeds update -a"
|
||||
./scripts/feeds update -a
|
||||
if ! ./scripts/feeds update -a; then
|
||||
[ -L feeds ] || { echo "[sdk] ERROR: feeds update failed"; exit 8; }
|
||||
echo "[sdk] WARNING: feeds update failed on cached checkouts — wiping cache, cloning fresh"
|
||||
find "$FEEDS_CACHE" -mindepth 1 -maxdepth 1 -exec rm -rf {} + 2>/dev/null || true
|
||||
./scripts/feeds update -a
|
||||
fi
|
||||
|
||||
echo "[sdk] feeds install (prefer shater feed)"
|
||||
./scripts/feeds install -p shater shaterd shater-core byedpi luci-app-shater
|
||||
|
||||
@@ -0,0 +1,84 @@
|
||||
// lx:begin health-board
|
||||
|
||||
// Health board (plan §5.A): failure tracking and verdict computation on top of
|
||||
// HistoryStorage. Successes keep flowing through StoreURLTestHistory; failures are
|
||||
// recorded with MarkFailed instead of deleting the entry (deletion stays reserved
|
||||
// for nodes removed from the configuration), and consumers classify a tag at read
|
||||
// time with Verdict. One store, one truth: whoever learns about a death — the
|
||||
// group's own checker, the observatory, or a failed user dial — marks it here.
|
||||
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
)
|
||||
|
||||
// HealthVerdict classifies a stored history entry at read time.
|
||||
type HealthVerdict int
|
||||
|
||||
const (
|
||||
// VerdictUntested means nothing fresh enough is known either way: no entry,
|
||||
// or every observation is older than the caller's TTL.
|
||||
VerdictUntested HealthVerdict = iota
|
||||
// VerdictAlive means the newest fresh observation is a success.
|
||||
VerdictAlive
|
||||
// VerdictDead means the newest fresh observation is a failure.
|
||||
VerdictDead
|
||||
)
|
||||
|
||||
func (v HealthVerdict) String() string {
|
||||
switch v {
|
||||
case VerdictAlive:
|
||||
return "alive"
|
||||
case VerdictDead:
|
||||
return "dead"
|
||||
default:
|
||||
return "untested"
|
||||
}
|
||||
}
|
||||
|
||||
// MarkFailed records a failed probe or dial for tag: LastFail is set to now while
|
||||
// LastOK/Delay of an existing entry are preserved, so a node that once worked keeps
|
||||
// its last known latency for display. The entry is never deleted here — a marked
|
||||
// tag stays distinguishable from "never measured" (plan §2 Д5).
|
||||
func (s *HistoryStorage) MarkFailed(tag string) {
|
||||
if s == nil {
|
||||
return
|
||||
}
|
||||
s.access.Lock()
|
||||
updated := &adapter.URLTestHistory{LastFail: time.Now()}
|
||||
if previous := s.delayHistory[tag]; previous != nil {
|
||||
updated.LastOK = previous.LastOK
|
||||
updated.Delay = previous.Delay
|
||||
}
|
||||
s.delayHistory[tag] = updated
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
}
|
||||
|
||||
// Verdict classifies tag against the wall clock: alive when the last success is
|
||||
// newer than the last failure and younger than ttl, dead when the last failure is
|
||||
// newer than the last success and younger than ttl, untested otherwise.
|
||||
func (s *HistoryStorage) Verdict(tag string, ttl time.Duration) HealthVerdict {
|
||||
return s.VerdictAt(tag, ttl, time.Now())
|
||||
}
|
||||
|
||||
// VerdictAt is Verdict against an explicit clock, for deterministic tests.
|
||||
func (s *HistoryStorage) VerdictAt(tag string, ttl time.Duration, now time.Time) HealthVerdict {
|
||||
history := s.LoadURLTestHistory(tag)
|
||||
if history == nil {
|
||||
return VerdictUntested
|
||||
}
|
||||
switch {
|
||||
case history.LastOK.After(history.LastFail) && now.Sub(history.LastOK) < ttl:
|
||||
return VerdictAlive
|
||||
case history.LastFail.After(history.LastOK) && now.Sub(history.LastFail) < ttl:
|
||||
return VerdictDead
|
||||
default:
|
||||
return VerdictUntested
|
||||
}
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -59,6 +59,17 @@ func (s *HistoryStorage) DeleteURLTestHistory(tag string) {
|
||||
|
||||
func (s *HistoryStorage) StoreURLTestHistory(tag string, history *adapter.URLTestHistory) {
|
||||
s.access.Lock()
|
||||
// lx:begin health-board
|
||||
// Health board (plan §5.A): overwriting an entry with a fresh success must not
|
||||
// erase the recorded failure — Verdict compares LastOK against LastFail, so
|
||||
// dropping LastFail here would forge an eternal "alive". Callers only ever set
|
||||
// LastOK/Delay on success; a caller that deliberately sets LastFail wins.
|
||||
if history.LastFail.IsZero() {
|
||||
if previous := s.delayHistory[tag]; previous != nil {
|
||||
history.LastFail = previous.LastFail
|
||||
}
|
||||
}
|
||||
// lx:end health-board
|
||||
s.delayHistory[tag] = history
|
||||
s.notifyUpdated()
|
||||
s.access.Unlock()
|
||||
|
||||
@@ -0,0 +1,169 @@
|
||||
// lx:begin health-board
|
||||
|
||||
package urltest
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
)
|
||||
|
||||
// TestVerdictTable drives VerdictAt through every classification the health board
|
||||
// must produce (plan §5.A): the newest FRESH observation decides, and anything
|
||||
// older than the TTL decays to untested.
|
||||
func TestVerdictTable(t *testing.T) {
|
||||
const ttl = 10 * time.Minute
|
||||
now := time.Now()
|
||||
at := func(ago time.Duration) time.Time { return now.Add(-ago) }
|
||||
|
||||
for _, tc := range []struct {
|
||||
name string
|
||||
history *adapter.URLTestHistory // nil = no entry stored
|
||||
want HealthVerdict
|
||||
}{
|
||||
{
|
||||
name: "fresh success is alive",
|
||||
history: &adapter.URLTestHistory{LastOK: at(time.Minute), Delay: 42},
|
||||
want: VerdictAlive,
|
||||
},
|
||||
{
|
||||
name: "fresh failure is dead",
|
||||
history: &adapter.URLTestHistory{LastFail: at(time.Minute)},
|
||||
want: VerdictDead,
|
||||
},
|
||||
{
|
||||
name: "stale success (older than TTL) decays to untested",
|
||||
history: &adapter.URLTestHistory{LastOK: at(ttl + time.Minute), Delay: 42},
|
||||
want: VerdictUntested,
|
||||
},
|
||||
{
|
||||
name: "failure after success is dead",
|
||||
history: &adapter.URLTestHistory{
|
||||
LastOK: at(2 * time.Minute),
|
||||
Delay: 42,
|
||||
LastFail: at(time.Minute),
|
||||
},
|
||||
want: VerdictDead,
|
||||
},
|
||||
{
|
||||
name: "success after failure is alive",
|
||||
history: &adapter.URLTestHistory{
|
||||
LastOK: at(time.Minute),
|
||||
Delay: 42,
|
||||
LastFail: at(2 * time.Minute),
|
||||
},
|
||||
want: VerdictAlive,
|
||||
},
|
||||
{
|
||||
name: "no entry is untested",
|
||||
history: nil,
|
||||
want: VerdictUntested,
|
||||
},
|
||||
{
|
||||
name: "empty entry (both timestamps zero) is untested",
|
||||
history: &adapter.URLTestHistory{},
|
||||
want: VerdictUntested,
|
||||
},
|
||||
{
|
||||
name: "stale failure (older than TTL) decays to untested",
|
||||
history: &adapter.URLTestHistory{
|
||||
LastOK: at(2 * ttl),
|
||||
Delay: 42,
|
||||
LastFail: at(ttl + time.Minute),
|
||||
},
|
||||
want: VerdictUntested,
|
||||
},
|
||||
} {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
if tc.history != nil {
|
||||
storage.delayHistory[tag] = tc.history
|
||||
}
|
||||
if got := storage.VerdictAt(tag, ttl, now); got != tc.want {
|
||||
t.Fatalf("VerdictAt(%+v) = %v, want %v", tc.history, got, tc.want)
|
||||
}
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// TestMarkFailedPreservesSuccess verifies MarkFailed records the failure without
|
||||
// deleting the entry or losing the last known success/latency (plan §2 Д5).
|
||||
func TestMarkFailedPreservesSuccess(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
lastOK := time.Now().Add(-time.Minute)
|
||||
storage.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: lastOK, Delay: 42})
|
||||
|
||||
storage.MarkFailed(tag)
|
||||
|
||||
history := storage.LoadURLTestHistory(tag)
|
||||
if history == nil {
|
||||
t.Fatal("MarkFailed deleted the entry; it must only set LastFail")
|
||||
}
|
||||
if !history.LastOK.Equal(lastOK) || history.Delay != 42 {
|
||||
t.Fatalf("MarkFailed lost the last success: got %+v", history)
|
||||
}
|
||||
if history.LastFail.IsZero() || !history.LastFail.After(lastOK) {
|
||||
t.Fatalf("MarkFailed did not record a fresh failure: got %+v", history)
|
||||
}
|
||||
if got := storage.Verdict(tag, 10*time.Minute); got != VerdictDead {
|
||||
t.Fatalf("Verdict after MarkFailed = %v, want %v", got, VerdictDead)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMarkFailedWithoutEntry verifies MarkFailed on a never-measured tag creates a
|
||||
// failure-only entry (dead, not untested) instead of doing nothing.
|
||||
func TestMarkFailedWithoutEntry(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
|
||||
storage.MarkFailed(tag)
|
||||
|
||||
history := storage.LoadURLTestHistory(tag)
|
||||
if history == nil {
|
||||
t.Fatal("MarkFailed on an unknown tag must create an entry")
|
||||
}
|
||||
if !history.LastOK.IsZero() || history.Delay != 0 {
|
||||
t.Fatalf("MarkFailed invented a success: got %+v", history)
|
||||
}
|
||||
if got := storage.Verdict(tag, 10*time.Minute); got != VerdictDead {
|
||||
t.Fatalf("Verdict = %v, want %v", got, VerdictDead)
|
||||
}
|
||||
}
|
||||
|
||||
// TestStorePreservesLastFail verifies a success overwrite keeps the previously
|
||||
// recorded failure timestamp, so Verdict can still order the two observations.
|
||||
func TestStorePreservesLastFail(t *testing.T) {
|
||||
storage := NewHistoryStorage()
|
||||
const tag = "node"
|
||||
|
||||
storage.MarkFailed(tag)
|
||||
failedAt := storage.LoadURLTestHistory(tag).LastFail
|
||||
|
||||
// A strictly later success: with the wall clock, LastOK could land on the SAME
|
||||
// tick as LastFail (Windows clock granularity) and read as untested (ties are
|
||||
// deliberately not alive — the ordering is strict).
|
||||
storage.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: failedAt.Add(time.Second), Delay: 7})
|
||||
|
||||
history := storage.LoadURLTestHistory(tag)
|
||||
if !history.LastFail.Equal(failedAt) {
|
||||
t.Fatalf("StoreURLTestHistory dropped LastFail: got %+v, want LastFail=%v", history, failedAt)
|
||||
}
|
||||
if got := storage.Verdict(tag, 10*time.Minute); got != VerdictAlive {
|
||||
t.Fatalf("Verdict after success-over-failure = %v, want %v", got, VerdictAlive)
|
||||
}
|
||||
}
|
||||
|
||||
// TestNilStorageVerdict guards the nil-receiver contract shared with
|
||||
// LoadURLTestHistory: reads on a nil store degrade to untested, never panic.
|
||||
func TestNilStorageVerdict(t *testing.T) {
|
||||
var storage *HistoryStorage
|
||||
if got := storage.Verdict("node", time.Minute); got != VerdictUntested {
|
||||
t.Fatalf("nil storage Verdict = %v, want %v", got, VerdictUntested)
|
||||
}
|
||||
storage.MarkFailed("node") // must be a no-op, not a panic
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -489,7 +489,7 @@ func (s *StartedService) readGroups() *Groups {
|
||||
item.Tag = itemTag
|
||||
item.Type = itemOutbound.Type()
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(itemOutbound)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix() // lx: health board §5.A — Time renamed to LastOK
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
g.Items = append(g.Items, &item)
|
||||
@@ -620,8 +620,8 @@ func (s *StartedService) URLTest(ctx context.Context, request *URLTestRequest) (
|
||||
historyStorage.DeleteURLTestHistory(outboundTag)
|
||||
} else {
|
||||
historyStorage.StoreURLTestHistory(outboundTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: t,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: t,
|
||||
})
|
||||
}
|
||||
return nil, nil
|
||||
@@ -1063,7 +1063,7 @@ func (s *StartedService) SubscribeOutbounds(_ *emptypb.Empty, server grpc.Server
|
||||
Type: ob.Type(),
|
||||
}
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(ob)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix() // lx: health board §5.A — Time renamed to LastOK
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
list.Outbounds = append(list.Outbounds, item)
|
||||
@@ -1074,7 +1074,7 @@ func (s *StartedService) SubscribeOutbounds(_ *emptypb.Empty, server grpc.Server
|
||||
Type: ep.Type(),
|
||||
}
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(ep)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix() // lx: health board §5.A — Time renamed to LastOK
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
list.Outbounds = append(list.Outbounds, item)
|
||||
|
||||
@@ -86,8 +86,8 @@ func (s *StartedService) URLTestOutbound(ctx context.Context, request *URLTestOu
|
||||
return &URLTestOutboundResponse{Error: err.Error()}, nil
|
||||
}
|
||||
boxService.urlTestHistoryStorage.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: delay,
|
||||
LastOK: time.Now(),
|
||||
Delay: delay,
|
||||
})
|
||||
return &URLTestOutboundResponse{Delay: uint32(delay)}, nil
|
||||
}
|
||||
@@ -168,7 +168,7 @@ func (s *StartedService) GetOutbounds(ctx context.Context, empty *emptypb.Empty)
|
||||
appendItem := func(detour adapter.Outbound) {
|
||||
item := &GroupItem{Tag: detour.Tag(), Type: detour.Type()}
|
||||
if history := historyStorage.LoadURLTestHistory(adapter.OutboundTag(detour)); history != nil {
|
||||
item.UrlTestTime = history.Time.Unix()
|
||||
item.UrlTestTime = history.LastOK.Unix()
|
||||
item.UrlTestDelay = int32(history.Delay)
|
||||
}
|
||||
list.Outbounds = append(list.Outbounds, item)
|
||||
|
||||
@@ -10,6 +10,81 @@ tracks only the fork. Versions are tagged `vX.Y.Z-lx.N`; releases are built by
|
||||
`lx-release.yml`. Tags carrying an `-rc.N` / `-alpha.N` / `-beta.N` suffix publish
|
||||
as GitHub **pre-releases** and never become "Latest".
|
||||
|
||||
#### Unreleased (shater)
|
||||
|
||||
**Fork-layer + control-plane rework of proxy health** — ships with `shaterd`
|
||||
(the shater router daemon), not as an lx release tag; recorded here because the
|
||||
load-bearing half lives in fork zones (`common/urltest`, `protocol/group`).
|
||||
Routing rules now always get a **live** node under every group strategy, and
|
||||
the background prober measures **only what the rules can reach**.
|
||||
|
||||
* **Health board with real verdicts (`common/urltest`).** A history entry used
|
||||
to record only success (`{Time, Delay}`) and a failed check *deleted* it, so a
|
||||
dead member was indistinguishable from a never-measured one. The entry is now
|
||||
`{LastOK, Delay, LastFail}`; a failure marks, never deletes (deletion is
|
||||
reserved for removing a node from the config). The verdict is computed on
|
||||
read — `alive` (success fresher than failure and younger than TTL), `dead`
|
||||
(failure fresher, while it is fresh), `untested` (nothing fresh) — with
|
||||
TTL = max(3 × global probe interval, 10 min), so a stale success decays to
|
||||
`untested` instead of reading as alive forever. Everyone who learns of a death
|
||||
(the group checker, the observatory, a failed user dial) writes here;
|
||||
selection, balancer slots and the panel read here. The private `engine.dead`
|
||||
overlay that only the panel could see is gone.
|
||||
|
||||
* **Selection is alive-only, and a failed dial retries (`protocol/group`).**
|
||||
least ping ranks by delay **among alive members only**, falls back to untested
|
||||
members in config order, and only then to first-by-config; round_robin /
|
||||
random / failover slot liveness is fed by board verdicts (TTL included), not
|
||||
by "a history entry exists". A failed user dial marks the member dead on the
|
||||
board and transparently re-picks — at most 3 candidates per connection, within
|
||||
the dial deadline (UDP retries only until the first send) — so the connection
|
||||
survives as long as any member is alive. A failed group check marks-fail
|
||||
instead of deleting; an immediate first check on PostStart shrinks the
|
||||
cold-start blind window to seconds. `selector` (manual pin) semantics and the
|
||||
SPEC 019 slot invariants are untouched. When a whole group dies, its rule
|
||||
blocks fail-closed — no silent fallback.
|
||||
|
||||
* **Observatory replaces the background sweep (`shater/engine`).** The probe
|
||||
plan is derived from the applied config by reachability — enabled-rule targets
|
||||
(plus `Final` and DNS detours) → groups → members / egress copies; a chain is
|
||||
probed end-to-end by dialing its exit tag through the whole hop path (the Xray
|
||||
observatory-through-`proxySettings` equivalent; chains were previously not
|
||||
probed at all), and chain group-hop members are probed through their path
|
||||
prefix. Schedule: 10 s tick, batch 24, concurrency 12, 5 s probe timeout; a
|
||||
tag refreshes on the global probe interval; a freshness gate skips tags the
|
||||
active group's own checker already measures; the cursor survives no-op
|
||||
reconciles, and the first pass after apply is immediate. Every alive↔dead
|
||||
flip is logged (info, tag + probe/dial reason) — the diagnostic trail for a
|
||||
dead chain hop and for flapping. The `GroupHealth` toggle now gates the
|
||||
observatory.
|
||||
|
||||
* **Probe URL / Interval are global-only.** The per-group `ProbeURL` /
|
||||
`ProbeInterval` overrides are deleted (model, UCI, generate, panel); probing
|
||||
is configured solely by the global Settings fields, which also drive the
|
||||
native group checker and the exit test (failover keeps its 30 s default
|
||||
interval). With one URL all delay measurements are comparable, and the "one
|
||||
node in two groups with different URLs" ambiguity disappears. Old UCI configs
|
||||
still carrying the options parse silently and drain on the next write.
|
||||
|
||||
* **Subscription nodes move out of UCI.** Each subscription's nodes are cached
|
||||
in `/etc/shater/subs/<name>.json` (atomic tmp+rename; tmpfs fallback
|
||||
`/tmp/shater-subs` when the overlay cannot take the write); UCI keeps only
|
||||
hand-added nodes and the `subscription` sections, so `sub update` rewrites its
|
||||
own cache file instead of the whole `/etc/config/shater`. Legacy `from_sub`
|
||||
nodes migrate into the cache on first read.
|
||||
|
||||
* **One manual test — the exit test, now for chains too.** The manual
|
||||
"Test all nodes" probe-all run is removed together with the sweep
|
||||
(`/api/nodes/test` returns 404). The remaining manual test is the per-target
|
||||
"Test" (delay + exit IP + country through the real path), extended from groups
|
||||
to chains via the chain exit tag.
|
||||
|
||||
* **"Unused" badge.** A group or chain reachable from no enabled routing rule is
|
||||
outside the observatory plan and reported `used:false`; the panel shows
|
||||
"unused" instead of health counters. Unreferenced nodes are deliberately never
|
||||
probed — they stay honestly `untested` until a rule references them, at which
|
||||
point the observatory covers them within seconds.
|
||||
|
||||
#### v1.14.0-lx.3
|
||||
|
||||
**Stable release** (published as "Latest", not a pre-release) — a promotion of
|
||||
|
||||
@@ -329,3 +329,98 @@ makes it affordable on the target hardware: measured on the testbed, StevenBlack
|
||||
2.4 MB of text became **80873 domains in a 491 KB `.srs`**, costing ~0.5 MB of a rootfs
|
||||
with ~33 MB free — so it ships in the production posture rather than being traded away
|
||||
for the 8 KB geosite ads list.
|
||||
|
||||
## D18 — Node health = a board with computed verdicts, NOT delete-on-failure + a dead-overlay
|
||||
Decided 2026-07-24 (proxy-health plan §5.A). Upstream `common/urltest` recorded
|
||||
only success (`{Time, Delay}`) and a failed check **deleted** the entry — a dead
|
||||
member was indistinguishable from a never-measured one. On top of that, deaths
|
||||
found by shater's own probing went to a private `engine.dead` overlay that only
|
||||
the panel read: selection never saw them, and an old success in history counted
|
||||
as "alive forever". Net effect, reproduced in the field: YouTube dead on a real
|
||||
device while the panel showed the group healthy.
|
||||
|
||||
**Decision: one health board, one source of truth.** The history entry becomes
|
||||
`{LastOK, Delay, LastFail}`; a failure *marks* (`MarkFailed`), never deletes —
|
||||
deletion is reserved for removing a node from the config. The verdict is a
|
||||
**method computed on read**, not a stored field: `alive` when the success is
|
||||
fresher than the failure and younger than TTL; `dead` while a fresher failure is
|
||||
itself fresh; `untested` otherwise, with TTL = max(3 × global probe interval,
|
||||
10 min). Everyone who learns of a death — the group's native checker, the
|
||||
observatory, a failed user dial — writes the same board; selection, balancer
|
||||
slots and the panel read the same board. The `engine.dead` overlay is deleted.
|
||||
- **Rejected: keep delete-on-failure and widen the overlay to selection.** Two
|
||||
stores of the same truth with undefined precedence; the overlay would need its
|
||||
own TTL/pruning and every reader would have to merge — exactly the
|
||||
divergence ("green panel, dead path") this decision exists to kill.
|
||||
- **Rejected: persist the board across reboots.** Measurements go stale faster
|
||||
than an overlay write is worth; after reboot everything is `untested` for
|
||||
seconds until the observatory's immediate first pass. In-memory, deliberately.
|
||||
- **Deferred, not rejected: Xray-style sliding-window stats / hysteresis.** The
|
||||
delay of the last success is enough to rank alive members in v1; the board
|
||||
keys and record shape leave room to widen without migration. Flapping is
|
||||
visible instead through the alive↔dead flip log (info) — the single
|
||||
diagnostic trail.
|
||||
Consequence: a stale success **decays** (alive → untested after TTL) instead of
|
||||
reading as alive forever, and a dead group blocks fail-closed — visible and
|
||||
alertable — rather than silently falling back.
|
||||
|
||||
## D19 — Background probing = an observatory driven by rule reachability, NOT a population sweep
|
||||
Decided 2026-07-24 (proxy-health plan §5.C, replaces `shater/engine/sweep.go`).
|
||||
The sweep probed the **whole population** round-robin — all nodes plus all
|
||||
per-group copies, used or not — yet chain copies were excluded entirely, so the
|
||||
one path users actually complained about ("rule → chain") was never measured.
|
||||
The manual "Test all nodes" run duplicated the same full-population walk on
|
||||
demand. On router budgets that is the wrong shape twice: work grows with the
|
||||
subscription size (~200 nodes), not with what the config uses.
|
||||
|
||||
**Decision: probe only what the rules can reach.** Following the Xray model
|
||||
(central observatory + balancers reading observations, §3 of the plan), a plan
|
||||
is built from the **applied** `option.Options` by reachability: enabled-rule
|
||||
targets (plus `Final` and DNS detours) → groups → members / egress copies; a
|
||||
chain is probed **end-to-end** by dialing its exit tag through the whole hop
|
||||
path (the Xray observatory-through-`proxySettings` equivalent); chain group-hop
|
||||
members are probed through their path prefix. Everything unreferenced stays
|
||||
`untested` and its group/chain is badged "unused" in the panel — so untested
|
||||
never looks like a health problem. A freshness gate skips tags an active
|
||||
group's own checker already measures; the cursor survives no-op reconciles
|
||||
(the sweep's release-blocker: cron reconciles must not restart the cycle);
|
||||
the manual probe-all is removed with the sweep.
|
||||
- **Rejected: keep the sweep.** Probes hundreds of unused nodes on a router
|
||||
budget and still misses chains; its "coverage" is what made per-node health
|
||||
look authoritative while the used path went unmeasured.
|
||||
- **Rejected: probe every chain hop individually.** Multiplied probe traffic
|
||||
for diagnostics that never affects selection — no choice depends on a middle
|
||||
hop's individual health. A dead middle hop makes the exit verdict honestly
|
||||
`dead`; localization is served by the verdict flip log.
|
||||
- **Rejected: auto-reroute rules when a group dies.** A dead group blocks
|
||||
fail-closed — visible and alertable. Silent rerouting would hide the outage
|
||||
and change routing semantics behind the user's back.
|
||||
Consequence: the probe budget is bounded by the config, not the subscription;
|
||||
cold start converges in seconds (immediate first pass after apply); wanting
|
||||
numbers for an unused group has one honest answer — reference it from a rule.
|
||||
|
||||
## D20 — Probe URL / Interval are global-only; per-group overrides deleted
|
||||
Decided 2026-07-24 (proxy-health plan §5.D). Groups carried optional
|
||||
`ProbeURL`/`ProbeInterval` overrides. That bred a documented ambiguity — one
|
||||
node shared by two groups with different URLs yields incomparable delays and
|
||||
needs "whose URL wins" dedup machinery (the `(dial, URL)` plan key) — and it
|
||||
breaks the D18 board: a verdict is a *(record, TTL)* pair with TTL derived from
|
||||
the probe interval, so per-group intervals would make the same record mean
|
||||
different things to different readers. Xray's observatory has exactly one
|
||||
global `probeURL`/`probeInterval`; that is the model we mapped onto (§3).
|
||||
|
||||
**Decision: only `Globals.ProbeURL`/`Globals.ProbeInterval`.** They drive the
|
||||
native group checker, the observatory and the manual exit test alike (failover
|
||||
keeps its 30 s default interval). Probe-plan dedup collapses to the dial path.
|
||||
Migration is the standard dead-option drainage: old UCI configs carrying
|
||||
`probe_url`/`probe_interval` on a group parse silently and the options vanish
|
||||
on the next render.
|
||||
- **Rejected: keep per-group overrides.** Incomparable measurements across
|
||||
groups, undefined semantics for shared members, and a per-group TTL that
|
||||
fractures the single-board verdict.
|
||||
- **Deferred: per-tier probe cadence** (rule-critical tags more often). If ever
|
||||
needed it is a field on the observatory's `ProbeJob` — a scheduling knob,
|
||||
not a return of per-group *configuration*.
|
||||
Consequence: all delay numbers are comparable (least ping ranks apples against
|
||||
apples), and group settings lose two footgun fields while Settings keeps the
|
||||
two that actually govern every check.
|
||||
|
||||
@@ -112,8 +112,8 @@ func getGroupDelay(server *Server) func(w http.ResponseWriter, r *http.Request)
|
||||
} else {
|
||||
server.logger.Debug("outbound ", tag, " available: ", t, "ms")
|
||||
server.urlTestHistory.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: t,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: t,
|
||||
})
|
||||
resultAccess.Lock()
|
||||
result[tag] = t
|
||||
|
||||
@@ -209,8 +209,8 @@ func getProxyDelay(server *Server) func(w http.ResponseWriter, r *http.Request)
|
||||
server.urlTestHistory.DeleteURLTestHistory(realTag)
|
||||
} else {
|
||||
server.urlTestHistory.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: delay,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: delay,
|
||||
})
|
||||
}
|
||||
}()
|
||||
|
||||
@@ -25,7 +25,7 @@ LUCI_DEPENDS:=+shater-core +rpcd
|
||||
LUCI_PKGARCH:=all
|
||||
|
||||
PKG_VERSION:=0.2.0
|
||||
PKG_RELEASE:=1
|
||||
PKG_RELEASE:=2
|
||||
|
||||
PKG_MAINTAINER:=Shater <maqrota@icloud.com>
|
||||
PKG_LICENSE:=GPL-3.0-or-later
|
||||
|
||||
@@ -14,7 +14,7 @@ include $(TOPDIR)/rules.mk
|
||||
|
||||
PKG_NAME:=shater-core
|
||||
PKG_VERSION:=0.2.0
|
||||
PKG_RELEASE:=2
|
||||
PKG_RELEASE:=3
|
||||
|
||||
PKG_MAINTAINER:=Shater <maqrota@icloud.com>
|
||||
PKG_LICENSE:=GPL-2.0-or-later
|
||||
|
||||
@@ -35,7 +35,7 @@ include $(TOPDIR)/rules.mk
|
||||
|
||||
PKG_NAME:=shaterd
|
||||
PKG_VERSION:=0.2.0
|
||||
PKG_RELEASE:=2
|
||||
PKG_RELEASE:=3
|
||||
|
||||
PKG_MAINTAINER:=Shater <maqrota@icloud.com>
|
||||
PKG_LICENSE:=GPL-3.0-or-later
|
||||
|
||||
@@ -85,61 +85,6 @@
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
|
||||
/* ---- query log wrapper + honest empty state ---- */
|
||||
.log-wrap {
|
||||
margin-top: calc(var(--u, 8px) * 3);
|
||||
}
|
||||
/* DNS-filter readout that sits above the live query log. */
|
||||
.filter-readout {
|
||||
display: grid;
|
||||
grid-template-columns: minmax(0, 1.4fr) minmax(0, 1fr);
|
||||
gap: calc(var(--u, 8px) * 2);
|
||||
align-items: center;
|
||||
margin-bottom: calc(var(--u, 8px) * 1.5);
|
||||
}
|
||||
@media (max-width: 560px) {
|
||||
.filter-readout {
|
||||
grid-template-columns: 1fr;
|
||||
}
|
||||
}
|
||||
.topblocked {
|
||||
list-style: none;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 4px;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
}
|
||||
.topblocked li {
|
||||
display: flex;
|
||||
justify-content: space-between;
|
||||
gap: 8px;
|
||||
padding: 2px 6px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 3px;
|
||||
background: var(--recess, transparent);
|
||||
}
|
||||
.topblocked .tb-dom {
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
color: var(--ink, inherit);
|
||||
}
|
||||
.topblocked .tb-n {
|
||||
color: var(--accent, #e8823c);
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
.qempty {
|
||||
margin: 10px 2px 0;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11.5px;
|
||||
line-height: 1.6;
|
||||
letter-spacing: 0.02em;
|
||||
color: var(--faint);
|
||||
}
|
||||
|
||||
/* ---- apply / confirm / rollback controls ---- */
|
||||
.controls {
|
||||
display: flex;
|
||||
|
||||
+1
-1
@@ -35,7 +35,7 @@ const SOON: Record<Exclude<Route, 'overview'>, string> = {
|
||||
dns: 'Pick resolvers and blocklists — DoH/DoT upstreams, fakeip pool, and per-source DNS rules.',
|
||||
devices: 'See LAN clients and their traffic — per-device policy and activity at a glance.',
|
||||
insights: 'Traffic & DNS insights — top domains, per-rule bytes, exits, and per-device volume.',
|
||||
profiles: 'WAN-mode / failover profiles — auto-switch routing by uplink, probe, or schedule.',
|
||||
profiles: 'WAN-mode / failover profiles — auto-switch routing by the active uplink.',
|
||||
settings: 'Global settings — kill-switch, IPv6, marks/tables, commit-confirm, health probe.',
|
||||
apply: 'Review, apply, and roll back config changes with the commit-confirm safety window.',
|
||||
}
|
||||
|
||||
+78
-137
@@ -308,11 +308,11 @@ export interface GroupMemberHealth {
|
||||
* auto 119 / 122 alive tested 122 of 298
|
||||
* stealth 2 / 2 alive
|
||||
*
|
||||
* Untested must NEVER be folded into dead. A group probes its members LAZILY —
|
||||
* only while it is being used — so a freshly booted router with a 376-node
|
||||
* subscription legitimately has almost no history, and "3 alive / 373 dead"
|
||||
* would be an alarm raised at the moment nothing is wrong. `dead` is only ever a
|
||||
* POSITIVE finding: a probe ran and failed.
|
||||
* Untested must NEVER be folded into dead. The daemon's observatory probes a
|
||||
* group only when an enabled routing rule can reach it (see `used`), and only a
|
||||
* probe that RAN and FAILED writes `dead` — so a freshly (re)started engine
|
||||
* legitimately reads mostly untested for a few seconds, and an unused group
|
||||
* reads untested forever. Neither is an alarm.
|
||||
*
|
||||
* `bound: true` means every member is a per-group egress COPY — this health was
|
||||
* measured through the group's own egress and is deliberately NOT comparable
|
||||
@@ -326,6 +326,12 @@ export interface GroupHealth {
|
||||
group: string
|
||||
type: string // "selector" | "urltest"
|
||||
bound: boolean
|
||||
/** An enabled routing rule (the Final target, a DNS-resolver detour, a device
|
||||
* target, …) reaches this group, so the observatory probes its members in
|
||||
* the background. false ⇒ nothing routes through the group: it is skipped by
|
||||
* the background probing and every member stays `untested`. That is an
|
||||
* "unused" note about the ROUTING CONFIG, never a health problem. */
|
||||
used: boolean
|
||||
/** Node name the group routes through right now; '' when it hasn't picked. */
|
||||
selected: string
|
||||
total: number
|
||||
@@ -338,61 +344,38 @@ export interface GroupHealth {
|
||||
members?: GroupMemberHealth[]
|
||||
}
|
||||
|
||||
/**
|
||||
* The daemon's background health sweep — the thing that keeps these numbers
|
||||
* filling in without anyone pressing a button. Absent on older daemons, in which
|
||||
* case the UI must not promise that untested members will resolve on their own.
|
||||
*/
|
||||
export interface HealthSweep {
|
||||
enabled: boolean
|
||||
cursor: number
|
||||
total: number
|
||||
cycles: number
|
||||
}
|
||||
|
||||
/**
|
||||
* GET /api/groups/health. `groups` is ALWAYS an array, never null.
|
||||
*
|
||||
* The `node_test_*` fields mirror GET /api/nodes/test: the probe-all run is what
|
||||
* turns a group's untested members into a real alive/dead verdict, so one poll of
|
||||
* this endpoint drives the numbers, the button and its progress readout.
|
||||
* The numbers fill in on their own: the daemon's observatory probes every group
|
||||
* and chain an enabled routing rule can reach (a ~10 s tick on the global Probe
|
||||
* URL / Interval), so a used group's untested members resolve to a real
|
||||
* alive/dead verdict within seconds. There is no manual run-everything button
|
||||
* any more and nothing to report here beyond the groups themselves.
|
||||
*/
|
||||
|
||||
/** Per-chain reachability, the chain analogue of {@link GroupHealth}.used (plan
|
||||
* §5.E): a chain no enabled routing rule routes through is outside the
|
||||
* observatory's plan, so its exit is never probed and the Targets card renders it
|
||||
* "unused" instead of an exit-test readout. A chain has no membership counters —
|
||||
* it is a fixed path, and its end-to-end health is the exit test's job. */
|
||||
export interface ChainHealth {
|
||||
name: string
|
||||
/** An enabled routing rule (the Final target, a DNS-resolver detour, a device
|
||||
* target, …) reaches this chain, so the observatory probes its exit in the
|
||||
* background. false ⇒ nothing routes through the chain: it is skipped by the
|
||||
* background probing and its end-to-end health stays untested. That is an
|
||||
* "unused" note about the ROUTING CONFIG, never a health problem. */
|
||||
used: boolean
|
||||
}
|
||||
|
||||
export interface GroupsHealth {
|
||||
groups: GroupHealth[]
|
||||
node_test_running: boolean
|
||||
node_test_done: number
|
||||
node_test_total: number
|
||||
sweep?: HealthSweep
|
||||
}
|
||||
|
||||
/**
|
||||
* The health run's scope, as the daemon publishes it (panel/api.go NodeTestScope).
|
||||
*
|
||||
* It is a CONSTANT, not a list, and that is the whole point: a health run measures
|
||||
* every node, every endpoint and every group's egress copies in one pass, so it can
|
||||
* never be attributed to one card. Render it as a single global progress indicator.
|
||||
* The scoped counterpart is {@link GroupTestStatus.scope}.
|
||||
*/
|
||||
export const NODE_TEST_SCOPE = 'all_nodes'
|
||||
export type NodeTestScope = typeof NODE_TEST_SCOPE
|
||||
|
||||
/**
|
||||
* GET /api/nodes/test — progress of a manual "Test all nodes" probe-all run.
|
||||
* `running` is true while a run is in flight; `done`/`total` count finished vs
|
||||
* targeted node probes. Idle (never run, or finished) reads `{running:false}`.
|
||||
*/
|
||||
export interface NodeTestStatus {
|
||||
running: boolean
|
||||
done: number
|
||||
total: number
|
||||
/** Always {@link NODE_TEST_SCOPE}; absent on daemons older than the split. */
|
||||
scope?: NodeTestScope
|
||||
}
|
||||
|
||||
/** POST /api/nodes/test reply: `started` when a fresh run began, else `running`. */
|
||||
export interface NodeTestStart {
|
||||
started?: boolean
|
||||
running?: boolean
|
||||
/** Per-chain reachability, same "unused" badge as groups (plan §5.E). Present on
|
||||
* the summary and `?members=` shapes; the single-group (`?group=`) shape is a
|
||||
* group detail request and omits it. Always an array when present; absent ⇒ not
|
||||
* reported by this daemon version (the page treats absence like "not known yet"). */
|
||||
chains?: ChainHealth[]
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -463,23 +446,6 @@ export interface Globals {
|
||||
EndpointResolver: string
|
||||
ProbeURL: string
|
||||
ProbeInterval: string
|
||||
/**
|
||||
* Tick period of the daemon's background health sweep — the walk that keeps
|
||||
* every node's health fresh so the group cards fill in without anyone pressing
|
||||
* a button (model.Globals.SweepInterval, UCI `sweep_interval`).
|
||||
*
|
||||
* `""` means ENABLED at the engine's default tick, NOT off. That default is
|
||||
* load-bearing: urltest groups probe only while they are being used and selector
|
||||
* groups never probe at all, so without the sweep a 376-node subscription reads
|
||||
* almost entirely "not measured".
|
||||
*
|
||||
* Accepted: a duration ("30s", "5m", "1h") or a bare integer of seconds; any of
|
||||
* `0` / `off` / `none` / `disabled` to switch the sweep off. Below 5 s the
|
||||
* daemon raises it to that floor. An UNRECOGNISED value does not disable the
|
||||
* sweep — the daemon warns and falls back to the default — so the panel refuses
|
||||
* it at the input instead of saving something that will be silently ignored.
|
||||
*/
|
||||
SweepInterval: string
|
||||
SchemaVersion: number
|
||||
ActiveProfile: string
|
||||
/**
|
||||
@@ -508,14 +474,17 @@ export interface Globals {
|
||||
DNSIntercept?: boolean // force ALL LAN plaintext DNS (:53) through the engine, incl. router-addressed queries
|
||||
BlockDoH?: boolean // block known public DoH resolvers (by host + IP:443 + Firefox canary) so clients fall back to plaintext :53
|
||||
/**
|
||||
* Our background health sweep of group members — the walk that fills the alive /
|
||||
* The daemon's observatory — the background probing that fills the alive /
|
||||
* dead / tested numbers the Targets page shows for each group (model.Globals
|
||||
* .GroupHealth, UCI `group_health`). Default ON.
|
||||
* .GroupHealth, UCI `group_health`). It probes only the groups and chains an
|
||||
* enabled routing rule can reach, on the global Probe URL / Interval, and
|
||||
* skips everything else. Default ON.
|
||||
*
|
||||
* This governs OUR sweep and the health/testing UI ONLY. It does NOT touch
|
||||
* sing-box's own internal urltest / least_test probes: a group always keeps
|
||||
* picking a live member under the hood regardless of this switch. Turning it off
|
||||
* just stops the extra sweep and hides the health statistics.
|
||||
* This gates the observatory and the health/testing UI ONLY. It does NOT
|
||||
* touch sing-box's own internal urltest / least_test probes: a group always
|
||||
* keeps picking a live member under the hood regardless of this switch.
|
||||
* Turning it off stops the background probing and hides the health
|
||||
* statistics.
|
||||
*
|
||||
* ABSENT ⇒ enabled (an older config never wrote the key), so the invariant the
|
||||
* panel reads by is `globals?.GroupHealth !== false` — never `=== true`.
|
||||
@@ -662,8 +631,6 @@ export interface Group {
|
||||
FilterProto?: string[] | null
|
||||
FilterCountry?: string[] | null
|
||||
Dedup?: boolean
|
||||
ProbeURL?: string
|
||||
ProbeInterval?: string
|
||||
/**
|
||||
* The egress EVERY node in this group dials its own server through — the same
|
||||
* binding `Node.Egress` gives one node, applied to the whole group. "" ⇒ the
|
||||
@@ -731,14 +698,6 @@ export interface RulesetStatus {
|
||||
rule_count: number
|
||||
}
|
||||
|
||||
/** A quick rule on/off bundle (config preset). */
|
||||
export interface Preset {
|
||||
Name: string
|
||||
Enabled: boolean
|
||||
Order?: number
|
||||
Target?: string
|
||||
}
|
||||
|
||||
/** A WAN-mode / failover conditional override (config profile). */
|
||||
export interface Profile {
|
||||
Name: string
|
||||
@@ -752,23 +711,8 @@ export interface Profile {
|
||||
// condition — so adding one SWITCHED OFF an otherwise working profile. Both were
|
||||
// deleted from the Go model; PUT decodes with DisallowUnknownFields, so sending
|
||||
// either key fails the whole write with 400.
|
||||
SchedDays?: string[] | null
|
||||
SchedStart?: string
|
||||
SchedEnd?: string
|
||||
/**
|
||||
* Minutes EAST of UTC that anchor the schedule's wall-clock times (Moscow =
|
||||
* 180, New York winter = −300). The router carries no IANA tzdata, so the
|
||||
* daemon evaluates the window at UTC+offset; the panel captures the editing
|
||||
* browser's offset whenever a schedule field is saved. 0/absent ⇒ UTC.
|
||||
* Known limitation: a fixed offset does not follow DST until re-saved.
|
||||
* (Replaces the deleted SchedTZ — an IANA name never resolved on the router,
|
||||
* so the promised local time was silently UTC; see model.go.)
|
||||
*/
|
||||
SchedUTCOffset?: number
|
||||
EnableRules?: string[] | null
|
||||
DisableRules?: string[] | null
|
||||
DefaultTarget?: string
|
||||
DefaultEgress?: string
|
||||
/** Per-profile override of the endpoint resolver (keyed by active WAN: SIM→yandex, WiFi→DoH). "" ⇒ no override (model.Profile.EndpointResolver, UCI `endpoint_resolver`). */
|
||||
EndpointResolver?: string
|
||||
}
|
||||
@@ -879,12 +823,19 @@ export interface Rule {
|
||||
Proto?: string // '' | tcp | udp | tls | http | quic | dns | stun | bittorrent | dtls | ssh | rdp | ntp
|
||||
Target?: string
|
||||
Egress?: string
|
||||
Kill?: string
|
||||
/**
|
||||
* Policy for when the target can't resolve at generate time (dead group,
|
||||
* broken chain, missing egress/node). The rule is always still emitted — its
|
||||
* traffic never falls through to the default route. ''/'default'/'closed' and
|
||||
* anything unrecognised block the traffic (fail-closed); 'open' is an explicit,
|
||||
* warned kill-switch bypass that sends it direct.
|
||||
*/
|
||||
Kill?: string // '' | default | closed | open
|
||||
SchedEnabled?: boolean
|
||||
SchedDays?: string[] | null
|
||||
SchedStart?: string
|
||||
SchedEnd?: string
|
||||
/** Minutes east of UTC anchoring SchedStart/SchedEnd/SchedDays — same contract as Profile.SchedUTCOffset. */
|
||||
/** Minutes east of UTC anchoring SchedStart/SchedEnd/SchedDays — the panel captures the editing browser's offset on save (the router has no tzdata). */
|
||||
SchedUTCOffset?: number
|
||||
}
|
||||
|
||||
@@ -958,7 +909,8 @@ export interface Interface {
|
||||
|
||||
/** GET /api/devices row — a discovered LAN client merged with its config, if any. */
|
||||
export interface DiscoveredDevice {
|
||||
ip: string
|
||||
ip: string // primary address (most recent lease)
|
||||
ips: string[] // every known address (v4+v6, multiple leases), primary first
|
||||
mac: string
|
||||
hostname: string
|
||||
online: boolean
|
||||
@@ -988,7 +940,6 @@ export interface Model {
|
||||
Devices?: Device[] | null
|
||||
Chains?: Chain[] | null
|
||||
Rulesets?: Ruleset[] | null
|
||||
Presets?: Preset[] | null
|
||||
Profiles?: Profile[] | null
|
||||
Inbounds?: Inbound[] | null
|
||||
DNSRules?: DNSRule[] | null
|
||||
@@ -1357,21 +1308,6 @@ export function updateSubscription(name: string): Promise<{ added: number }> {
|
||||
: req('api/subscription/update', { method: 'POST', body: JSON.stringify({ name }) })
|
||||
}
|
||||
|
||||
/**
|
||||
* POST /api/nodes/test — force a health probe of every node. The daemon runs it in
|
||||
* the background (singleton: a second POST while one is running is a no-op that
|
||||
* reports `{running:true}`); the normal /api/stats poll then surfaces each fresh
|
||||
* result. Resolves to `{started}` or `{running}`.
|
||||
*/
|
||||
export function postNodesTest(): Promise<NodeTestStart> {
|
||||
return MOCK ? mock.postNodesTest() : req<NodeTestStart>('api/nodes/test', { method: 'POST' })
|
||||
}
|
||||
|
||||
/** GET /api/nodes/test — progress of the current/last probe-all run. */
|
||||
export function getNodesTest(): Promise<NodeTestStatus> {
|
||||
return MOCK ? mock.getNodesTest() : req<NodeTestStatus>('api/nodes/test')
|
||||
}
|
||||
|
||||
/**
|
||||
* GET /api/groups/health — per-group membership health (see {@link GroupHealth}).
|
||||
*
|
||||
@@ -1397,20 +1333,24 @@ export function getGroupsHealth(
|
||||
}
|
||||
|
||||
/**
|
||||
* One group's last test: which member the balancer picked, how fast it answered,
|
||||
* and what the internet saw as the source address.
|
||||
* One group's (or chain's) last test: which member the balancer picked, how fast
|
||||
* it answered, and what the internet saw as the source address.
|
||||
*
|
||||
* `ok:true` with an EMPTY `exit_ip`/`exit_country` is a valid, successful result,
|
||||
* not a partial failure: the delay was measured but the exit address could not be
|
||||
* determined (the lookup service was unreachable, or the answer wasn't parseable).
|
||||
* Render it as a success with an unknown address — never as an error.
|
||||
*
|
||||
* Chains ride the same endpoint. For a chain row, `group` carries the CHAIN's
|
||||
* name and `selected` the node its last group hop picked ('' when the exit hop
|
||||
* isn't a group). Everything else reads the same way.
|
||||
*
|
||||
* `ok:false` ⇒ the test failed and `error` carries the human reason; every other
|
||||
* field is meaningless. `tested_unix` is the router's clock, in seconds.
|
||||
*/
|
||||
export interface GroupTestResult {
|
||||
group: string
|
||||
selected: string // the member node the group chose for this test
|
||||
group: string // group name — or a chain name for a chain row
|
||||
selected: string // the member node the group (or the chain's exit group) chose
|
||||
delay_ms: number
|
||||
exit_ip: string // may be '' even when ok
|
||||
exit_country: string // ISO code; may be '' even when ok
|
||||
@@ -1421,25 +1361,26 @@ export interface GroupTestResult {
|
||||
|
||||
/**
|
||||
* GET /api/groups/test — progress plus every result so far. `results` is ALWAYS
|
||||
* an array (never null); `done`/`total` count finished vs targeted groups while
|
||||
* `running` is true. Idle reads `{running:false}` with the last run's results
|
||||
* still attached, so a reload after a test still shows what it found.
|
||||
* an array (never null); `done`/`total` count finished vs targeted groups and
|
||||
* chains while `running` is true. Idle reads `{running:false}` with the last
|
||||
* run's results still attached, so a reload after a test still shows what it found.
|
||||
*/
|
||||
export interface GroupTestStatus {
|
||||
running: boolean
|
||||
done: number
|
||||
total: number
|
||||
/**
|
||||
* The group names THIS run covers. Always an array (never JSON null); absent
|
||||
* only on daemons older than the split.
|
||||
* The group and chain names THIS run covers. Always an array (never JSON
|
||||
* null); absent only on daemons older than the split.
|
||||
*
|
||||
* It is what makes `running` usable. On its own that flag says only "a group
|
||||
* test is happening somewhere", which is why pressing Test on one group used to
|
||||
* put "measuring…" on every card. The rule: show the in-progress indicator on
|
||||
* card g iff `running && scope.includes(g)`. A run started with a name carries
|
||||
* exactly that name; a run started with no name carries every group, and then
|
||||
* the indicator on every card is correct. The scope PERSISTS after the run
|
||||
* ends, so displayed results stay attributable to the cards they came from.
|
||||
* exactly that name; a run started with no name carries every group and
|
||||
* every chain, and then the indicator on every card is correct. The scope
|
||||
* PERSISTS after the run ends, so displayed results stay attributable to the
|
||||
* cards they came from.
|
||||
*/
|
||||
scope?: string[]
|
||||
results: GroupTestResult[]
|
||||
@@ -1455,10 +1396,10 @@ export interface GroupTestStart {
|
||||
}
|
||||
|
||||
/**
|
||||
* POST /api/groups/test — measure a group's delay and exit address. Pass a group
|
||||
* name to test one; pass nothing (or '') to test every group. Singleton: a second
|
||||
* call while a run is in flight resolves to `{started:false, reason:'already
|
||||
* running'}` rather than failing.
|
||||
* POST /api/groups/test — measure a target's delay and exit address. Pass a
|
||||
* group or chain name to test one; pass nothing (or '') to test every group
|
||||
* and every chain. Singleton: a second call while a run is in flight resolves
|
||||
* to `{started:false, reason:'already running'}` rather than failing.
|
||||
*/
|
||||
export function postGroupsTest(name = ''): Promise<GroupTestStart> {
|
||||
return MOCK
|
||||
|
||||
+80
-140
@@ -6,7 +6,7 @@
|
||||
// state mutates in-memory so the Apply / Confirm / Rollback flow is exercisable.
|
||||
//
|
||||
// Type-only imports from api.ts (erased at build) keep this free of a runtime cycle.
|
||||
import type { ApplyResult, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, NodeTestStart, NodeTestStatus, QueryLogEntry, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning } from './api'
|
||||
import type { ApplyResult, ChainHealth, ConnLogEntry, DiscoveredDevice, GroupHealth, GroupMemberHealth, GroupsHealth, GroupTestResult, GroupTestStart, GroupTestStatus, Interface, Model, QueryLogEntry, RulesetCategories, RulesetCheck, RulesetStatus, Stats, StatsLogPage, StatsLogQuery, Status, StatusWarning } from './api'
|
||||
|
||||
let armed = false // a pending commit-confirm auto-rollback
|
||||
let hasLastGood = false // a predecessor config exists to roll back to (post-apply)
|
||||
@@ -26,9 +26,6 @@ const CONFIG: Model = {
|
||||
EndpointResolver: '',
|
||||
ProbeURL: 'https://www.gstatic.com/generate_204',
|
||||
ProbeInterval: '60s',
|
||||
// '' = the engine's default tick (every 10 s), NOT off. Left blank on purpose
|
||||
// so `?mock` shows the recommended state and the placeholder that says so.
|
||||
SweepInterval: '',
|
||||
SchemaVersion: 2,
|
||||
// Points at the iface-driven profile below, so `?mock` lands in the WAN-watcher
|
||||
// AUTO-PIN state (not a manual override): the plate must say the router pins this
|
||||
@@ -39,7 +36,7 @@ const CONFIG: Model = {
|
||||
Untunnelable: new URLSearchParams(typeof location === 'undefined' ? '' : location.search).get('untun') ?? 'block',
|
||||
DNSIntercept: true, // force ALL LAN plaintext DNS (:53) through the engine
|
||||
BlockDoH: false, // block known public DoH resolvers so clients fall back to plaintext :53
|
||||
GroupHealth: true, // background group-member health sweep + Targets health stats (default on)
|
||||
GroupHealth: true, // observatory: background probing of used groups/chains + Targets health stats (default on)
|
||||
StatsRingSize: 500, // fixed cap — shows the "limit" rendering (500 rows)
|
||||
StatsTimelineMinutes: 0, // 0 ⇒ "Unlimited" rendering + memory warning
|
||||
StatsMaxDomains: 5000, // fixed cap — shows the "limit" rendering (5000)
|
||||
@@ -132,6 +129,9 @@ const CONFIG: Model = {
|
||||
{ Name: 'via-tunnel', Source: 'subscription', Subscription: 'primary', Strategy: 'leastping', Egress: 'awg' },
|
||||
{ Name: 'fallback', Source: 'subscription', Subscription: 'backup', Strategy: 'roundrobin', Egress: '' },
|
||||
],
|
||||
// One multi-hop chain so `?mock` exercises the chain card's Test button and
|
||||
// its result readout: enters through the awg tunnel, exits via the auto group.
|
||||
Chains: [{ Name: 'relay', Hops: ['egress:awg', 'group:auto'] }],
|
||||
Egresses: [
|
||||
{ Name: 'wan', Type: 'interface', Interface: 'wan' },
|
||||
// An AmneziaWG tunnel — the whole point of a group-level egress binding.
|
||||
@@ -240,20 +240,14 @@ const CONFIG: Model = {
|
||||
Enabled: true,
|
||||
Priority: 20,
|
||||
MatchIface: ['wan1'],
|
||||
SchedDays: [],
|
||||
EnableRules: ['default-tunnel'],
|
||||
DisableRules: ['block-ads', 'ru-bypass', 'private-direct'],
|
||||
DefaultTarget: 'group:auto',
|
||||
},
|
||||
{
|
||||
Name: 'evening-direct',
|
||||
Enabled: true,
|
||||
Priority: 10,
|
||||
MatchIface: [],
|
||||
SchedDays: ['mon', 'tue', 'wed', 'thu', 'fri'],
|
||||
SchedStart: '19:00',
|
||||
SchedEnd: '23:00',
|
||||
DefaultTarget: 'direct',
|
||||
},
|
||||
],
|
||||
}
|
||||
@@ -861,11 +855,13 @@ export async function getStatsConnsPage(q: StatsLogQuery = {}): Promise<StatsLog
|
||||
// derives them the same way — otherwise managing a device in `?mock` would leave the
|
||||
// discovery row stubbornly claiming the opposite. Includes lan + guest networks.
|
||||
const MOCK_HOSTS: ReadonlyArray<Omit<DiscoveredDevice, 'configured' | 'name' | 'blockCount'>> = [
|
||||
{ ip: '192.168.1.20', mac: 'a4:83:e7:11:22:33', hostname: 'macbook-max', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.31', mac: 'f0:18:98:aa:bb:cc', hostname: 'iphone-lena', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.42', mac: '3c:22:fb:44:55:66', hostname: 'ipad-kids', online: true, state: 'idle', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.55', mac: 'dc:a6:32:77:88:99', hostname: 'tv-livingroom', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '10.20.0.14', mac: 'b8:27:eb:aa:00:11', hostname: '', online: false, state: 'offline', network: 'guest', iface: 'br-guest' },
|
||||
// Dual-stacked (v4 lease + link-local v6) — exercises the merged-row display:
|
||||
// one card, primary IP prominent, the extra address as a secondary chip.
|
||||
{ ip: '192.168.1.20', ips: ['192.168.1.20', 'fe80::a683:e7ff:fe11:2233'], mac: 'a4:83:e7:11:22:33', hostname: 'macbook-max', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.31', ips: ['192.168.1.31'], mac: 'f0:18:98:aa:bb:cc', hostname: 'iphone-lena', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.42', ips: ['192.168.1.42'], mac: '3c:22:fb:44:55:66', hostname: 'ipad-kids', online: true, state: 'idle', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '192.168.1.55', ips: ['192.168.1.55'], mac: 'dc:a6:32:77:88:99', hostname: 'tv-livingroom', online: true, state: 'online', network: 'lan', iface: 'br-lan' },
|
||||
{ ip: '10.20.0.14', ips: ['10.20.0.14'], mac: 'b8:27:eb:aa:00:11', hostname: '', online: false, state: 'offline', network: 'guest', iface: 'br-guest' },
|
||||
]
|
||||
|
||||
export async function getDevices(): Promise<DiscoveredDevice[]> {
|
||||
@@ -906,9 +902,8 @@ export async function getInterfaces(): Promise<Interface[]> {
|
||||
//
|
||||
// The fixture is a real member LIST per group, not four hand-written counters:
|
||||
// the counters are derived from it, so the daemon's invariants (tested ==
|
||||
// alive+dead, alive+dead+untested == total) hold by construction and cannot drift
|
||||
// as the mock probe-all run flips members. Between them the four groups cover
|
||||
// every state the panel has to render:
|
||||
// alive+dead, alive+dead+untested == total) hold by construction and cannot
|
||||
// drift. Between them the four groups cover every state the panel has to render:
|
||||
//
|
||||
// auto 298 members, 122 measured — 119 alive / 3 dead, 176 untested.
|
||||
// The ordinary case: healthy, with a real dead finding AND a large
|
||||
@@ -918,11 +913,9 @@ export async function getInterfaces(): Promise<Interface[]> {
|
||||
// via-tunnel the same subscription `auto` draws from, dialled through `awg` —
|
||||
// and 0 of 6 alive through it. Two groups, one node set, two
|
||||
// verdicts: the entire reason health is reported per group.
|
||||
// fallback 24 members, NOTHING measured. Neither healthy nor broken — the
|
||||
// state that invites a measurement instead of raising an alarm.
|
||||
//
|
||||
// `?mock&nodetest=1` starts the page with a probe-all already in flight, so the
|
||||
// running/progress rendering is reachable without racing a click.
|
||||
// fallback 24 members, NOTHING measured, and `used:false` — no enabled rule
|
||||
// routes through it, so the observatory never probes it. The card
|
||||
// renders the quiet "unused" note instead of health counters.
|
||||
|
||||
/** City pool for synthesised member names — the real feed looks like this. */
|
||||
const MEMBER_CITIES = [
|
||||
@@ -941,40 +934,42 @@ interface HealthShape {
|
||||
dead: number
|
||||
/** true ⇒ members are per-group egress copies (health measured through it). */
|
||||
bound: boolean
|
||||
/** An enabled rule reaches this group (GroupHealth.used). false ⇒ the
|
||||
* observatory skips it and its members stay untested. */
|
||||
used: boolean
|
||||
type: string
|
||||
/** Node name the group routes through right now; '' when it hasn't picked. */
|
||||
selectedIndex: number | null
|
||||
}
|
||||
|
||||
const HEALTH_SHAPE: Record<string, HealthShape> = {
|
||||
auto: { total: 298, alive: 119, dead: 3, bound: false, type: 'urltest', selectedIndex: 1 },
|
||||
stealth: { total: 2, alive: 2, dead: 0, bound: true, type: 'urltest', selectedIndex: 0 },
|
||||
'via-tunnel': { total: 6, alive: 0, dead: 6, bound: true, type: 'urltest', selectedIndex: null },
|
||||
fallback: { total: 24, alive: 0, dead: 0, bound: false, type: 'urltest', selectedIndex: null },
|
||||
auto: { total: 298, alive: 119, dead: 3, bound: false, used: true, type: 'urltest', selectedIndex: 1 },
|
||||
stealth: { total: 2, alive: 2, dead: 0, bound: true, used: true, type: 'urltest', selectedIndex: 0 },
|
||||
'via-tunnel': { total: 6, alive: 0, dead: 6, bound: true, used: true, type: 'urltest', selectedIndex: null },
|
||||
fallback: { total: 24, alive: 0, dead: 0, bound: false, used: false, type: 'urltest', selectedIndex: null },
|
||||
}
|
||||
|
||||
/**
|
||||
* `?mock&bias=1|2|3` reproduces the optimistic-ratio defect the owner hit on the
|
||||
* live router, on `auto` — 298 members, the same size he reported — and walks it
|
||||
* through the three states the rendering has to tell apart.
|
||||
* `?mock&bias=1|2` reproduces the optimistic-ratio defect the owner hit on the
|
||||
* live router, on `auto` — 298 members, the same size he reported.
|
||||
*
|
||||
* The mechanism: a group writes an entry when a member answers and DELETES it when
|
||||
* one doesn't, so until the background sweep has been over the group its history
|
||||
* holds nothing but successes. `11 / 11 alive` was that, not health.
|
||||
* The mechanism: a group's own probes write an entry when a member answers and
|
||||
* DELETE it when one doesn't, so until the daemon's board has a failure on
|
||||
* record the history holds nothing but successes. `11 / 11 alive` was that,
|
||||
* not health.
|
||||
*
|
||||
* bias=1 → 11 alive, 0 dead, 287 unchecked, sweep mid-first-pass → NO RATIO
|
||||
* bias=2 → 111 alive, 74 dead, 113 unchecked, sweep mid-first-pass → ratio (a
|
||||
* failure is on record, so something is writing both outcomes)
|
||||
* bias=3 → the same counts with a completed pass → ratio
|
||||
* bias=1 → 11 alive, 0 dead, 287 unchecked → NO RATIO (one-sided sample)
|
||||
* bias=2 → 111 alive, 74 dead, 113 unchecked → ratio (a failure is on record,
|
||||
* so something is writing both outcomes)
|
||||
*
|
||||
* Read 1 → 2 → 3 in order: the reading must get MORE PRECISE, never "good, then
|
||||
* Read 1 → 2 in order: the reading must get MORE PRECISE, never "good, then
|
||||
* suddenly bad".
|
||||
*/
|
||||
const BIAS_STAGE =
|
||||
typeof location === 'undefined' ? null : new URLSearchParams(location.search).get('bias')
|
||||
if (BIAS_STAGE === '1') {
|
||||
HEALTH_SHAPE.auto = { ...HEALTH_SHAPE.auto, alive: 11, dead: 0 }
|
||||
} else if (BIAS_STAGE === '2' || BIAS_STAGE === '3') {
|
||||
} else if (BIAS_STAGE === '2') {
|
||||
HEALTH_SHAPE.auto = { ...HEALTH_SHAPE.auto, alive: 111, dead: 74 }
|
||||
}
|
||||
|
||||
@@ -1000,13 +995,13 @@ function buildMembers(group: string, shape: HealthShape): GroupMemberHealth[] {
|
||||
return out
|
||||
}
|
||||
|
||||
/** Live member state, keyed by group name. Mutated by the mock probe-all run. */
|
||||
/** Live member state, keyed by group name. */
|
||||
const GROUP_MEMBERS = new Map<string, GroupMemberHealth[]>(
|
||||
Object.entries(HEALTH_SHAPE).map(([g, s]) => [g, buildMembers(g, s)]),
|
||||
)
|
||||
|
||||
/** Derive one group's summary from its member list — never hand-written, so the
|
||||
* daemon's counter invariants hold no matter what the probe-all run did. */
|
||||
* daemon's counter invariants hold by construction. */
|
||||
function summarise(group: string, members: GroupMemberHealth[]): GroupHealth {
|
||||
const shape = HEALTH_SHAPE[group]
|
||||
let alive = 0
|
||||
@@ -1024,6 +1019,7 @@ function summarise(group: string, members: GroupMemberHealth[]): GroupHealth {
|
||||
group,
|
||||
type: shape?.type ?? 'urltest',
|
||||
bound: shape?.bound ?? false,
|
||||
used: shape?.used ?? true,
|
||||
selected: sel != null && members[sel] ? members[sel].node : '',
|
||||
total: members.length,
|
||||
tested: alive + dead,
|
||||
@@ -1050,88 +1046,31 @@ function healthList(): GroupHealth[] {
|
||||
return (CONFIG.Groups ?? []).map((g) => summarise(g.Name, GROUP_MEMBERS.get(g.Name) ?? []))
|
||||
}
|
||||
|
||||
// Mock "Test all nodes" probe-all: a run that advances a batch of members per GET
|
||||
// poll so the progress readout and — the point of the run — untested members
|
||||
// turning into a real alive/dead verdict are both exercisable offline.
|
||||
let nodeTest: NodeTestStatus = { running: false, done: 0, total: 0 }
|
||||
/** Members still to be measured by the in-flight run, as [group, index]. */
|
||||
let nodeTestQueue: Array<[string, number]> = []
|
||||
|
||||
function startNodeTest(): void {
|
||||
nodeTestQueue = []
|
||||
for (const [group, members] of GROUP_MEMBERS) {
|
||||
members.forEach((_, i) => nodeTestQueue.push([group, i]))
|
||||
}
|
||||
nodeTest = { running: true, done: 0, total: nodeTestQueue.length }
|
||||
/** Per-chain reachability for the Targets page's "unused" badge (plan §5.E) — the
|
||||
* chain analogue of healthList's `used` field. The mock's single chain `relay` is
|
||||
* NOT referenced by any rule in CONFIG.Rules (they target group:auto / block /
|
||||
* direct), so it reads used=false and its card renders "unused" — exactly the case
|
||||
* the badge exists to surface. A stopped engine reports no chains. */
|
||||
function chainHealthList(): ChainHealth[] {
|
||||
if (!mockPlane().engine) return []
|
||||
return (CONFIG.Chains ?? []).map((c) => ({ name: c.Name, used: chainUsed(c.Name) }))
|
||||
}
|
||||
|
||||
/** Advance the run by one poll's worth of probes, flipping untested members to a
|
||||
* verdict. Roughly 1 in 12 comes back dead, so the numbers move believably. */
|
||||
function advanceNodeTest(): void {
|
||||
if (!nodeTest.running) return
|
||||
const batch = Math.max(1, Math.round(nodeTest.total / 14))
|
||||
for (let n = 0; n < batch; n++) {
|
||||
const next = nodeTestQueue.shift()
|
||||
if (!next) break
|
||||
const [group, i] = next
|
||||
const m = GROUP_MEMBERS.get(group)?.[i]
|
||||
if (m) {
|
||||
// A bound group stays dead through its tunnel — measuring it again does not
|
||||
// make it work, and pretending otherwise would hide the case the feature
|
||||
// exists to show.
|
||||
const dead = HEALTH_SHAPE[group]?.bound && HEALTH_SHAPE[group]?.alive === 0 ? true : i % 12 === 7
|
||||
m.state = dead ? 'dead' : 'alive'
|
||||
m.delay_ms = dead ? 0 : 38 + ((i * 37) % 460)
|
||||
m.age_seconds = 1 + (i % 5)
|
||||
}
|
||||
nodeTest = { ...nodeTest, done: nodeTest.done + 1 }
|
||||
}
|
||||
if (nodeTestQueue.length === 0) nodeTest = { ...nodeTest, running: false }
|
||||
}
|
||||
|
||||
// ?mock&nodetest=1 lands straight in the running state (see the note above). This
|
||||
// is the HEALTH run — global by nature, so the panel must render it as ONE
|
||||
// indicator in the section header and never as a per-card badge.
|
||||
if (typeof location !== 'undefined' && new URLSearchParams(location.search).has('nodetest')) {
|
||||
startNodeTest()
|
||||
}
|
||||
|
||||
/**
|
||||
* The background sweep's progress, as GET /api/groups/health reports it.
|
||||
*
|
||||
* `cycles` is the field with teeth: 0 means the sweep has not been everywhere yet,
|
||||
* so "not measured" is simply "not reached". Once it is ≥ 1 the sweep HAS been
|
||||
* everywhere, and anything still unmeasured lost a reading it used to have.
|
||||
*
|
||||
* ?mock&sweep=first → mid first pass (cycles 0)
|
||||
* ?mock&sweep=off → sweep disabled; nothing fills in on its own
|
||||
*/
|
||||
function mockSweep(): { enabled: boolean; cursor: number; total: number; cycles: number } {
|
||||
const mode = typeof location === 'undefined' ? null : new URLSearchParams(location.search).get('sweep')
|
||||
if (mode === 'off') return { enabled: false, cursor: 0, total: 0, cycles: 0 }
|
||||
if (mode === 'first') return { enabled: true, cursor: 184, total: 707, cycles: 0 }
|
||||
// The bias walkthrough drives the sweep too: stages 1 and 2 are mid-first-pass
|
||||
// (the cursor advances between them, exactly as the live router's did), stage 3
|
||||
// has completed one. See BIAS_STAGE.
|
||||
if (BIAS_STAGE === '1') return { enabled: true, cursor: 312, total: 896, cycles: 0 }
|
||||
if (BIAS_STAGE === '2') return { enabled: true, cursor: 696, total: 896, cycles: 0 }
|
||||
if (BIAS_STAGE === '3') return { enabled: true, cursor: 148, total: 896, cycles: 1 }
|
||||
return { enabled: true, cursor: 184, total: 707, cycles: 3 }
|
||||
/** A chain is "used" when some enabled routing rule (or Final, or a DNS detour)
|
||||
* targets `chain:<name>` — the same reachability the daemon's observatory derives.
|
||||
* The mock's rules never target a chain, so every chain reads used=false; a real
|
||||
* config would mark the ones rules point at used=true. */
|
||||
function chainUsed(name: string): boolean {
|
||||
const target = `chain:${name}`
|
||||
return (CONFIG.Rules ?? []).some(
|
||||
(r) => r.Enabled !== false && (r.Target === target),
|
||||
)
|
||||
}
|
||||
|
||||
export async function getGroupsHealth(
|
||||
opts: { group?: string; members?: boolean } = {},
|
||||
): Promise<GroupsHealth> {
|
||||
await wait(70)
|
||||
advanceNodeTest()
|
||||
const base: Omit<GroupsHealth, 'groups'> = {
|
||||
node_test_running: nodeTest.running,
|
||||
node_test_done: nodeTest.done,
|
||||
node_test_total: nodeTest.total,
|
||||
// The daemon's background sweep, on by default — it is what lets the panel
|
||||
// promise that untested members resolve without anyone pressing anything.
|
||||
sweep: mockSweep(),
|
||||
}
|
||||
if (opts.group) {
|
||||
const members = GROUP_MEMBERS.get(opts.group)
|
||||
// A stopped engine has no groups at all, so every name is a 404 — the same
|
||||
@@ -1141,21 +1080,20 @@ export async function getGroupsHealth(
|
||||
throw new ApiErrorLike(404, 'no such group in the running engine')
|
||||
}
|
||||
return {
|
||||
...base,
|
||||
groups: [{ ...summarise(opts.group, members), members: members.map((m) => ({ ...m })) }],
|
||||
}
|
||||
}
|
||||
const groups = healthList()
|
||||
if (opts.members) {
|
||||
return {
|
||||
...base,
|
||||
groups: groups.map((g) => ({
|
||||
...g,
|
||||
members: (GROUP_MEMBERS.get(g.group) ?? []).map((m) => ({ ...m })),
|
||||
})),
|
||||
chains: chainHealthList(),
|
||||
}
|
||||
}
|
||||
return { ...base, groups }
|
||||
return { groups, chains: chainHealthList() }
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1173,27 +1111,13 @@ class ApiErrorLike extends Error {
|
||||
}
|
||||
}
|
||||
|
||||
export async function postNodesTest(): Promise<NodeTestStart> {
|
||||
await wait(60)
|
||||
if (nodeTest.running) return { running: true }
|
||||
startNodeTest()
|
||||
return { started: true }
|
||||
}
|
||||
|
||||
export async function getNodesTest(): Promise<NodeTestStatus> {
|
||||
await wait(60)
|
||||
advanceNodeTest()
|
||||
// The constant the daemon publishes (api.ts NODE_TEST_SCOPE, spelled inline
|
||||
// because this module may only import TYPES from api.ts): this run is GLOBAL and
|
||||
// can never be attributed to one group's card.
|
||||
return { ...nodeTest, scope: 'all_nodes' }
|
||||
}
|
||||
|
||||
// Mock group test. Deliberately covers every state the UI has to render, one per
|
||||
// group, so a single offline run exercises all of them:
|
||||
// Mock group/chain test. Deliberately covers every state the UI has to render,
|
||||
// one per target, so a single offline run exercises all of them:
|
||||
// auto → ok WITH an exit address
|
||||
// stealth → ok WITHOUT one (delay measured, address undeterminable) — a
|
||||
// SUCCESS, and the case the UI most easily gets wrong
|
||||
// relay → the chain: same wire shape, `group` carries the CHAIN's name and
|
||||
// `selected` the node its exit group picked
|
||||
// fallback → a failure carrying a human reason
|
||||
// Results land one per GET poll, so the running/progress state is visible too.
|
||||
const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_unix'>> = {
|
||||
@@ -1213,6 +1137,15 @@ const GROUP_TEST_SHAPE: Record<string, Omit<GroupTestResult, 'group' | 'tested_u
|
||||
ok: true,
|
||||
error: '',
|
||||
},
|
||||
// The chain — Selected is the node the chain's exit group (auto) picked.
|
||||
relay: {
|
||||
selected: 'nl-reality-2',
|
||||
delay_ms: 61,
|
||||
exit_ip: '185.12.34.56',
|
||||
exit_country: 'NL',
|
||||
ok: true,
|
||||
error: '',
|
||||
},
|
||||
// Dead through its tunnel, exactly as its membership health says — the exit
|
||||
// test and the member health tell the same story about the same group.
|
||||
'via-tunnel': {
|
||||
@@ -1251,9 +1184,13 @@ let groupTestPinned = false
|
||||
export async function postGroupsTest(name = ''): Promise<GroupTestStart> {
|
||||
await wait(60)
|
||||
if (groupTest.running) return { started: false, reason: 'already running' }
|
||||
const all = (CONFIG.Groups ?? []).map((g) => g.Name)
|
||||
// Empty name = every group AND every chain, exactly like the daemon.
|
||||
const all = [
|
||||
...(CONFIG.Groups ?? []).map((g) => g.Name),
|
||||
...(CONFIG.Chains ?? []).map((c) => c.Name),
|
||||
]
|
||||
const targets = name ? all.filter((g) => g === name) : all
|
||||
if (targets.length === 0) return { started: false, reason: `no group named “${name}”` }
|
||||
if (targets.length === 0) return { started: false, reason: `no group or chain named “${name}”` }
|
||||
groupTestQueue = [...targets]
|
||||
groupTestPinned = false
|
||||
groupTest = {
|
||||
@@ -1302,7 +1239,10 @@ export async function getGroupsTest(): Promise<GroupTestStatus> {
|
||||
if (typeof location !== 'undefined') {
|
||||
const want = new URLSearchParams(location.search).get('grouptest')
|
||||
if (want) {
|
||||
const all = (CONFIG.Groups ?? []).map((g) => g.Name)
|
||||
const all = [
|
||||
...(CONFIG.Groups ?? []).map((g) => g.Name),
|
||||
...(CONFIG.Chains ?? []).map((c) => c.Name),
|
||||
]
|
||||
const targets = want === 'all' || want === '1' ? all : all.filter((g) => g === want)
|
||||
if (targets.length > 0) {
|
||||
groupTestQueue = [...targets]
|
||||
|
||||
@@ -200,6 +200,16 @@
|
||||
.dev-ip {
|
||||
color: var(--dim);
|
||||
}
|
||||
/* secondary addresses of a merged (multi-IP) device — quiet chips beside the primary */
|
||||
.dev-ip-extra {
|
||||
padding: 1px 6px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 5px;
|
||||
background: color-mix(in srgb, var(--sink) 60%, transparent);
|
||||
color: var(--faint);
|
||||
font-size: 10.5px;
|
||||
cursor: help;
|
||||
}
|
||||
.dev-mac {
|
||||
color: var(--faint);
|
||||
cursor: help;
|
||||
|
||||
@@ -61,7 +61,8 @@ const STATE_LED: Record<string, LedVariant> = {
|
||||
/** A discovered client merged with its config policy (if any). */
|
||||
interface DeviceRow {
|
||||
key: string
|
||||
ip: string
|
||||
ip: string // primary address (most recent lease)
|
||||
ips: string[] // every known address, primary first; [] only for a config-only row without IP
|
||||
mac: string
|
||||
hostname: string
|
||||
state: string // online | idle | offline
|
||||
@@ -186,6 +187,7 @@ export default function Devices() {
|
||||
out.push({
|
||||
key: d.mac || d.ip || d.hostname,
|
||||
ip: d.ip,
|
||||
ips: d.ips?.length ? d.ips : d.ip ? [d.ip] : [],
|
||||
mac: d.mac,
|
||||
hostname: d.hostname,
|
||||
state: d.state || (d.online ? 'online' : 'offline'),
|
||||
@@ -202,6 +204,7 @@ export default function Devices() {
|
||||
out.push({
|
||||
key: d.MAC || d.IP || d.Name,
|
||||
ip: d.IP ?? '',
|
||||
ips: d.IP ? [d.IP] : [],
|
||||
mac: d.MAC ?? '',
|
||||
hostname: d.Name,
|
||||
state: 'offline',
|
||||
@@ -515,6 +518,11 @@ function DeviceCard({
|
||||
</div>
|
||||
<div className="dev-id-l2 mono">
|
||||
<span className="dev-ip">{row.ip || '—'}</span>
|
||||
{row.ips.slice(1).map((ip) => (
|
||||
<span key={ip} className="dev-ip-extra" title="Additional address of this device">
|
||||
{ip}
|
||||
</span>
|
||||
))}
|
||||
<span className="dev-mac" title="Hardware (MAC) address">
|
||||
{row.mac || 'no mac'}
|
||||
</span>
|
||||
|
||||
@@ -797,23 +797,6 @@
|
||||
.ins-conn--dns {
|
||||
grid-template-columns: 62px minmax(60px, 116px) 12px minmax(0, 1fr) minmax(72px, 160px) auto;
|
||||
}
|
||||
/* router-originated lookups (urltest probes / sub fetches): a dimmed pill so they
|
||||
* read distinctly from a real LAN client's IP/hostname. */
|
||||
.ins-conn-dev--router {
|
||||
justify-self: start;
|
||||
max-width: 100%;
|
||||
padding: 1px 8px;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.03em;
|
||||
color: var(--faint);
|
||||
background: color-mix(in srgb, var(--raised) 70%, transparent);
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 999px;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
}
|
||||
.ins-conn--dns .tag {
|
||||
justify-self: end;
|
||||
}
|
||||
|
||||
@@ -177,26 +177,19 @@ function ConnRow({ c, fresh }: { c: ConnLogEntry; fresh: boolean }) {
|
||||
* (block/proxy/pass). Mirrors ConnRow's `device → dest` reading (same `.ins-conn`
|
||||
* chrome + accent arrow) so the DNS log and the Connections log read identically:
|
||||
* the "who asked" is the LAN device, the resolver that answered is the trailing
|
||||
* secondary column. Router-originated lookups (urltest probes / sub fetches) carry
|
||||
* the literal device `router` and render as a dimmed chip so they're distinct from
|
||||
* real LAN clients; an empty device (older data) falls back to a dash.
|
||||
* secondary column. The backend drops the router's own resolutions (urltest probes /
|
||||
* sub fetches) at ingestion, so every row here is a real client; an empty device
|
||||
* (older data) falls back to a dash.
|
||||
*/
|
||||
function DnsRow({ r, fresh }: { r: QueryLogEntry; fresh: boolean }) {
|
||||
const tag = actionTag(r.action)
|
||||
const dev = r.device.trim()
|
||||
const isRouter = dev.toLowerCase() === 'router'
|
||||
return (
|
||||
<li className={fresh ? 'ins-conn ins-conn--dns new' : 'ins-conn ins-conn--dns'}>
|
||||
<span className="ins-conn-t mono">{fmtClock(r.unix)}</span>
|
||||
{isRouter ? (
|
||||
<span className="ins-conn-dev ins-conn-dev--router" title="router-originated lookup">
|
||||
router
|
||||
</span>
|
||||
) : (
|
||||
<span className="ins-conn-dev mono" title={dev || 'source device unknown'}>
|
||||
{dev || '—'}
|
||||
</span>
|
||||
)}
|
||||
<span className="ins-conn-dev mono" title={dev || 'source device unknown'}>
|
||||
{dev || '—'}
|
||||
</span>
|
||||
<span className="ins-conn-arrow" aria-hidden="true">
|
||||
→
|
||||
</span>
|
||||
@@ -1236,7 +1229,7 @@ export default function Insights() {
|
||||
{/* 8 · DNS log — the live DNS DECISIONS (time · device → domain · resolver ·
|
||||
block/proxy/pass). Same shell/scroll/actions as Connections above, and
|
||||
now the same device → target reading: the "who asked" is the LAN client
|
||||
(or `router` for the appliance's own lookups), the resolver trails. */}
|
||||
(router-originated lookups are dropped at ingestion), the resolver trails. */}
|
||||
<Section
|
||||
title="DNS log · decisions"
|
||||
led={dns.err ? 'crit' : dns.rows.length ? 'on' : 'off'}
|
||||
|
||||
+14
-44
@@ -389,55 +389,29 @@ select.fp-input {
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
/* protocol checkboxes */
|
||||
.opt-checks {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 8px 16px;
|
||||
}
|
||||
.opt-check {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
font-size: 12px;
|
||||
color: var(--dim);
|
||||
cursor: pointer;
|
||||
}
|
||||
.opt-check input {
|
||||
accent-color: var(--accent);
|
||||
cursor: pointer;
|
||||
}
|
||||
|
||||
/* chip list */
|
||||
.chip-field {
|
||||
/* extra-headers key/value rows */
|
||||
.hdr-field {
|
||||
gap: 8px;
|
||||
}
|
||||
.chip-row {
|
||||
.hdr-rows {
|
||||
list-style: none;
|
||||
margin: 0;
|
||||
padding: 0;
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
flex-direction: column;
|
||||
gap: 6px;
|
||||
}
|
||||
.chip {
|
||||
display: inline-flex;
|
||||
.hdr-row {
|
||||
display: grid;
|
||||
grid-template-columns: minmax(9rem, 1fr) 2fr auto;
|
||||
gap: 8px;
|
||||
align-items: center;
|
||||
gap: 6px;
|
||||
padding: 3px 4px 3px 9px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 6px;
|
||||
background: color-mix(in srgb, var(--raised) 70%, transparent);
|
||||
}
|
||||
.chip-text {
|
||||
font-size: 11px;
|
||||
color: var(--ink);
|
||||
max-width: 32ch;
|
||||
overflow: hidden;
|
||||
text-overflow: ellipsis;
|
||||
white-space: nowrap;
|
||||
.hdr-key {
|
||||
font-family: var(--font-mono);
|
||||
font-variant-numeric: tabular-nums;
|
||||
}
|
||||
.chip-x {
|
||||
.hdr-x {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
justify-content: center;
|
||||
@@ -453,18 +427,14 @@ select.fp-input {
|
||||
cursor: pointer;
|
||||
transition: color 0.15s, background 0.15s;
|
||||
}
|
||||
.chip-x:hover:not(:disabled) {
|
||||
.hdr-x:hover:not(:disabled) {
|
||||
color: var(--crit);
|
||||
background: color-mix(in srgb, var(--crit) 14%, transparent);
|
||||
}
|
||||
.chip-x:disabled {
|
||||
.hdr-x:disabled {
|
||||
opacity: 0.5;
|
||||
cursor: default;
|
||||
}
|
||||
.chip-add {
|
||||
display: flex;
|
||||
gap: 8px;
|
||||
}
|
||||
|
||||
/* editor footer */
|
||||
.opt-actions {
|
||||
|
||||
+133
-218
@@ -576,15 +576,40 @@ export default function Nodes() {
|
||||
)
|
||||
|
||||
// Commit an options edit for one subscription. The editor hands back a fully
|
||||
// patched Subscription (Name/Enabled/URL preserved); we splice it in and reuse
|
||||
// the same save→apply machinery as every other mutation.
|
||||
// patched Subscription (Enabled and every off-form field preserved); we splice
|
||||
// it in and reuse the same save→apply machinery as every other mutation.
|
||||
//
|
||||
// A rename cascades: FromSub is the ONLY link between a sub and its cached
|
||||
// nodes, so every node pointing at the old name is rewritten in the same save
|
||||
// (or the whole group would fall out of the sub's bucket and turn orphan). A
|
||||
// collision with another sub's name blocks the save outright — names are the
|
||||
// Subscriptions' primary key.
|
||||
const editSub = useCallback(
|
||||
(idx: number, patch: Subscription): Promise<boolean> => {
|
||||
if (!config) return Promise.resolve(false)
|
||||
const next = subs.map((s, i) => (i === idx ? patch : s))
|
||||
return save({ ...config, Subscriptions: next }, `Updated ${patch.Name}`)
|
||||
async (idx: number, patch: Subscription): Promise<boolean> => {
|
||||
if (!config) return false
|
||||
const oldName = subs[idx].Name
|
||||
const renamed = patch.Name !== oldName
|
||||
if (renamed && subs.some((s, i) => i !== idx && s.Name === patch.Name)) {
|
||||
flash(`A subscription named “${patch.Name}” already exists.`)
|
||||
return false
|
||||
}
|
||||
const next: Model = { ...config, Subscriptions: subs.map((s, i) => (i === idx ? patch : s)) }
|
||||
if (renamed) {
|
||||
next.Nodes = nodes.map((n) => (n.FromSub === oldName ? { ...n, FromSub: patch.Name } : n))
|
||||
}
|
||||
const ok = await save(next, `Updated ${patch.Name}`)
|
||||
// Carry the group's open/closed override to the new key only once the
|
||||
// rename actually stuck (openMap is keyed by FromSub).
|
||||
if (ok && renamed) {
|
||||
setOpenMap((prev) => {
|
||||
if (!(oldName in prev)) return prev
|
||||
const { [oldName]: was, ...rest } = prev
|
||||
return { ...rest, [patch.Name]: was }
|
||||
})
|
||||
}
|
||||
return ok
|
||||
},
|
||||
[config, subs, save],
|
||||
[config, subs, nodes, save, flash],
|
||||
)
|
||||
|
||||
// Fetch one subscription now (feedback #8): pull it (through the tunnel when
|
||||
@@ -1296,79 +1321,71 @@ const FETCH_VIA: { value: string; label: string }[] = [
|
||||
{ value: 'direct', label: 'Direct' },
|
||||
{ value: 'proxy', label: 'Proxy (through the tunnel)' },
|
||||
]
|
||||
const FORMATS = ['auto', 'clash', 'xray', 'singbox', 'links'] as const
|
||||
const PROTO_FILTERS = ['vless', 'vmess', 'trojan', 'ss'] as const
|
||||
|
||||
/** Flat, all-strings-and-bools shape the form binds to (arrays stay arrays). */
|
||||
/** One editable Key/Value row of the extra-headers list. */
|
||||
interface HeaderRow {
|
||||
key: string
|
||||
value: string
|
||||
}
|
||||
|
||||
/** Split a stored raw "Key: value" header into an editable row on the FIRST
|
||||
* colon. A malformed entry (no colon at all) loads as a key-only row rather
|
||||
* than being silently dropped — the user can fix or delete it. */
|
||||
function toHeaderRow(raw: string): HeaderRow {
|
||||
const i = raw.indexOf(':')
|
||||
if (i < 0) return { key: raw.trim(), value: '' }
|
||||
return { key: raw.slice(0, i).trim(), value: raw.slice(i + 1).trim() }
|
||||
}
|
||||
|
||||
/** Flat, all-strings shape the form binds to (headers stay rows). */
|
||||
interface SubDraft {
|
||||
Name: string
|
||||
URL: string
|
||||
UpdateInterval: string
|
||||
FetchVia: string
|
||||
FetchDetour: string // canonical: direct | group:<n> | node:<n> | egress:<n>
|
||||
Format: string
|
||||
UA: string
|
||||
HWID: string
|
||||
DeviceOS: string
|
||||
VerOS: string
|
||||
DeviceModel: string
|
||||
Headers: string[]
|
||||
Include: string[]
|
||||
Exclude: string[]
|
||||
FilterProto: string[]
|
||||
FilterCountry: string // comma-separated ISO codes; leading "!" excludes
|
||||
Dedup: boolean
|
||||
ExpireAlertDays: string
|
||||
Headers: HeaderRow[]
|
||||
}
|
||||
|
||||
function toDraft(s: Subscription, cat: DetourCatalog): SubDraft {
|
||||
return {
|
||||
Name: s.Name,
|
||||
URL: s.URL,
|
||||
UpdateInterval: s.UpdateInterval ?? '',
|
||||
FetchVia: s.FetchVia === 'proxy' ? 'proxy' : 'direct',
|
||||
FetchDetour: canonDetour(s.FetchDetour, cat),
|
||||
Format: s.Format && s.Format.trim() ? s.Format : 'auto',
|
||||
UA: s.UA ?? '',
|
||||
HWID: s.HWID ?? '',
|
||||
DeviceOS: s.DeviceOS ?? '',
|
||||
VerOS: s.VerOS ?? '',
|
||||
DeviceModel: s.DeviceModel ?? '',
|
||||
Headers: asArray(s.Headers),
|
||||
Include: asArray(s.Include),
|
||||
Exclude: asArray(s.Exclude),
|
||||
FilterProto: asArray(s.FilterProto),
|
||||
FilterCountry: asArray(s.FilterCountry).join(', '),
|
||||
Dedup: s.Dedup ?? false,
|
||||
ExpireAlertDays: s.ExpireAlertDays != null ? String(s.ExpireAlertDays) : '',
|
||||
Headers: asArray(s.Headers).map(toHeaderRow),
|
||||
}
|
||||
}
|
||||
|
||||
/** Fold a draft back onto the base sub — empties collapse to undefined so the
|
||||
* JSON stays lean, and Name/Enabled/URL (incl. the secret token) ride untouched. */
|
||||
* JSON stays lean. Spreading `base` first keeps every field the form no longer
|
||||
* shows (Format, device identity, filters, dedup, expire alert …) exactly as
|
||||
* stored; a blanked Name or URL falls back to the old value instead of wiping
|
||||
* the sub. */
|
||||
function applyDraft(base: Subscription, d: SubDraft): Subscription {
|
||||
const s = (v: string): string | undefined => (v.trim() ? v.trim() : undefined)
|
||||
const list = (a: string[]): string[] | undefined => (a.length ? a : undefined)
|
||||
const countries = d.FilterCountry.split(',')
|
||||
.map((c) => c.trim())
|
||||
.filter(Boolean)
|
||||
const days = Number.parseInt(d.ExpireAlertDays, 10)
|
||||
const proxy = d.FetchVia === 'proxy'
|
||||
// Rows with an empty key are dropped; the rest serialize back to the model's
|
||||
// raw "Key: value" strings, single space after the colon.
|
||||
const headers = d.Headers.filter((h) => h.key.trim()).map(
|
||||
(h) => `${h.key.trim()}: ${h.value.trim()}`,
|
||||
)
|
||||
return {
|
||||
...base,
|
||||
Name: d.Name.trim() || base.Name,
|
||||
URL: d.URL.trim() || base.URL,
|
||||
UpdateInterval: s(d.UpdateInterval),
|
||||
FetchVia: proxy ? 'proxy' : undefined,
|
||||
// Detour only rides along when fetching via proxy and it isn't plain Direct.
|
||||
FetchDetour: proxy && d.FetchDetour !== 'direct' ? d.FetchDetour : undefined,
|
||||
Format: d.Format && d.Format !== 'auto' ? d.Format : undefined,
|
||||
UA: s(d.UA),
|
||||
HWID: s(d.HWID),
|
||||
DeviceOS: s(d.DeviceOS),
|
||||
VerOS: s(d.VerOS),
|
||||
DeviceModel: s(d.DeviceModel),
|
||||
Headers: list(d.Headers),
|
||||
Include: list(d.Include),
|
||||
Exclude: list(d.Exclude),
|
||||
FilterProto: list(d.FilterProto),
|
||||
FilterCountry: countries.length ? countries : undefined,
|
||||
Dedup: d.Dedup ? true : undefined,
|
||||
ExpireAlertDays: Number.isFinite(days) && days > 0 ? days : undefined,
|
||||
Headers: headers.length ? headers : undefined,
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1419,19 +1436,33 @@ function SubOptions({
|
||||
|
||||
const proxy = draft.FetchVia === 'proxy'
|
||||
const disabled = busy || saving
|
||||
const toggleProto = (p: string) =>
|
||||
set(
|
||||
'FilterProto',
|
||||
draft.FilterProto.includes(p)
|
||||
? draft.FilterProto.filter((x) => x !== p)
|
||||
: [...draft.FilterProto, p],
|
||||
)
|
||||
|
||||
return (
|
||||
<div id={id} className="sub-options">
|
||||
<fieldset className="opt-group" disabled={disabled}>
|
||||
<legend className="opt-legend">Fetch</legend>
|
||||
<div className="opt-grid">
|
||||
<OptField label="Name" hint="Renaming keeps the sub’s cached nodes attached">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.Name}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('Name', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="URL" wide hint="The feed address — a token in it stays stored, only masked in the list">
|
||||
<input
|
||||
className="fp-input"
|
||||
type="text"
|
||||
inputMode="url"
|
||||
value={draft.URL}
|
||||
placeholder="https://provider.example/sub?token=…"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('URL', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Update interval" hint="e.g. 30m · 6h · 24h — blank means manual only">
|
||||
<input
|
||||
className="fp-input"
|
||||
@@ -1467,22 +1498,6 @@ function SubOptions({
|
||||
/>
|
||||
</OptField>
|
||||
)}
|
||||
<OptField
|
||||
label="Format"
|
||||
hint="Auto reads whatever the feed turns out to be. Pick one to force that parser when auto guesses wrong."
|
||||
>
|
||||
<select
|
||||
className="fp-input"
|
||||
value={draft.Format}
|
||||
onChange={(e) => set('Format', e.target.value)}
|
||||
>
|
||||
{FORMATS.map((f) => (
|
||||
<option key={f} value={f}>
|
||||
{f}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
</OptField>
|
||||
</div>
|
||||
</fieldset>
|
||||
|
||||
@@ -1513,110 +1528,8 @@ function SubOptions({
|
||||
onChange={(e) => set('HWID', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Device OS">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.DeviceOS}
|
||||
placeholder="android"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('DeviceOS', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Ver OS">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.VerOS}
|
||||
placeholder="14"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('VerOS', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Device model">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.DeviceModel}
|
||||
placeholder="Pixel 8"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('DeviceModel', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
</div>
|
||||
<ChipList
|
||||
label="Extra headers"
|
||||
hint="One Key: value per chip — sent verbatim with the fetch"
|
||||
placeholder="X-Api-Key: …"
|
||||
value={draft.Headers}
|
||||
disabled={disabled}
|
||||
onChange={(v) => set('Headers', v)}
|
||||
/>
|
||||
</fieldset>
|
||||
|
||||
<fieldset className="opt-group" disabled={disabled}>
|
||||
<legend className="opt-legend">Filters</legend>
|
||||
<ChipList
|
||||
label="Include"
|
||||
hint="Keep only nodes whose name matches one of these regexes"
|
||||
placeholder="^🇩🇪|premium"
|
||||
value={draft.Include}
|
||||
disabled={disabled}
|
||||
onChange={(v) => set('Include', v)}
|
||||
/>
|
||||
<ChipList
|
||||
label="Exclude"
|
||||
hint="Drop nodes whose name matches — applied after include"
|
||||
placeholder="expire|traffic"
|
||||
value={draft.Exclude}
|
||||
disabled={disabled}
|
||||
onChange={(v) => set('Exclude', v)}
|
||||
/>
|
||||
<div className="opt-grid">
|
||||
<OptField label="Protocols" wide hint="Keep only these link types (none = keep all)">
|
||||
<div className="opt-checks">
|
||||
{PROTO_FILTERS.map((p) => (
|
||||
<label key={p} className="opt-check">
|
||||
<input
|
||||
type="checkbox"
|
||||
checked={draft.FilterProto.includes(p)}
|
||||
onChange={() => toggleProto(p)}
|
||||
/>
|
||||
<span className="mono">{p}</span>
|
||||
</label>
|
||||
))}
|
||||
</div>
|
||||
</OptField>
|
||||
<OptField label="Countries" hint="ISO codes, comma-separated; prefix ! to exclude">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={draft.FilterCountry}
|
||||
placeholder="de, nl, !ru"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
onChange={(e) => set('FilterCountry', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Expire-alert days" hint="Warn this many days before the sub expires">
|
||||
<input
|
||||
className="fp-input"
|
||||
type="number"
|
||||
inputMode="numeric"
|
||||
min={0}
|
||||
value={draft.ExpireAlertDays}
|
||||
placeholder="0"
|
||||
onChange={(e) => set('ExpireAlertDays', e.target.value)}
|
||||
/>
|
||||
</OptField>
|
||||
<OptField label="Dedup" hint="Collapse identical endpoints pulled more than once">
|
||||
<Toggle
|
||||
pressed={draft.Dedup}
|
||||
onChange={(v) => set('Dedup', v)}
|
||||
label={`${draft.Dedup ? 'Disable' : 'Enable'} dedup for ${sub.Name}`}
|
||||
disabled={disabled}
|
||||
/>
|
||||
</OptField>
|
||||
</div>
|
||||
<HeaderRows value={draft.Headers} disabled={disabled} onChange={(v) => set('Headers', v)} />
|
||||
</fieldset>
|
||||
|
||||
<div className="opt-actions">
|
||||
@@ -1666,42 +1579,54 @@ function OptField({
|
||||
)
|
||||
}
|
||||
|
||||
function ChipList({
|
||||
label,
|
||||
hint,
|
||||
placeholder,
|
||||
/** Key/Value row editor for the extra request headers. Rows map 1:1 onto the
|
||||
* model's raw "Key: value" strings (toHeaderRow / applyDraft do the split and
|
||||
* join); a row left with an empty key is dropped on save rather than
|
||||
* serialising a nameless header. */
|
||||
function HeaderRows({
|
||||
value,
|
||||
disabled,
|
||||
onChange,
|
||||
}: {
|
||||
label: string
|
||||
hint?: string
|
||||
placeholder: string
|
||||
value: string[]
|
||||
value: HeaderRow[]
|
||||
disabled?: boolean
|
||||
onChange: (next: string[]) => void
|
||||
onChange: (next: HeaderRow[]) => void
|
||||
}) {
|
||||
const [text, setText] = useState('')
|
||||
const add = () => {
|
||||
const v = text.trim()
|
||||
if (!v) return
|
||||
if (!value.includes(v)) onChange([...value, v])
|
||||
setText('')
|
||||
}
|
||||
const setRow = (i: number, patch: Partial<HeaderRow>) =>
|
||||
onChange(value.map((h, k) => (k === i ? { ...h, ...patch } : h)))
|
||||
return (
|
||||
<div className="opt-field opt-field--wide chip-field">
|
||||
<span className="opt-label mono">{label}</span>
|
||||
<div className="opt-field opt-field--wide hdr-field">
|
||||
<span className="opt-label mono">Extra headers</span>
|
||||
{value.length > 0 && (
|
||||
<ul className="chip-row">
|
||||
{value.map((v, i) => (
|
||||
<li key={`${v}-${i}`} className="chip">
|
||||
<span className="chip-text mono">{v}</span>
|
||||
<ul className="hdr-rows">
|
||||
{value.map((h, i) => (
|
||||
<li key={i} className="hdr-row">
|
||||
<input
|
||||
className="fp-input hdr-key"
|
||||
value={h.key}
|
||||
placeholder="X-Api-Key"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
aria-label={`Header ${i + 1} name`}
|
||||
disabled={disabled}
|
||||
onChange={(e) => setRow(i, { key: e.target.value })}
|
||||
/>
|
||||
<input
|
||||
className="fp-input"
|
||||
value={h.value}
|
||||
placeholder="value"
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
aria-label={`Header ${i + 1} value`}
|
||||
disabled={disabled}
|
||||
onChange={(e) => setRow(i, { value: e.target.value })}
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
className="chip-x"
|
||||
className="hdr-x"
|
||||
onClick={() => onChange(value.filter((_, k) => k !== i))}
|
||||
disabled={disabled}
|
||||
aria-label={`Remove ${v}`}
|
||||
aria-label={`Remove header ${h.key.trim() || i + 1}`}
|
||||
>
|
||||
×
|
||||
</button>
|
||||
@@ -1709,28 +1634,18 @@ function ChipList({
|
||||
))}
|
||||
</ul>
|
||||
)}
|
||||
<div className="chip-add">
|
||||
<input
|
||||
className="fp-input"
|
||||
value={text}
|
||||
placeholder={placeholder}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
aria-label={`Add ${label}`}
|
||||
<div>
|
||||
<Button
|
||||
className="row-del"
|
||||
onClick={() => onChange([...value, { key: '', value: '' }])}
|
||||
disabled={disabled}
|
||||
onChange={(e) => setText(e.target.value)}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === 'Enter') {
|
||||
e.preventDefault()
|
||||
add()
|
||||
}
|
||||
}}
|
||||
/>
|
||||
<Button className="row-del" onClick={add} disabled={disabled || !text.trim()}>
|
||||
Add
|
||||
>
|
||||
Add header
|
||||
</Button>
|
||||
</div>
|
||||
{hint && <span className="opt-hint">{hint}</span>}
|
||||
<span className="opt-hint">
|
||||
Sent verbatim with the fetch — a row without a name is dropped on save
|
||||
</span>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -1,14 +1,13 @@
|
||||
import { useCallback, useEffect, useRef, useState } from 'react'
|
||||
import { Button, Led, Module, QueryLog, SegMeter } from '../components'
|
||||
import type { LedVariant, QueryEntry, QueryTag } from '../components'
|
||||
import { fmtClock, fmtDateTime, fmtDuration } from '../format'
|
||||
import { Button, Led, Module } from '../components'
|
||||
import type { LedVariant } from '../components'
|
||||
import { fmtDateTime, fmtDuration } from '../format'
|
||||
import {
|
||||
apply as apiApply,
|
||||
confirm as apiConfirm,
|
||||
rollback as apiRollback,
|
||||
getConfig,
|
||||
getStats,
|
||||
getStatsLog,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
import type { Model, Stats, Status, StatusWarning } from '../api'
|
||||
@@ -28,12 +27,6 @@ function short(hash: string): string {
|
||||
return h.length > 12 ? h.slice(0, 12) : h
|
||||
}
|
||||
|
||||
function actionTag(action?: string): QueryTag {
|
||||
if (action === 'block') return 'block'
|
||||
if (action === 'proxy') return 'proxy'
|
||||
return 'pass'
|
||||
}
|
||||
|
||||
type ControlKind = 'apply' | 'confirm' | 'rollback'
|
||||
|
||||
/**
|
||||
@@ -100,50 +93,19 @@ export function Overview({
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// ---- live query log + filter stats: poll both, degrade to honest empty states ----
|
||||
const [entries, setEntries] = useState<QueryEntry[]>([])
|
||||
// ---- live filter stats: poll the aggregate snapshot, degrade to honest empty states ----
|
||||
const [stats, setStats] = useState<Stats | null>(null)
|
||||
const [statsOk, setStatsOk] = useState(true)
|
||||
// `loggingOff` is read back out of the snapshot each tick so the poll can stop
|
||||
// asking for a log the daemon isn't keeping. A ref (not state) keeps the effect
|
||||
// stable — it must not re-subscribe every time the snapshot lands.
|
||||
const loggingOffRef = useRef(false)
|
||||
useEffect(() => {
|
||||
let alive = true
|
||||
const tick = async () => {
|
||||
// Nothing to poll for while the tab is in the background.
|
||||
if (document.hidden) return
|
||||
try {
|
||||
// Logging off ⇒ the log endpoint has nothing to give; skip it and keep
|
||||
// polling the snapshot alone so the page notices it being switched back on.
|
||||
const [log, s] = await Promise.all([
|
||||
loggingOffRef.current ? Promise.resolve([]) : getStatsLog(50),
|
||||
getStats(),
|
||||
])
|
||||
const s = await getStats()
|
||||
if (!alive) return
|
||||
loggingOffRef.current = s.backend === 'off'
|
||||
setStatsOk(true)
|
||||
setStats(s)
|
||||
const rows = Array.isArray(log) ? log : []
|
||||
setEntries(
|
||||
rows.slice(0, 10).map((r) => ({
|
||||
// The resolution unix-time + domain is a stable identity for the row,
|
||||
// so React only replays the slide-in for genuinely new queries.
|
||||
id: `${r.unix}-${r.domain}-${r.qtype}`,
|
||||
// Format from unix in the browser's local timezone (shared helper) so
|
||||
// this log matches the Insights logs — not the server's UTC `time`.
|
||||
time: fmtClock(r.unix) || (r.time ?? ''),
|
||||
domain: r.domain,
|
||||
// The REAL query source, same as the Insights DNS log: the LAN
|
||||
// client's hostname/IP, or 'router' for the appliance's own lookups.
|
||||
// (The resolver lives in the Insights log's own column.) '' only on
|
||||
// rows persisted before device attribution existed.
|
||||
device: r.device || '—',
|
||||
tag: actionTag(r.action),
|
||||
})),
|
||||
)
|
||||
} catch {
|
||||
if (alive) setStatsOk(false)
|
||||
// Transient — the interval retries, and the modules keep their last reading.
|
||||
}
|
||||
}
|
||||
void tick()
|
||||
@@ -463,64 +425,6 @@ export function Overview({
|
||||
/>
|
||||
</div>
|
||||
|
||||
<div className="log-wrap">
|
||||
{hasStats && (
|
||||
<div className="filter-readout">
|
||||
<SegMeter label="Filtered" value={blockedPct} max={100} unit="%" />
|
||||
{topBlocked.length > 0 ? (
|
||||
<ul className="topblocked" aria-label="Top blocked domains">
|
||||
{topBlocked.map((d) => (
|
||||
<li key={d.domain}>
|
||||
<span className="tb-dom">{d.domain}</span>
|
||||
<span className="tb-n">{d.blocked}</span>
|
||||
</li>
|
||||
))}
|
||||
</ul>
|
||||
) : (
|
||||
<p className="qempty" style={{ margin: 0 }}>
|
||||
{blocked > 0 ? 'blocked queries logged' : 'nothing blocked yet'}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
<QueryLog
|
||||
title="Query log"
|
||||
led={!statsOk ? 'crit' : loggingOff ? 'off' : entries.length > 0 ? 'on' : 'amber'}
|
||||
hint={
|
||||
!statsOk
|
||||
? 'stats unavailable'
|
||||
: loggingOff
|
||||
? 'logging off'
|
||||
: hasStats
|
||||
? `${queries} queries · ${blocked} blocked`
|
||||
: // Aggregates can reset on a daemon restart while the persistent
|
||||
// log still streams rows — don't claim "waiting" when the log is
|
||||
// clearly live; only show it when there's genuinely nothing.
|
||||
entries.length > 0
|
||||
? 'recent queries'
|
||||
: 'waiting for engine stats'
|
||||
}
|
||||
entries={entries}
|
||||
/>
|
||||
{entries.length === 0 && (
|
||||
<p className="qempty">
|
||||
{!statsOk ? (
|
||||
'Stats endpoint unreachable — retrying every few seconds.'
|
||||
) : loggingOff ? (
|
||||
<>
|
||||
Logging is off — no DNS, connection, or per-device history is recorded.{' '}
|
||||
<a className="linkish" href="#/settings">
|
||||
Turn it on in Settings → Logging backend
|
||||
</a>
|
||||
.
|
||||
</>
|
||||
) : (
|
||||
'No query data yet — the live stream lights up once the engine resolves DNS (needs an active subscriber + traffic).'
|
||||
)}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* Apply / Confirm / Rollback — active voice, honest results. */}
|
||||
<div className="controls" role="group" aria-label="Config actions">
|
||||
<div className="controls-btns">
|
||||
|
||||
@@ -142,9 +142,6 @@
|
||||
width: 7rem;
|
||||
flex: none;
|
||||
}
|
||||
.pf-field--time {
|
||||
width: auto;
|
||||
}
|
||||
.pf-flabel {
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
@@ -315,9 +312,6 @@
|
||||
color: var(--faint);
|
||||
font-style: italic;
|
||||
}
|
||||
.pf-chip--target {
|
||||
border-color: color-mix(in srgb, var(--accent) 30%, var(--groove));
|
||||
}
|
||||
.pf-chip--on {
|
||||
border-color: color-mix(in srgb, var(--led-on) 45%, var(--groove));
|
||||
color: var(--led-on);
|
||||
@@ -432,9 +426,6 @@
|
||||
.pf-erow-ctl {
|
||||
min-width: 0;
|
||||
}
|
||||
/* (The .pf-erow--muted dimming that used to sit here is gone with its reason:
|
||||
the watcher DOES evaluate an iface-driven profile's schedule now, so the
|
||||
schedule fields are never inert.) */
|
||||
.pf-ehint {
|
||||
margin: 6px 0 0;
|
||||
font-family: var(--font-sans);
|
||||
@@ -477,63 +468,6 @@
|
||||
outline-offset: 1px;
|
||||
border-radius: 2px;
|
||||
}
|
||||
.pf-chipinput-in {
|
||||
flex: 1 1 8rem;
|
||||
border: 0;
|
||||
background: none;
|
||||
box-shadow: none;
|
||||
padding: 3px 4px;
|
||||
}
|
||||
.pf-chipinput-in:focus-visible {
|
||||
outline: none;
|
||||
}
|
||||
.pf-chipinput:focus-within {
|
||||
border-color: var(--accent);
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 1px;
|
||||
}
|
||||
|
||||
/* ---- schedule ---- */
|
||||
.pf-days {
|
||||
display: inline-flex;
|
||||
flex-wrap: wrap;
|
||||
gap: 4px;
|
||||
}
|
||||
.pf-day {
|
||||
padding: 6px 9px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 6px;
|
||||
background: var(--sink);
|
||||
color: var(--dim);
|
||||
font-family: var(--font-mono);
|
||||
font-size: 11px;
|
||||
letter-spacing: 0.02em;
|
||||
cursor: pointer;
|
||||
transition: color 0.15s, border-color 0.15s, background 0.15s;
|
||||
}
|
||||
.pf-day:hover:not(:disabled) {
|
||||
color: var(--ink);
|
||||
}
|
||||
.pf-day.on {
|
||||
color: var(--accent);
|
||||
border-color: color-mix(in srgb, var(--accent) 55%, var(--groove));
|
||||
background: color-mix(in srgb, var(--accent) 12%, transparent);
|
||||
}
|
||||
.pf-day:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: 2px;
|
||||
}
|
||||
.pf-day:disabled {
|
||||
opacity: 0.55;
|
||||
cursor: default;
|
||||
}
|
||||
.pf-times {
|
||||
display: flex;
|
||||
flex-wrap: wrap;
|
||||
align-items: flex-end;
|
||||
gap: 12px;
|
||||
margin-top: 10px;
|
||||
}
|
||||
|
||||
/* ---- rule on/off matrix ---- */
|
||||
.pf-rules {
|
||||
@@ -594,56 +528,6 @@
|
||||
cursor: default;
|
||||
}
|
||||
|
||||
/* ---- preset packs ---- */
|
||||
.pf-pack-desc {
|
||||
margin: 6px 0 0;
|
||||
font-family: var(--font-sans);
|
||||
font-size: 12px;
|
||||
line-height: 1.5;
|
||||
color: var(--dim);
|
||||
max-width: 60ch;
|
||||
}
|
||||
.pf-pack-target {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
align-items: flex-end;
|
||||
gap: 6px;
|
||||
}
|
||||
.pf-adv {
|
||||
border-top: 1px solid var(--groove);
|
||||
}
|
||||
.pf-adv-summary {
|
||||
padding: 9px 14px;
|
||||
cursor: pointer;
|
||||
list-style: none;
|
||||
font-family: var(--font-mono);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: var(--track-label, 0.16em);
|
||||
text-transform: uppercase;
|
||||
color: var(--faint);
|
||||
}
|
||||
.pf-adv-summary::-webkit-details-marker {
|
||||
display: none;
|
||||
}
|
||||
.pf-adv-summary::before {
|
||||
content: '▸ ';
|
||||
color: var(--faint);
|
||||
}
|
||||
.pf-adv[open] .pf-adv-summary::before {
|
||||
content: '▾ ';
|
||||
}
|
||||
.pf-adv-summary:focus-visible {
|
||||
outline: 2px solid var(--accent);
|
||||
outline-offset: -2px;
|
||||
border-radius: 4px;
|
||||
}
|
||||
.pf-adv-body {
|
||||
display: flex;
|
||||
flex-direction: column;
|
||||
gap: 2px;
|
||||
padding: 0 14px 14px;
|
||||
}
|
||||
|
||||
/* ---- empty + skeleton ---- */
|
||||
.pf-empty {
|
||||
margin-top: calc(var(--u, 8px) * 2);
|
||||
@@ -706,10 +590,6 @@
|
||||
grid-column: 1 / -1;
|
||||
justify-content: flex-end;
|
||||
}
|
||||
.pf-pack-target {
|
||||
grid-column: 1 / -1;
|
||||
align-items: stretch;
|
||||
}
|
||||
}
|
||||
|
||||
@media (prefers-reduced-motion: reduce) {
|
||||
@@ -717,7 +597,6 @@
|
||||
animation: none;
|
||||
}
|
||||
.pf-input,
|
||||
.pf-day,
|
||||
.pf-edit,
|
||||
.pf-del {
|
||||
transition: none;
|
||||
|
||||
+101
-535
@@ -1,19 +1,16 @@
|
||||
import './Profiles.css'
|
||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react'
|
||||
import { Button, Led, Toggle } from '../components'
|
||||
import { apply as apiApply, getConfig, putConfig, ApiError } from '../api'
|
||||
import type { Model, Preset, Profile } from '../api'
|
||||
import { apply as apiApply, getConfig, getInterfaces, putConfig, ApiError } from '../api'
|
||||
import type { Interface, Model, Profile } from '../api'
|
||||
|
||||
// The Profiles page is a thin editor over the desired-state Model — the same
|
||||
// save→apply split as Settings / DNS / Routing. Every edit rewrites a slice in
|
||||
// place, PUTs the whole Model (save), marks the config dirty, and only Apply
|
||||
// pushes it onto the live data plane. Two sections:
|
||||
//
|
||||
// 1. PROFILES — named WAN-mode / failover overrides that activate by
|
||||
// condition (uplink iface, schedule). Manual override lives in
|
||||
// Globals.ActiveProfile ("" = auto, evaluated on the router).
|
||||
// 2. PRESET PACKS — three built-in curated rule bundles (block-ads / ru-bypass
|
||||
// / private). Toggling one upserts the matching entry in Model.Presets.
|
||||
// pushes it onto the live data plane. One section: PROFILES — named WAN-mode /
|
||||
// failover overrides that activate when the router's default-route uplink
|
||||
// matches. Manual override lives in Globals.ActiveProfile ("" = auto,
|
||||
// evaluated on the router).
|
||||
|
||||
// ---- helpers ---------------------------------------------------------------
|
||||
|
||||
@@ -27,59 +24,7 @@ const asArray = <T,>(a: T[] | null | undefined): T[] => (a ? a : [])
|
||||
const ruleSummary = (names: string[], max = 2): string =>
|
||||
names.length <= max ? names.join(', ') : `${names.slice(0, max).join(', ')} +${names.length - max} more`
|
||||
|
||||
// Weekday tokens as stored on the model (mon..sun), in display order.
|
||||
const DAYS = ['mon', 'tue', 'wed', 'thu', 'fri', 'sat', 'sun'] as const
|
||||
const cap = (s: string): string => (s ? s[0].toUpperCase() + s.slice(1) : s)
|
||||
|
||||
// ---- schedule timezone anchoring --------------------------------------------
|
||||
// The router carries no IANA tzdata, so the daemon cannot resolve a timezone
|
||||
// NAME — it evaluates schedule windows at a fixed UTC offset (minutes east,
|
||||
// model SchedUTCOffset). The panel's job is to capture the editing browser's
|
||||
// offset alongside every schedule edit, so "22:00" means 22:00 on the clock of
|
||||
// whoever wrote it. Known limitation (stated in the hint): a fixed offset does
|
||||
// not follow DST until the schedule is re-saved.
|
||||
|
||||
/** The browser's current UTC offset in minutes EAST (Moscow ⇒ 180). */
|
||||
const browserUTCOffsetMin = (): number => -new Date().getTimezoneOffset()
|
||||
|
||||
/** "UTC+03:00" / "UTC−05:00" for a minutes-east offset. */
|
||||
const utcOffsetLabel = (min: number): string => {
|
||||
const sign = min < 0 ? '−' : '+'
|
||||
const a = Math.abs(min)
|
||||
const hh = String(Math.floor(a / 60)).padStart(2, '0')
|
||||
const mm = String(a % 60).padStart(2, '0')
|
||||
return `UTC${sign}${hh}:${mm}`
|
||||
}
|
||||
|
||||
/** IANA-style name of the browser zone ("Europe/Moscow"), best-effort. */
|
||||
const browserTZName = (): string => {
|
||||
try {
|
||||
return Intl.DateTimeFormat().resolvedOptions().timeZone || 'local time'
|
||||
} catch {
|
||||
return 'local time'
|
||||
}
|
||||
}
|
||||
|
||||
// The three built-in packs. Fixed set — the operator toggles them, never adds or
|
||||
// removes. Name is the on-disk `config preset` name the engine expands.
|
||||
const PACKS: ReadonlyArray<{ name: string; title: string; desc: string }> = [
|
||||
{ name: 'block-ads', title: 'Block ads', desc: 'Blocks known ad & tracker domains.' },
|
||||
{ name: 'ru-bypass', title: 'Russia bypass', desc: 'Routes Russian services direct (no proxy).' },
|
||||
{ name: 'private', title: 'Private ranges', desc: 'Keeps LAN / private ranges off the proxy.' },
|
||||
]
|
||||
|
||||
// ---- target picker ---------------------------------------------------------
|
||||
// A routing target in the model's canonical form: `direct` | `block` |
|
||||
// `group:<n>` | `chain:<n>` | `node:<n>` | `egress:<n>`. Built entirely from the
|
||||
// live Model, exactly like the Routing / DNS pickers.
|
||||
|
||||
interface TargetCatalog {
|
||||
groups: string[]
|
||||
chains: string[]
|
||||
nodes: string[]
|
||||
egresses: { name: string; type: string }[]
|
||||
}
|
||||
|
||||
// Pull Name strings out of a model slice (Rules / Resolvers) for the pickers.
|
||||
type Named = { Name?: unknown }
|
||||
function namesOf(v: unknown): string[] {
|
||||
if (!Array.isArray(v)) return []
|
||||
@@ -88,16 +33,6 @@ function namesOf(v: unknown): string[] {
|
||||
.filter((n): n is string => typeof n === 'string' && n.length > 0)
|
||||
}
|
||||
|
||||
/** Every valid canonical target value for a catalog (excludes the empty option). */
|
||||
function targetValues(cat: TargetCatalog): Set<string> {
|
||||
const s = new Set<string>(['direct', 'block'])
|
||||
for (const g of cat.groups) s.add(`group:${g}`)
|
||||
for (const c of cat.chains) s.add(`chain:${c}`)
|
||||
for (const n of cat.nodes) s.add(`node:${n}`)
|
||||
for (const e of cat.egresses) s.add(`egress:${e.name}`)
|
||||
return s
|
||||
}
|
||||
|
||||
// ---- page ------------------------------------------------------------------
|
||||
|
||||
export default function Profiles() {
|
||||
@@ -117,6 +52,24 @@ export default function Profiles() {
|
||||
void loadConfig()
|
||||
}, [loadConfig])
|
||||
|
||||
// The router's UCI interfaces feed the uplink-condition picker. Best-effort:
|
||||
// if the list can't be fetched the editor keeps existing chips removable and
|
||||
// simply offers nothing to add (same degradation as the Targets egress picker).
|
||||
const [interfaces, setInterfaces] = useState<Interface[]>([])
|
||||
useEffect(() => {
|
||||
let alive = true
|
||||
getInterfaces()
|
||||
.then((ifs) => {
|
||||
if (alive) setInterfaces(ifs)
|
||||
})
|
||||
.catch(() => {
|
||||
/* leave empty → chips stay removable, nothing to add */
|
||||
})
|
||||
return () => {
|
||||
alive = false
|
||||
}
|
||||
}, [])
|
||||
|
||||
// ---- toast + persistent apply banner (mirrors DNS / Settings) -------------
|
||||
const [toast, setToast] = useState<string | null>(null)
|
||||
const toastTimer = useRef<number | undefined>(undefined)
|
||||
@@ -174,22 +127,9 @@ export default function Profiles() {
|
||||
// ---- derived slices -------------------------------------------------------
|
||||
const globals = config?.Globals
|
||||
const profiles = useMemo<Profile[]>(() => asArray(config?.Profiles), [config])
|
||||
const presets = useMemo<Preset[]>(() => asArray(config?.Presets), [config])
|
||||
const ruleNames = useMemo(() => namesOf(config?.Rules), [config])
|
||||
const egresses = useMemo(() => asArray(config?.Egresses), [config])
|
||||
const resolverNames = useMemo(() => namesOf(config?.Resolvers), [config])
|
||||
|
||||
const catalog = useMemo<TargetCatalog>(
|
||||
() => ({
|
||||
groups: namesOf(config?.Groups),
|
||||
chains: namesOf(config?.['Chains']),
|
||||
nodes: namesOf(config?.Nodes),
|
||||
egresses: egresses.map((e) => ({ name: e.Name, type: e.Type })),
|
||||
}),
|
||||
[config, egresses],
|
||||
)
|
||||
const targetValid = useMemo(() => targetValues(catalog), [catalog])
|
||||
|
||||
const profileNames = useMemo(() => new Set(profiles.map((p) => p.Name)), [profiles])
|
||||
const active = globals?.ActiveProfile ?? ''
|
||||
const busy = saving || applying
|
||||
@@ -226,7 +166,6 @@ export default function Profiles() {
|
||||
Enabled: true,
|
||||
Priority: priority,
|
||||
MatchIface: [],
|
||||
SchedDays: [],
|
||||
EnableRules: [],
|
||||
DisableRules: [],
|
||||
}
|
||||
@@ -266,19 +205,6 @@ export default function Profiles() {
|
||||
[config, profiles, save],
|
||||
)
|
||||
|
||||
// ---- preset-pack upsert ---------------------------------------------------
|
||||
const setPreset = useCallback(
|
||||
(name: string, patch: Partial<Preset>, msg: string) => {
|
||||
if (!config) return
|
||||
const exists = presets.some((p) => p.Name === name)
|
||||
const next = exists
|
||||
? presets.map((p) => (p.Name === name ? { ...p, ...patch } : p))
|
||||
: [...presets, { Name: name, Enabled: false, ...patch }]
|
||||
void save({ ...config, Presets: next }, msg)
|
||||
},
|
||||
[config, presets, save],
|
||||
)
|
||||
|
||||
// ---- expansion (only one profile editor open at a time) -------------------
|
||||
const [openName, setOpenName] = useState<string | null>(null)
|
||||
// Rename remounts nothing (row keyed by index) but the open pointer must follow.
|
||||
@@ -319,7 +245,7 @@ export default function Profiles() {
|
||||
const loading = config === null && loadError === null
|
||||
|
||||
return (
|
||||
<section className="page pf-page" aria-label="Profiles and preset packs">
|
||||
<section className="page pf-page" aria-label="Profiles">
|
||||
{loadError && (
|
||||
<p className="page-error" role="alert">
|
||||
Couldn’t read config — {loadError}.{' '}
|
||||
@@ -351,7 +277,7 @@ export default function Profiles() {
|
||||
onChange={setActive}
|
||||
/>
|
||||
|
||||
{/* ---- 1. PROFILES ---- */}
|
||||
{/* ---- PROFILES ---- */}
|
||||
<div className="pf-section" aria-label="Profiles">
|
||||
<header className="pf-sec-hd">
|
||||
<h2 className="pf-sec-title">Profiles</h2>
|
||||
@@ -360,9 +286,9 @@ export default function Profiles() {
|
||||
</span>
|
||||
</header>
|
||||
<p className="pf-sec-note">
|
||||
A profile is a named override that switches on by condition — which uplink is carrying the
|
||||
router, or a time window. When active it can flip rules on or off and change the default
|
||||
route. Higher priority wins; a manual override above beats every condition.
|
||||
A profile is a named override that switches on by condition — which uplink is carrying
|
||||
the router. When active it can flip rules on or off. Higher priority wins; a manual
|
||||
override above beats every condition.
|
||||
</p>
|
||||
|
||||
<AddProfileForm busy={busy} disabled={!config} taken={profileNames} onAdd={addProfile} />
|
||||
@@ -376,7 +302,7 @@ export default function Profiles() {
|
||||
<div className="pf-empty">
|
||||
<span className="pf-empty-title mono">No profiles</span>
|
||||
<p className="pf-empty-body">
|
||||
No profiles — add one to auto-switch routing by uplink or schedule.
|
||||
No profiles — add one to auto-switch routing by the active uplink.
|
||||
</p>
|
||||
</div>
|
||||
) : (
|
||||
@@ -391,10 +317,8 @@ export default function Profiles() {
|
||||
onOpen={() => setOpenName(openName === p.Name ? null : p.Name)}
|
||||
busy={busy}
|
||||
ruleNames={ruleNames}
|
||||
egresses={egresses}
|
||||
interfaces={interfaces}
|
||||
resolverNames={resolverNames}
|
||||
catalog={catalog}
|
||||
valid={targetValid}
|
||||
taken={profileNames}
|
||||
onPatch={(patch, msg) => patchProfile(p.Name, patch, msg)}
|
||||
onRename={(nn) => onRename(p.Name, nn)}
|
||||
@@ -405,38 +329,6 @@ export default function Profiles() {
|
||||
)}
|
||||
</div>
|
||||
|
||||
{/* ---- 2. PRESET PACKS ---- */}
|
||||
<div className="pf-section" aria-label="Preset packs">
|
||||
<header className="pf-sec-hd">
|
||||
<h2 className="pf-sec-title">Preset packs</h2>
|
||||
<span className="pf-sec-count mono">
|
||||
{presets.filter((p) => p.Enabled).length} / {PACKS.length} on
|
||||
</span>
|
||||
</header>
|
||||
<p className="pf-sec-note">
|
||||
Curated rule bundles, built in. Turn one on to add its rules to the router; set a target to
|
||||
override where its traffic goes.
|
||||
</p>
|
||||
|
||||
<ul className="pf-rows">
|
||||
{PACKS.map((pack) => {
|
||||
const entry = presets.find((p) => p.Name === pack.name)
|
||||
return (
|
||||
<PackRow
|
||||
key={pack.name}
|
||||
pack={pack}
|
||||
preset={entry}
|
||||
busy={busy}
|
||||
disabled={!config}
|
||||
catalog={catalog}
|
||||
valid={targetValid}
|
||||
onSet={(patch, msg) => setPreset(pack.name, patch, msg)}
|
||||
/>
|
||||
)
|
||||
})}
|
||||
</ul>
|
||||
</div>
|
||||
|
||||
{toast && (
|
||||
<div className="toast" role="status">
|
||||
{toast}
|
||||
@@ -660,10 +552,8 @@ function ProfileRow({
|
||||
onOpen,
|
||||
busy,
|
||||
ruleNames,
|
||||
egresses,
|
||||
interfaces,
|
||||
resolverNames,
|
||||
catalog,
|
||||
valid,
|
||||
taken,
|
||||
onPatch,
|
||||
onRename,
|
||||
@@ -676,10 +566,8 @@ function ProfileRow({
|
||||
onOpen: () => void
|
||||
busy: boolean
|
||||
ruleNames: string[]
|
||||
egresses: { Name: string }[]
|
||||
interfaces: Interface[]
|
||||
resolverNames: string[]
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
taken: Set<string>
|
||||
onPatch: (patch: Partial<Profile>, msg: string) => void
|
||||
onRename: (newName: string) => void
|
||||
@@ -688,8 +576,6 @@ function ProfileRow({
|
||||
const iface = asArray(profile.MatchIface)
|
||||
const enableRules = asArray(profile.EnableRules)
|
||||
const disableRules = asArray(profile.DisableRules)
|
||||
const days = asArray(profile.SchedDays)
|
||||
const hasSchedule = days.length > 0 || !!profile.SchedStart || !!profile.SchedEnd
|
||||
|
||||
return (
|
||||
<li className={profile.Enabled ? 'pf-row' : 'pf-row off'}>
|
||||
@@ -721,42 +607,16 @@ function ProfileRow({
|
||||
)}
|
||||
</div>
|
||||
<div className="pf-row-l2">
|
||||
{!hasSchedule && iface.length === 0 ? (
|
||||
{iface.length === 0 ? (
|
||||
<span className="pf-chip pf-chip--muted">always — no condition set</span>
|
||||
) : (
|
||||
<>
|
||||
{iface.length > 0 && (
|
||||
<span className="pf-chip">
|
||||
<b>iface</b>
|
||||
<span className="mono">{iface.join(', ')}</span>
|
||||
</span>
|
||||
)}
|
||||
{hasSchedule && (
|
||||
<span className="pf-chip">
|
||||
<b>time</b>
|
||||
<span className="mono">
|
||||
{days.length ? days.map(cap).join(',') + ' ' : ''}
|
||||
{(profile.SchedStart || '00:00') + '–' + (profile.SchedEnd || '24:00')}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
</>
|
||||
)}
|
||||
{(enableRules.length > 0 ||
|
||||
disableRules.length > 0 ||
|
||||
!!profile.DefaultTarget ||
|
||||
!!profile.DefaultEgress) && <span className="pf-chip-sep" aria-hidden="true">→</span>}
|
||||
{profile.DefaultTarget && (
|
||||
<span className="pf-chip pf-chip--target">
|
||||
<b>route</b>
|
||||
<span className="mono">{profile.DefaultTarget}</span>
|
||||
<span className="pf-chip">
|
||||
<b>iface</b>
|
||||
<span className="mono">{iface.join(', ')}</span>
|
||||
</span>
|
||||
)}
|
||||
{profile.DefaultEgress && (
|
||||
<span className="pf-chip pf-chip--target">
|
||||
<b>egress</b>
|
||||
<span className="mono">{profile.DefaultEgress}</span>
|
||||
</span>
|
||||
{(enableRules.length > 0 || disableRules.length > 0) && (
|
||||
<span className="pf-chip-sep" aria-hidden="true">→</span>
|
||||
)}
|
||||
{enableRules.length > 0 && (
|
||||
<span
|
||||
@@ -803,10 +663,8 @@ function ProfileRow({
|
||||
profile={profile}
|
||||
busy={busy}
|
||||
ruleNames={ruleNames}
|
||||
egresses={egresses}
|
||||
interfaces={interfaces}
|
||||
resolverNames={resolverNames}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
taken={taken}
|
||||
onPatch={onPatch}
|
||||
onRename={onRename}
|
||||
@@ -822,10 +680,8 @@ function ProfileEditor({
|
||||
profile,
|
||||
busy,
|
||||
ruleNames,
|
||||
egresses,
|
||||
interfaces,
|
||||
resolverNames,
|
||||
catalog,
|
||||
valid,
|
||||
taken,
|
||||
onPatch,
|
||||
onRename,
|
||||
@@ -833,10 +689,8 @@ function ProfileEditor({
|
||||
profile: Profile
|
||||
busy: boolean
|
||||
ruleNames: string[]
|
||||
egresses: { Name: string }[]
|
||||
interfaces: Interface[]
|
||||
resolverNames: string[]
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
taken: Set<string>
|
||||
onPatch: (patch: Partial<Profile>, msg: string) => void
|
||||
onRename: (newName: string) => void
|
||||
@@ -844,7 +698,6 @@ function ProfileEditor({
|
||||
const iface = asArray(profile.MatchIface)
|
||||
const enableRules = asArray(profile.EnableRules)
|
||||
const disableRules = asArray(profile.DisableRules)
|
||||
const days = asArray(profile.SchedDays)
|
||||
|
||||
// Name is the only field that can't be a bare instant-save (needs a uniqueness
|
||||
// guard + it moves the ActiveProfile pointer), so it commits on blur/Enter.
|
||||
@@ -862,18 +715,6 @@ function ProfileEditor({
|
||||
const removeIface = (dev: string) =>
|
||||
onPatch({ MatchIface: iface.filter((x) => x !== dev) }, `iface − ${dev}`)
|
||||
|
||||
// Every schedule edit re-anchors the window to the editing browser's UTC
|
||||
// offset — the daemon has no tzdata, so the offset IS the timezone (see the
|
||||
// schedule helpers at the top of the file).
|
||||
const patchSched = (patch: Partial<Profile>, msg: string) =>
|
||||
onPatch({ ...patch, SchedUTCOffset: browserUTCOffsetMin() }, msg)
|
||||
|
||||
const toggleDay = (d: string) =>
|
||||
patchSched(
|
||||
{ SchedDays: days.includes(d) ? days.filter((x) => x !== d) : [...days, d] },
|
||||
'Schedule updated',
|
||||
)
|
||||
|
||||
// A rule sits in at most one override list. Checking it in one clears the other.
|
||||
const toggleRule = (rule: string, list: 'enable' | 'disable') => {
|
||||
if (list === 'enable') {
|
||||
@@ -917,15 +758,14 @@ function ProfileEditor({
|
||||
|
||||
{/* -- conditions -- */}
|
||||
<fieldset className="pf-eblock">
|
||||
<legend className="pf-elegend">Conditions — all must hold</legend>
|
||||
<legend className="pf-elegend">Condition</legend>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Uplink interface</span>
|
||||
<div className="pf-erow-ctl">
|
||||
<ChipInput
|
||||
<IfaceChips
|
||||
chips={iface}
|
||||
placeholder="wwan0, usb0…"
|
||||
ariaLabel="Uplink interface names"
|
||||
interfaces={interfaces}
|
||||
busy={busy}
|
||||
onAdd={addIface}
|
||||
onRemove={removeIface}
|
||||
@@ -934,131 +774,17 @@ function ProfileEditor({
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* A "Probe" condition (URL + activate-when-up/down) used to sit here. It
|
||||
never probed anything, and it did worse than nothing: a profile that
|
||||
carried one was treated as having a condition that could never hold, so
|
||||
filling this in SWITCHED OFF an otherwise working profile. Both fields
|
||||
are gone from the daemon; the conditions that remain are the two below,
|
||||
which are read for real. */}
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Schedule</span>
|
||||
<div className="pf-erow-ctl">
|
||||
{iface.length > 0 && (
|
||||
<p className="pf-ehint">
|
||||
Applies together with the uplink match: the profile is active only while the uplink
|
||||
matches AND the window is open (the watcher re-checks both every ~25 s).
|
||||
</p>
|
||||
)}
|
||||
<div className="pf-days" role="group" aria-label="Active days (none = every day)">
|
||||
{DAYS.map((d) => {
|
||||
const on = days.includes(d)
|
||||
return (
|
||||
<button
|
||||
type="button"
|
||||
key={d}
|
||||
className={on ? 'pf-day on' : 'pf-day'}
|
||||
aria-pressed={on}
|
||||
onClick={() => toggleDay(d)}
|
||||
disabled={busy}
|
||||
>
|
||||
{cap(d)}
|
||||
</button>
|
||||
)
|
||||
})}
|
||||
</div>
|
||||
<div className="pf-times">
|
||||
<label className="pf-field pf-field--time">
|
||||
<span className="pf-flabel">From</span>
|
||||
<input
|
||||
className="pf-input mono"
|
||||
type="time"
|
||||
value={profile.SchedStart ?? ''}
|
||||
onChange={(e) => patchSched({ SchedStart: e.target.value }, 'Schedule updated')}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
<label className="pf-field pf-field--time">
|
||||
<span className="pf-flabel">To</span>
|
||||
<input
|
||||
className="pf-input mono"
|
||||
type="time"
|
||||
value={profile.SchedEnd ?? ''}
|
||||
onChange={(e) => patchSched({ SchedEnd: e.target.value }, 'Schedule updated')}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
</div>
|
||||
<p className="pf-ehint">
|
||||
No days = every day · empty times = all day · times are in your browser's timezone (
|
||||
{browserTZName()}, {utcOffsetLabel(browserUTCOffsetMin())}) — the offset is saved with the
|
||||
schedule; DST shifts apply after a re-save.
|
||||
</p>
|
||||
{(profile.SchedUTCOffset ?? 0) !== browserUTCOffsetMin() &&
|
||||
((profile.SchedStart ?? '') !== '' ||
|
||||
(profile.SchedEnd ?? '') !== '' ||
|
||||
days.length > 0) && (
|
||||
<p className="pf-ehint">
|
||||
This schedule was saved at {utcOffsetLabel(profile.SchedUTCOffset ?? 0)} — the times
|
||||
above are on that clock. Editing any schedule field re-anchors it to your timezone.
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
</div>
|
||||
{/* A "Probe" condition (URL + activate-when-up/down) used to sit here, and
|
||||
a schedule window below it. The probe never probed anything and treated
|
||||
a profile carrying one as unsatisfiable — filling it in SWITCHED OFF an
|
||||
otherwise working profile; the schedule editor left with the feature.
|
||||
The uplink match above is the one condition left, read for real. */}
|
||||
</fieldset>
|
||||
|
||||
{/* -- overrides -- */}
|
||||
<fieldset className="pf-eblock">
|
||||
<legend className="pf-elegend">Overrides while active</legend>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Default route</span>
|
||||
<div className="pf-erow-ctl">
|
||||
<TargetSelect
|
||||
value={profile.DefaultTarget ?? ''}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
includeNoOverride
|
||||
noOverrideLabel="No override — keep default"
|
||||
busy={busy}
|
||||
ariaLabel="Default route target"
|
||||
onChange={(v) =>
|
||||
onPatch({ DefaultTarget: v }, v ? `Default route → ${v}` : 'Default route override cleared')
|
||||
}
|
||||
/>
|
||||
<p className="pf-ehint">Where unmatched traffic goes while this profile is active.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Default egress</span>
|
||||
<div className="pf-erow-ctl">
|
||||
<select
|
||||
className="pf-select"
|
||||
value={profile.DefaultEgress ?? ''}
|
||||
onChange={(e) =>
|
||||
onPatch(
|
||||
{ DefaultEgress: e.target.value },
|
||||
e.target.value ? `Egress → ${e.target.value}` : 'Egress override cleared',
|
||||
)
|
||||
}
|
||||
disabled={busy}
|
||||
aria-label="Default egress"
|
||||
>
|
||||
<option value="">No override</option>
|
||||
{egresses.map((eg) => (
|
||||
<option key={eg.Name} value={eg.Name}>
|
||||
{eg.Name}
|
||||
</option>
|
||||
))}
|
||||
{profile.DefaultEgress && !egresses.some((eg) => eg.Name === profile.DefaultEgress) && (
|
||||
<option value={profile.DefaultEgress}>{profile.DefaultEgress} (missing)</option>
|
||||
)}
|
||||
</select>
|
||||
<p className="pf-ehint">Pins the outbound interface / egress for this profile.</p>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="pf-erow">
|
||||
<span className="pf-elabel">Endpoint resolver</span>
|
||||
<div className="pf-erow-ctl">
|
||||
@@ -1141,172 +867,8 @@ function ProfileEditor({
|
||||
)
|
||||
}
|
||||
|
||||
// ---- a preset pack row -----------------------------------------------------
|
||||
|
||||
function PackRow({
|
||||
pack,
|
||||
preset,
|
||||
busy,
|
||||
disabled,
|
||||
catalog,
|
||||
valid,
|
||||
onSet,
|
||||
}: {
|
||||
pack: { name: string; title: string; desc: string }
|
||||
preset: Preset | undefined
|
||||
busy: boolean
|
||||
disabled: boolean
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
onSet: (patch: Partial<Preset>, msg: string) => void
|
||||
}) {
|
||||
const enabled = preset?.Enabled ?? false
|
||||
const target = preset?.Target ?? ''
|
||||
const [advOpen, setAdvOpen] = useState(false)
|
||||
|
||||
return (
|
||||
<li className={enabled ? 'pf-row pf-pack' : 'pf-row pf-pack off'}>
|
||||
<div className="pf-row-head">
|
||||
<Toggle
|
||||
pressed={enabled}
|
||||
onChange={(on) => onSet({ Enabled: on }, `${pack.name} ${on ? 'enabled' : 'disabled'}`)}
|
||||
label={`${enabled ? 'Disable' : 'Enable'} ${pack.title}`}
|
||||
disabled={busy || disabled}
|
||||
/>
|
||||
<div className="pf-row-main">
|
||||
<div className="pf-row-l1">
|
||||
<span className="pf-row-name">{pack.title}</span>
|
||||
<span className="pf-badge mono">{pack.name}</span>
|
||||
</div>
|
||||
<p className="pf-pack-desc">{pack.desc}</p>
|
||||
</div>
|
||||
<label className="pf-pack-target">
|
||||
<span className="pf-active-ctl-label mono">Target</span>
|
||||
<TargetSelect
|
||||
value={target}
|
||||
catalog={catalog}
|
||||
valid={valid}
|
||||
includeNoOverride
|
||||
noOverrideLabel="Pack default"
|
||||
busy={busy}
|
||||
disabled={disabled}
|
||||
ariaLabel={`${pack.title} target`}
|
||||
onChange={(v) => onSet({ Target: v }, v ? `${pack.name} → ${v}` : `${pack.name} → pack default`)}
|
||||
/>
|
||||
</label>
|
||||
</div>
|
||||
|
||||
<details
|
||||
className="pf-adv"
|
||||
open={advOpen}
|
||||
onToggle={(e) => setAdvOpen((e.target as HTMLDetailsElement).open)}
|
||||
>
|
||||
<summary className="pf-adv-summary">Advanced</summary>
|
||||
<div className="pf-adv-body">
|
||||
<label className="pf-field pf-field--num">
|
||||
<span className="pf-flabel">Order</span>
|
||||
<BlurField
|
||||
value={preset?.Order != null ? String(preset.Order) : ''}
|
||||
onCommit={(v) => {
|
||||
const s = v.trim()
|
||||
if (s === '') {
|
||||
onSet({ Order: undefined }, `${pack.name} order cleared`)
|
||||
} else if (/^\d+$/.test(s)) {
|
||||
onSet({ Order: Number(s) }, `${pack.name} order → ${s}`)
|
||||
}
|
||||
}}
|
||||
placeholder="auto"
|
||||
ariaLabel={`${pack.title} order`}
|
||||
busy={busy}
|
||||
width="6rem"
|
||||
inputMode="numeric"
|
||||
/>
|
||||
</label>
|
||||
<p className="pf-ehint">Lower runs first among packs. Leave blank for the built-in order.</p>
|
||||
</div>
|
||||
</details>
|
||||
</li>
|
||||
)
|
||||
}
|
||||
|
||||
// ---- shared controls -------------------------------------------------------
|
||||
|
||||
/** Canonical target picker (direct/block + groups/chains/nodes/egresses). */
|
||||
function TargetSelect({
|
||||
value,
|
||||
catalog,
|
||||
valid,
|
||||
includeNoOverride,
|
||||
noOverrideLabel,
|
||||
busy,
|
||||
disabled,
|
||||
ariaLabel,
|
||||
onChange,
|
||||
}: {
|
||||
value: string
|
||||
catalog: TargetCatalog
|
||||
valid: Set<string>
|
||||
includeNoOverride?: boolean
|
||||
noOverrideLabel?: string
|
||||
busy: boolean
|
||||
disabled?: boolean
|
||||
ariaLabel: string
|
||||
onChange: (v: string) => void
|
||||
}) {
|
||||
const missing = value !== '' && !valid.has(value)
|
||||
return (
|
||||
<select
|
||||
className="pf-select"
|
||||
value={value}
|
||||
onChange={(e) => onChange(e.target.value)}
|
||||
disabled={busy || disabled}
|
||||
aria-label={ariaLabel}
|
||||
>
|
||||
{includeNoOverride && <option value="">{noOverrideLabel ?? 'No override'}</option>}
|
||||
<option value="direct">Direct (no proxy)</option>
|
||||
<option value="block">Block</option>
|
||||
{catalog.groups.length > 0 && (
|
||||
<optgroup label="Groups">
|
||||
{catalog.groups.map((g) => (
|
||||
<option key={g} value={`group:${g}`}>
|
||||
Group {g} (balancer)
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.chains.length > 0 && (
|
||||
<optgroup label="Chains">
|
||||
{catalog.chains.map((c) => (
|
||||
<option key={c} value={`chain:${c}`}>
|
||||
Chain {c}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.nodes.length > 0 && (
|
||||
<optgroup label="Nodes">
|
||||
{catalog.nodes.map((n) => (
|
||||
<option key={n} value={`node:${n}`}>
|
||||
Node {n}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{catalog.egresses.length > 0 && (
|
||||
<optgroup label="Interfaces / egresses">
|
||||
{catalog.egresses.map((e) => (
|
||||
<option key={e.name} value={`egress:${e.name}`}>
|
||||
Interface/egress {e.name}
|
||||
{e.type ? ` (${e.type})` : ''}
|
||||
</option>
|
||||
))}
|
||||
</optgroup>
|
||||
)}
|
||||
{missing && <option value={value}>{value} (missing)</option>}
|
||||
</select>
|
||||
)
|
||||
}
|
||||
|
||||
/** A text input that keeps a local draft and commits on blur / Enter, reverts on Escape. */
|
||||
function BlurField({
|
||||
value,
|
||||
@@ -1361,65 +923,69 @@ function BlurField({
|
||||
)
|
||||
}
|
||||
|
||||
/** A chips editor: type + Enter to add, ✕ to remove. */
|
||||
function ChipInput({
|
||||
/** Uplink-iface chips + a dropdown of the router's UCI interfaces to add from.
|
||||
* A stored iface missing from the live list still renders as a chip (marked
|
||||
* stale) so it stays removable; with no interfaces fetched there is simply
|
||||
* nothing to add. */
|
||||
function IfaceChips({
|
||||
chips,
|
||||
placeholder,
|
||||
ariaLabel,
|
||||
interfaces,
|
||||
busy,
|
||||
onAdd,
|
||||
onRemove,
|
||||
}: {
|
||||
chips: string[]
|
||||
placeholder?: string
|
||||
ariaLabel: string
|
||||
interfaces: Interface[]
|
||||
busy: boolean
|
||||
onAdd: (v: string) => void
|
||||
onRemove: (v: string) => void
|
||||
}) {
|
||||
const [draft, setDraft] = useState('')
|
||||
|
||||
const commit = () => {
|
||||
const v = draft.trim()
|
||||
if (!v) return
|
||||
onAdd(v)
|
||||
setDraft('')
|
||||
}
|
||||
|
||||
const known = new Set(interfaces.map((i) => i.name))
|
||||
const available = interfaces.filter((i) => !chips.includes(i.name))
|
||||
return (
|
||||
<div className="pf-chipinput">
|
||||
{chips.map((c) => (
|
||||
<span key={c} className="pf-chip pf-chip--edit">
|
||||
<span className="mono">{c}</span>
|
||||
<button
|
||||
type="button"
|
||||
className="pf-chip-x"
|
||||
aria-label={`Remove ${c}`}
|
||||
onClick={() => onRemove(c)}
|
||||
disabled={busy}
|
||||
{chips.map((c) => {
|
||||
const stale = interfaces.length > 0 && !known.has(c)
|
||||
return (
|
||||
<span
|
||||
key={c}
|
||||
className="pf-chip pf-chip--edit"
|
||||
title={stale ? 'Not among the router’s interfaces' : undefined}
|
||||
>
|
||||
✕
|
||||
</button>
|
||||
</span>
|
||||
))}
|
||||
<input
|
||||
className="pf-input pf-chipinput-in mono"
|
||||
type="text"
|
||||
value={draft}
|
||||
placeholder={placeholder}
|
||||
aria-label={ariaLabel}
|
||||
autoComplete="off"
|
||||
spellCheck={false}
|
||||
disabled={busy}
|
||||
onChange={(e) => setDraft(e.target.value)}
|
||||
onKeyDown={(e) => {
|
||||
if (e.key === 'Enter' || e.key === ',') {
|
||||
e.preventDefault()
|
||||
commit()
|
||||
}
|
||||
}}
|
||||
onBlur={commit}
|
||||
/>
|
||||
<span className="mono">
|
||||
{c}
|
||||
{stale ? ' (stale)' : ''}
|
||||
</span>
|
||||
<button
|
||||
type="button"
|
||||
className="pf-chip-x"
|
||||
aria-label={`Remove ${c}`}
|
||||
onClick={() => onRemove(c)}
|
||||
disabled={busy}
|
||||
>
|
||||
✕
|
||||
</button>
|
||||
</span>
|
||||
)
|
||||
})}
|
||||
{available.length > 0 && (
|
||||
<select
|
||||
className="pf-select"
|
||||
value=""
|
||||
onChange={(e) => {
|
||||
if (e.target.value) onAdd(e.target.value)
|
||||
}}
|
||||
disabled={busy}
|
||||
aria-label="Add uplink interface"
|
||||
>
|
||||
<option value="">Add interface…</option>
|
||||
{available.map((i) => (
|
||||
<option key={i.name} value={i.name}>
|
||||
{i.name} ({i.device || '?'}){i.up ? '' : ' — down'}
|
||||
</option>
|
||||
))}
|
||||
</select>
|
||||
)}
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
@@ -103,64 +103,6 @@ function parseDuration(raw: string): ParseResult<string> {
|
||||
return { ok: true, value: s.toLowerCase() }
|
||||
}
|
||||
|
||||
// ---- health sweep interval -------------------------------------------------
|
||||
//
|
||||
// The daemon's own vocabulary (model.SweepDisabledValues / SweepIntervalMin /
|
||||
// SweepSchedule), mirrored here so the input accepts exactly what the router
|
||||
// accepts and nothing else.
|
||||
|
||||
/** The four spellings that switch the background sweep OFF. */
|
||||
const SWEEP_OFF = ['0', 'off', 'none', 'disabled']
|
||||
/** Anything shorter is raised to this by the daemon, so the field raises it here. */
|
||||
const SWEEP_MIN_S = 5
|
||||
const UNIT_S: Record<string, number> = { ms: 0.001, s: 1, m: 60, h: 3600 }
|
||||
|
||||
/** Seconds in a duration this field already validated, or null. */
|
||||
function durationSeconds(s: string): number | null {
|
||||
const m = /^(\d+(?:\.\d+)?)(ms|s|m|h)$/.exec(s)
|
||||
return m ? Number(m[1]) * UNIT_S[m[2]] : null
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse the sweep tick.
|
||||
*
|
||||
* The rule worth stating: an unrecognised value does NOT mean "off". The daemon
|
||||
* warns about it and quietly falls back to its default, so a value typed as a way
|
||||
* to stop the sweep would leave the sweep running. The field refuses it here
|
||||
* instead, where the person can still see what they typed.
|
||||
*
|
||||
* '' → the engine default (every 10 s)
|
||||
* 0 / off / none / disabled → stopped
|
||||
* 30, 30s, 5m, 1h → that tick, raised to the 5 s floor
|
||||
*/
|
||||
function parseSweepInterval(raw: string): ParseResult<string> {
|
||||
const s = raw.trim().toLowerCase()
|
||||
if (s === '') return { ok: true, value: '' }
|
||||
if (SWEEP_OFF.includes(s)) return { ok: true, value: s }
|
||||
// A bare integer is seconds, as everywhere else in the model — normalised so the
|
||||
// stored value says which unit it meant.
|
||||
const normalised = /^\d+$/.test(s) ? `${s}s` : s
|
||||
const seconds = DURATION_RE.test(normalised) ? durationSeconds(normalised) : null
|
||||
if (seconds === null) {
|
||||
return {
|
||||
ok: false,
|
||||
error: 'Use a duration like 30s, 5m, or 1h — or “off” to stop the sweep.',
|
||||
}
|
||||
}
|
||||
// The floor, applied where it is visible. Left to the daemon it would be a
|
||||
// warning in a log nobody reads while the field went on showing 1s.
|
||||
if (seconds < SWEEP_MIN_S) return { ok: true, value: `${SWEEP_MIN_S}s` }
|
||||
return { ok: true, value: normalised }
|
||||
}
|
||||
|
||||
/** What a saved sweep value means, as the toast that confirms it. */
|
||||
function sweepSavedMsg(v: string): string {
|
||||
if (v === '') return 'Health sweep → the default, every 10 s'
|
||||
if (SWEEP_OFF.includes(v)) return 'Health sweep off — most nodes will read “not measured”'
|
||||
if (v === `${SWEEP_MIN_S}s`) return `Health sweep → ${v} (the shortest allowed)`
|
||||
return `Health sweep → every ${v}`
|
||||
}
|
||||
|
||||
// `none` really does silence the engine log — it is emitted as the engine's own
|
||||
// log-disable switch, not as a quieter level.
|
||||
const LOG_LEVELS: ReadonlyArray<{ value: string; label: string }> = [
|
||||
@@ -310,9 +252,8 @@ export default function Settings() {
|
||||
[flash],
|
||||
)
|
||||
|
||||
// Our group-member health sweep + all the health/testing stats on the Targets
|
||||
// page. Absent ⇒ enabled (older config), so read it as `!== false`. When off, the
|
||||
// sweep-interval field below is meaningless, so it is dimmed (value kept).
|
||||
// Our background group-member health probing + all the health/testing stats on
|
||||
// the Targets page. Absent ⇒ enabled (older config), so read it as `!== false`.
|
||||
const groupHealthOn = globals?.GroupHealth !== false
|
||||
|
||||
const killSwitch = globals?.KillSwitch === 'open' ? 'open' : 'closed'
|
||||
@@ -588,7 +529,7 @@ export default function Settings() {
|
||||
</p>
|
||||
<Field
|
||||
label="Group health checks"
|
||||
note="Runs a background sweep that checks each group’s member nodes and powers the alive / tested health stats on the Targets page. Turn it off to stop that sweep and hide those stats — your groups still keep picking a live node on their own (sing-box probes them under the hood). The probe URL and interval below still feed those built-in checks."
|
||||
note="Lets the daemon probe the groups and chains your routing rules actually use, in the background, and powers the alive / tested health stats on the Targets page. Anything no enabled rule routes through is skipped and simply reads as unused. Turn it off to stop that background probing and hide those stats — your groups still keep picking a live node on their own (sing-box probes them under the hood, using the probe URL and interval below)."
|
||||
>
|
||||
<Toggle
|
||||
pressed={groupHealthOn}
|
||||
@@ -596,7 +537,7 @@ export default function Settings() {
|
||||
setGlobal(
|
||||
'GroupHealth',
|
||||
on,
|
||||
on ? 'Group health checks on' : 'Group health checks off — sweep stopped',
|
||||
on ? 'Group health checks on' : 'Group health checks off — background probing stopped',
|
||||
)
|
||||
}
|
||||
label={groupHealthOn ? 'Turn off group health checks' : 'Turn on group health checks'}
|
||||
@@ -631,26 +572,6 @@ export default function Settings() {
|
||||
onCommit={(v) => setGlobal('ProbeInterval', v, `Probe interval → ${v}`)}
|
||||
/>
|
||||
</Field>
|
||||
<Field
|
||||
label="Health sweep interval"
|
||||
note={
|
||||
groupHealthOn
|
||||
? 'How often the router probes nodes in the background so the health numbers on the Targets page stay fresh. Blank = every 10 s (recommended). Each tick measures only the nodes nobody else is checking, a few at a time, so a full pass takes roughly 5–10 minutes. Set “off” to stop it — at the cost of most nodes reading “not measured”.'
|
||||
: 'Group health checks are off, so the sweep isn’t running and this interval has no effect. Turn group health checks back on to use it. (Your saved value is kept.)'
|
||||
}
|
||||
>
|
||||
<InlineEdit<string>
|
||||
value={globals?.SweepInterval ?? ''}
|
||||
format={(s) => s}
|
||||
parse={parseSweepInterval}
|
||||
width="8rem"
|
||||
placeholder="10s"
|
||||
ariaLabel="Health sweep interval"
|
||||
busy={busy}
|
||||
disabled={!ready || !groupHealthOn}
|
||||
onCommit={(v) => setGlobal('SweepInterval', v, sweepSavedMsg(v))}
|
||||
/>
|
||||
</Field>
|
||||
</Group>
|
||||
|
||||
{/* ---- STATISTICS & LOGGING ---- */}
|
||||
|
||||
+24
-13
@@ -627,13 +627,12 @@
|
||||
padding: 6px 14px;
|
||||
font-size: 11px;
|
||||
}
|
||||
/* ---- run indicators (header) ----
|
||||
* Two background runs report here, and they are different actions with different
|
||||
* reach — a health refresh covers every group at once, an exit test covers only
|
||||
* the groups it names. Each says WHAT it is and HOW FAR it has got, so neither can
|
||||
* be mistaken for the other, and neither is ever repeated as a badge on the cards.
|
||||
* The third is the sweep: not a run someone started, so no LED and no pulse — a
|
||||
* quiet gauge that explains why untested members fill in by themselves. */
|
||||
/* ---- run indicator (header) ----
|
||||
* The one manual run left — the exit test — reports here. It says WHAT it is
|
||||
* and HOW FAR it has got, and it is never repeated as a badge on cards outside
|
||||
* its scope. The observatory's background probing deliberately has no gauge:
|
||||
* it is not a run someone started, and its whole visible effect is that the
|
||||
* numbers below stay fresh on their own. */
|
||||
.tg-run {
|
||||
display: inline-flex;
|
||||
align-items: center;
|
||||
@@ -655,16 +654,11 @@
|
||||
font-size: 10.5px;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* An in-flight run is amber-edged; the sweep is not a run and stays neutral. */
|
||||
.tg-run--health,
|
||||
/* An in-flight run is amber-edged. */
|
||||
.tg-run--exit {
|
||||
border-color: color-mix(in srgb, var(--amber) 45%, var(--groove));
|
||||
color: var(--ink);
|
||||
}
|
||||
.tg-run--sweep {
|
||||
cursor: help;
|
||||
color: var(--faint);
|
||||
}
|
||||
.tg-test-err {
|
||||
margin: 10px 2px 0;
|
||||
font-family: var(--font-sans);
|
||||
@@ -853,6 +847,23 @@
|
||||
font-size: 12px;
|
||||
color: var(--faint);
|
||||
}
|
||||
/* The unused note (GroupHealth.used === false). A routing fact, not a fault: no
|
||||
enabled rule reaches the group, so the observatory never probes it and there
|
||||
are no counters to show. Muted groove-badge styling on purpose — dim and
|
||||
bordered, never amber or red — because "nothing routes here" is not an
|
||||
alarm. */
|
||||
.gh-unused {
|
||||
display: inline-block;
|
||||
padding: 1px 7px;
|
||||
border: 1px solid var(--groove);
|
||||
border-radius: 999px;
|
||||
background: color-mix(in srgb, var(--sink) 45%, transparent);
|
||||
font-size: 10.5px;
|
||||
letter-spacing: 0.08em;
|
||||
text-transform: uppercase;
|
||||
color: var(--dim);
|
||||
cursor: help;
|
||||
}
|
||||
.gh-bound {
|
||||
padding: 1px 7px;
|
||||
border: 1px dashed color-mix(in srgb, var(--dim) 45%, var(--groove));
|
||||
|
||||
+158
-250
@@ -10,7 +10,6 @@ import {
|
||||
getInterfaces,
|
||||
getStatus,
|
||||
postGroupsTest,
|
||||
postNodesTest,
|
||||
putConfig,
|
||||
ApiError,
|
||||
} from '../api'
|
||||
@@ -22,7 +21,6 @@ import type {
|
||||
GroupsHealth,
|
||||
GroupTestResult,
|
||||
GroupTestStatus,
|
||||
HealthSweep,
|
||||
Chain,
|
||||
Egress,
|
||||
Interface,
|
||||
@@ -77,7 +75,8 @@ const without = (set: Set<string>, self: string): Set<string> =>
|
||||
// spellings: a PREFIXED target-or-detour string ("group:auto", "egress:wan") and,
|
||||
// for egresses only, a BARE binding ("Egress":"wan" on a node/rule/profile).
|
||||
// Renaming or deleting one without touching those references silently orphans
|
||||
// them — the daemon then falls back to the default route with no warning. These
|
||||
// them — the daemon then warns and BLOCKS the traffic of every rule left pointing
|
||||
// at nothing (a rule's traffic never falls through to the default route). These
|
||||
// helpers find and rewrite every site, so a rename carries and a delete can say
|
||||
// exactly what it will break.
|
||||
|
||||
@@ -123,10 +122,6 @@ function findReferences(m: Model, kind: RefKind, name: string): RefSite[] {
|
||||
for (const s of asArray(m.Subscriptions)) {
|
||||
if (isPrefixed(s.FetchDetour, kind, name)) out.push({ label: `subscription “${s.Name}” fetch` })
|
||||
}
|
||||
for (const p of asArray(m.Profiles)) {
|
||||
if (isPrefixed(p.DefaultTarget, kind, name)) out.push({ label: `profile “${p.Name}” target` })
|
||||
if (bare && isBare(p.DefaultEgress, name)) out.push({ label: `profile “${p.Name}” egress` })
|
||||
}
|
||||
if (bare) {
|
||||
for (const n of asArray(m.Nodes)) {
|
||||
if (isBare(n.Egress, name)) out.push({ label: `node “${n.Name}” egress` })
|
||||
@@ -160,12 +155,6 @@ function renameReferences(m: Model, kind: RefKind, from: string, to: string): Mo
|
||||
if (m.Alerts) next.Alerts = m.Alerts.map((a) => ({ ...a, Via: pfx(a.Via) }))
|
||||
if (m.Subscriptions)
|
||||
next.Subscriptions = m.Subscriptions.map((s) => ({ ...s, FetchDetour: pfx(s.FetchDetour) }))
|
||||
if (m.Profiles)
|
||||
next.Profiles = m.Profiles.map((p) => ({
|
||||
...p,
|
||||
DefaultTarget: pfx(p.DefaultTarget),
|
||||
DefaultEgress: br(p.DefaultEgress),
|
||||
}))
|
||||
if (bare && m.Nodes) next.Nodes = m.Nodes.map((n) => ({ ...n, Egress: br(n.Egress) ?? n.Egress }))
|
||||
if (bare && m.Groups) next.Groups = m.Groups.map((g) => ({ ...g, Egress: br(g.Egress) }))
|
||||
return next
|
||||
@@ -182,8 +171,8 @@ function refWarning(refs: RefSite[]): string {
|
||||
const more = refs.length - shown.length
|
||||
const list = `${shown.join(', ')}${more > 0 ? `, and ${more} more` : ''}`
|
||||
return refs.length === 1
|
||||
? ` It is referenced by ${list}, which falls back to the default route.`
|
||||
: ` It is referenced by ${refs.length} places — ${list} — which fall back to the default route.`
|
||||
? ` It is referenced by ${list}, whose traffic will be blocked (an unresolved target never falls through to the default route).`
|
||||
: ` It is referenced by ${refs.length} places — ${list} — whose traffic will be blocked (an unresolved target never falls through to the default route).`
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -306,12 +295,12 @@ const DPI_TYPES = new Set(['interface', 'direct'])
|
||||
// The rendering contract, from the daemon (see engine/grouphealth.go):
|
||||
// - "alive out of TESTED", with the untested remainder as a quiet aside shown
|
||||
// ONLY when it is non-zero.
|
||||
// - untested is never folded into dead. Groups probe lazily, so a freshly
|
||||
// booted router with a 376-node subscription is legitimately almost all
|
||||
// untested; rendering that as dead raises an alarm at the exact moment
|
||||
// - untested is never folded into dead. The observatory needs a few seconds
|
||||
// after an engine (re)start to reach a used group, and it never probes an
|
||||
// unused one; rendering either as dead raises an alarm at the exact moment
|
||||
// nothing is wrong, which teaches the operator to distrust every reading.
|
||||
|
||||
// ---- the sample is not always symmetric -------------------------------------
|
||||
// ---- where dead verdicts come from -------------------------------------------
|
||||
//
|
||||
// A group's own history is written by TWO parties, and only one of them can
|
||||
// record a failure:
|
||||
@@ -320,38 +309,38 @@ const DPI_TYPES = new Set(['interface', 'direct'])
|
||||
// failure DELETES the entry (protocol/group/urltest.go:512). It is physically
|
||||
// incapable of recording "dead" — its failures are indistinguishable from
|
||||
// "never probed";
|
||||
// - the background sweep (our overlay) is the only writer that puts a
|
||||
// trustworthy "dead" on record.
|
||||
// - the daemon's health board is the only writer that puts a trustworthy
|
||||
// "dead" on record: the observatory probes every group and chain an enabled
|
||||
// rule can reach (~10 s tick) and records BOTH outcomes, and the group
|
||||
// checkers feed it too.
|
||||
//
|
||||
// So until the sweep has been over a group, its history holds ONLY successes.
|
||||
// "11 / 11 alive" then does not mean the group is healthy; it means the failures
|
||||
// erased themselves. The real reading, minutes later, was 111 alive / 74 dead.
|
||||
//
|
||||
// That is an OPTIMISTIC lie, which is the worst kind here: it invites someone to
|
||||
// route traffic through a group where 40% of the members are down. So while the
|
||||
// sample is knowably one-sided the panel does not present it as a ratio at all —
|
||||
// a ratio promises a denominator that was checked in both directions, and this
|
||||
// one wasn't.
|
||||
// A used group's untested members therefore resolve to a real verdict within
|
||||
// seconds of the engine coming up. Until then the history holds ONLY successes,
|
||||
// and "11 / 11 alive" does not mean the group is healthy — the failures erased
|
||||
// themselves. That is an OPTIMISTIC lie, the worst kind here: it invites someone
|
||||
// to route traffic through a group where 40% of the members are down. So while
|
||||
// the sample is knowably one-sided the panel does not present it as a ratio at
|
||||
// all — a ratio promises a denominator that was checked in both directions, and
|
||||
// this one wasn't. The window is seconds now, but honesty in it costs nothing.
|
||||
|
||||
/** What a group's numbers add up to, in the six states they actually have. */
|
||||
type Verdict = 'good' | 'partial' | 'degraded' | 'down' | 'unmeasured' | 'empty'
|
||||
|
||||
/**
|
||||
* Is this group's sample knowably one-sided — only successes on record, with
|
||||
* members still unaccounted for and nothing yet able to have recorded a failure?
|
||||
* members still unaccounted for and no failure yet on the board?
|
||||
*
|
||||
* All three conditions matter:
|
||||
* Both conditions matter:
|
||||
* - `dead === 0` — the moment ONE failure is on record, something has been
|
||||
* writing both outcomes and the ratio is honest;
|
||||
* - `untested > 0` — a fully measured group has no room for hidden failures;
|
||||
* - `!sweptOnce` — once the sweep has completed a pass it has had its say
|
||||
* about every member, so silence now means something.
|
||||
* - `untested > 0` — a fully measured group has no room for hidden failures.
|
||||
*
|
||||
* It self-clears on the first failure or the first completed pass, whichever
|
||||
* comes first — no timers, no flags, nothing to get stuck.
|
||||
* It self-clears on the first failure or the first full coverage — and with the
|
||||
* observatory over every used group, that is seconds away. An UNUSED group never
|
||||
* reaches this code: its card says "unused" instead of rendering counters.
|
||||
*/
|
||||
function isBiasedSample(h: GroupHealth, sweptOnce: boolean): boolean {
|
||||
return h.dead === 0 && h.untested > 0 && !sweptOnce
|
||||
function isBiasedSample(h: GroupHealth): boolean {
|
||||
return h.dead === 0 && h.untested > 0
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -362,12 +351,12 @@ function isBiasedSample(h: GroupHealth, sweptOnce: boolean): boolean {
|
||||
* no basis for a proportion. Both decline to make a health claim rather than
|
||||
* make a flattering one.
|
||||
*/
|
||||
function verdictOf(h: GroupHealth, sweptOnce: boolean): Verdict {
|
||||
function verdictOf(h: GroupHealth): Verdict {
|
||||
if (h.total === 0) return 'empty'
|
||||
if (h.tested === 0) return 'unmeasured'
|
||||
if (h.alive === 0) return 'down'
|
||||
if (h.dead > 0) return 'degraded'
|
||||
if (isBiasedSample(h, sweptOnce)) return 'partial'
|
||||
if (isBiasedSample(h)) return 'partial'
|
||||
return 'good'
|
||||
}
|
||||
|
||||
@@ -428,12 +417,7 @@ function sortMembers(members: GroupMemberHealth[]): GroupMemberHealth[] {
|
||||
}
|
||||
|
||||
/** Nothing read yet — an empty, honest starting state (groups is never null). */
|
||||
const IDLE_HEALTH: GroupsHealth = {
|
||||
groups: [],
|
||||
node_test_running: false,
|
||||
node_test_done: 0,
|
||||
node_test_total: 0,
|
||||
}
|
||||
const IDLE_HEALTH: GroupsHealth = { groups: [] }
|
||||
|
||||
/** No test has run (or the page hasn't read one yet). */
|
||||
const IDLE_TEST: GroupTestStatus = { running: false, done: 0, total: 0, scope: [], results: [] }
|
||||
@@ -452,14 +436,14 @@ const normalizeTest = (st: GroupTestStatus): GroupTestStatus => ({
|
||||
})
|
||||
|
||||
/**
|
||||
* How the header names the reach of a running group test. The name matters more
|
||||
* How the header names the reach of a running exit test. The name matters more
|
||||
* than the number when there is only one: "auto" tells the operator which button
|
||||
* they pressed; "1 group" tells them nothing they didn't already know.
|
||||
* they pressed; "1 target" tells them nothing they didn't already know.
|
||||
*/
|
||||
function scopeLabel(scope: string[], groupCount: number): string {
|
||||
function scopeLabel(scope: string[], targetCount: number): string {
|
||||
if (scope.length === 1) return scope[0]
|
||||
if (scope.length === 0) return 'group exits' // pre-scope daemon — say nothing false
|
||||
return scope.length >= groupCount ? 'every group' : `${scope.length} groups`
|
||||
if (scope.length === 0) return 'exits' // pre-scope daemon — say nothing false
|
||||
return scope.length >= targetCount ? 'every exit' : `${scope.length} exits`
|
||||
}
|
||||
|
||||
/** Which editor (add or edit-by-name) is open within a section. */
|
||||
@@ -542,10 +526,11 @@ export default function Targets() {
|
||||
toastTimer.current = window.setTimeout(() => setToast(null), 2600)
|
||||
}, [])
|
||||
|
||||
// Our group-member health sweep + every health/testing control on this page. A
|
||||
// The daemon's observatory + every health/testing control on this page. A
|
||||
// group still picks a live node without it (sing-box probes internally); this
|
||||
// switch (Settings → Health check) governs only OUR sweep and the stats it
|
||||
// feeds. Absent ⇒ on. When off we stop polling and hide all of the health UI.
|
||||
// switch (Settings → Health check) gates only the observatory's background
|
||||
// probing of used groups/chains and the stats it feeds. Absent ⇒ on. When off
|
||||
// we stop polling and hide all of the health UI.
|
||||
const groupHealthOn = config?.Globals?.GroupHealth !== false
|
||||
|
||||
// ---- per-group membership health -------------------------------------------
|
||||
@@ -555,7 +540,6 @@ export default function Targets() {
|
||||
// opened — see GroupMembers.
|
||||
const [health, setHealth] = useState<GroupsHealth>(IDLE_HEALTH)
|
||||
const [healthErr, setHealthErr] = useState<string | null>(null)
|
||||
const [measuring, setMeasuring] = useState(false)
|
||||
// Has the endpoint ever answered? Until it has, a group with no health entry
|
||||
// means "we don't know", not "the engine hasn't built it" — and a row must not
|
||||
// blame the operator's config for our failed read.
|
||||
@@ -596,32 +580,19 @@ export default function Targets() {
|
||||
[health],
|
||||
)
|
||||
|
||||
/**
|
||||
* Measure every group member now (POST /api/nodes/test). This is the only thing
|
||||
* that can turn an untested member into a real dead verdict, and it covers each
|
||||
* group's own egress copies — not just the base outbounds — which is why the
|
||||
* control lives here beside the numbers it moves rather than on the Nodes page.
|
||||
*/
|
||||
const measureAll = useCallback(async () => {
|
||||
setMeasuring(true)
|
||||
try {
|
||||
const r = await postNodesTest()
|
||||
flash(r.started ? 'Measuring every group member…' : 'A measurement is already running')
|
||||
const h = await getGroupsHealth()
|
||||
setHealth({ ...h, groups: asArray(h.groups) })
|
||||
setHealthErr(null)
|
||||
setHealthRead(true)
|
||||
} catch (e) {
|
||||
flash(`Couldn’t start the measurement — ${errText(e)}`)
|
||||
} finally {
|
||||
setMeasuring(false)
|
||||
}
|
||||
}, [flash])
|
||||
// Per-chain reachability (used/unused), indexed by chain name — the chain
|
||||
// analogue of healthByGroup, for the "unused" badge on a chain card (plan §5.E).
|
||||
// `chains` is absent from the single-group (?group=) health shape and from a
|
||||
// daemon version that doesn't report it yet; treat both as "not known".
|
||||
const healthByChain = useMemo(
|
||||
() => new Map((health.chains ?? []).map((c) => [c.name, c])),
|
||||
[health],
|
||||
)
|
||||
|
||||
// ---- group test: how fast, through which node, out which address ------------
|
||||
// Same shape as the "Test all nodes" probe-all on the Nodes page: the POST only
|
||||
// kicks a run off, and a 2 s poll of the GET carries progress plus every result
|
||||
// so far. Reused deliberately rather than invented a second time.
|
||||
// ---- group/chain exit test: how fast, through which node, out which address --
|
||||
// The POST only kicks a run off, and a 2 s poll of the GET carries progress
|
||||
// plus every result so far. One endpoint covers groups and chains alike:
|
||||
// POST with a group or chain name tests that one; an empty name tests them all.
|
||||
const [gtest, setGtest] = useState<GroupTestStatus>(IDLE_TEST)
|
||||
const [gtestErr, setGtestErr] = useState<string | null>(null)
|
||||
const [polling, setPolling] = useState(false)
|
||||
@@ -680,7 +651,7 @@ export default function Targets() {
|
||||
if (r.started) {
|
||||
setGtestErr(null)
|
||||
setPolling(true)
|
||||
flash(name ? `Testing ${name}…` : 'Testing every group…')
|
||||
flash(name ? `Testing ${name}…` : 'Testing every exit…')
|
||||
void readTest()
|
||||
} else if (r.reason === 'already running') {
|
||||
setPolling(true) // pick up the run someone else started
|
||||
@@ -701,7 +672,7 @@ export default function Targets() {
|
||||
)
|
||||
|
||||
/**
|
||||
* The groups the current run covers. This — not `running` — is what puts the
|
||||
* The groups and chains the current run covers. This — not `running` — is what puts the
|
||||
* in-progress badge on a card: `running` alone only says a test is happening
|
||||
* SOMEWHERE, which is why testing one group used to light up all four.
|
||||
*
|
||||
@@ -770,10 +741,6 @@ export default function Targets() {
|
||||
const chainNames = useMemo(() => new Set(chains.map((c) => c.Name)), [chains])
|
||||
const egressNames = useMemo(() => new Set(egresses.map((e) => e.Name)), [egresses])
|
||||
|
||||
const globals = config?.Globals
|
||||
const defaultProbeURL = globals?.ProbeURL ?? ''
|
||||
const defaultProbeInterval = globals?.ProbeInterval ?? ''
|
||||
|
||||
// Hop picker options for chains: every group and node in the config.
|
||||
const hopOptions = useMemo<Opt[]>(
|
||||
() => [
|
||||
@@ -805,7 +772,8 @@ export default function Targets() {
|
||||
return save({ ...config, Groups: [...groups, { ...draft, Name: nm }] }, `Added ${nm}`)
|
||||
}
|
||||
// A rename must carry every reference with it (rule targets, chain hops,
|
||||
// resolver detours…) or those silently fall back to the default route.
|
||||
// resolver detours…) or those are orphaned — rules left pointing at nothing
|
||||
// have their traffic blocked.
|
||||
const carried = renameReferences(config, 'group', editing, draft.Name)
|
||||
const moved = editing === draft.Name ? 0 : findReferences(config, 'group', editing).length
|
||||
return save(
|
||||
@@ -925,70 +893,33 @@ export default function Targets() {
|
||||
<header className="tg-sec-hd">
|
||||
<h2 className="tg-sec-title">Groups</h2>
|
||||
<span className="tg-sec-count mono">{groups.length} configured</span>
|
||||
{/* Two runs live here and they are NOT the same action. A health
|
||||
refresh measures every node in every group at once — it is global by
|
||||
construction, so it gets exactly one indicator, here, and never a
|
||||
badge on a card. A group exit test is scoped to the groups it names,
|
||||
so its progress says WHICH, and its badge lands only on those cards. */}
|
||||
{/* The observatory's background probing is invisible by design — it
|
||||
keeps every used group's and chain's numbers fresh on its own. The
|
||||
one manual run left is the exit test: it is scoped to the groups
|
||||
and chains it names, so its progress says WHICH, and its badge
|
||||
lands only on those cards. */}
|
||||
<div className="tg-sec-ctl">
|
||||
{groupHealthOn && (
|
||||
<>
|
||||
{health.node_test_running && (
|
||||
<span
|
||||
className="tg-run tg-run--health"
|
||||
role="status"
|
||||
title="One health run measures every member of every group, including each group’s own egress copies. It covers all groups at once, so it is reported here and not on any single card."
|
||||
>
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-run-what">health · all groups</span>
|
||||
<span className="tg-run-n mono">
|
||||
{health.node_test_done}/{health.node_test_total}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
{gtest.running && (
|
||||
<span
|
||||
className="tg-run tg-run--exit"
|
||||
role="status"
|
||||
title="An exit test sends one connection through each group it covers and reports the delay and the address the internet sees."
|
||||
title="An exit test sends one connection through each group or chain it covers and reports the delay and the address the internet sees."
|
||||
>
|
||||
<Led variant="amber" pulse />
|
||||
<span className="tg-run-what">
|
||||
exit test · {scopeLabel(asArray(gtest.scope), groups.length)}
|
||||
exit test · {scopeLabel(asArray(gtest.scope), groups.length + chains.length)}
|
||||
</span>
|
||||
<span className="tg-run-n mono">
|
||||
{gtest.done}/{gtest.total}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
{!health.node_test_running && health.sweep?.enabled && health.sweep.total > 0 && (
|
||||
<span
|
||||
className="tg-run tg-run--sweep"
|
||||
title={
|
||||
health.sweep.cycles === 0
|
||||
? 'The router re-probes nodes in the background so these numbers stay fresh. It is still on its first pass, so members it hasn’t reached yet read “not measured”.'
|
||||
: `The router re-probes nodes in the background so these numbers stay fresh. It has completed ${health.sweep.cycles} full pass${health.sweep.cycles === 1 ? '' : 'es'}, so anything still unmeasured lost a reading it used to have.`
|
||||
}
|
||||
>
|
||||
<span className="tg-run-what">
|
||||
sweep{health.sweep.cycles === 0 ? ' · first pass' : ''}
|
||||
</span>
|
||||
<span className="tg-run-n mono">
|
||||
{health.sweep.cursor}/{health.sweep.total}
|
||||
</span>
|
||||
</span>
|
||||
)}
|
||||
<Button
|
||||
onClick={() => void measureAll()}
|
||||
disabled={measuring || health.node_test_running || groups.length === 0}
|
||||
title="Probe every member of every group now, including each group’s own egress copies. One run, all groups — health also refreshes on its own in the background."
|
||||
>
|
||||
{health.node_test_running ? 'Measuring…' : 'Refresh health'}
|
||||
</Button>
|
||||
<Button
|
||||
onClick={() => void runTest()}
|
||||
disabled={busy || !config || groups.length === 0 || gtest.running}
|
||||
title="Send one connection through each group and report the delay and the exit address the internet sees"
|
||||
disabled={busy || !config || (groups.length === 0 && chains.length === 0) || gtest.running}
|
||||
title="Send one connection through each group and chain and report the delay and the exit address the internet sees"
|
||||
>
|
||||
{gtest.running ? 'Testing…' : 'Test every exit'}
|
||||
</Button>
|
||||
@@ -1035,8 +966,6 @@ export default function Targets() {
|
||||
egresses={egresses}
|
||||
taken={groupNames}
|
||||
busy={busy}
|
||||
defaultProbeURL={defaultProbeURL}
|
||||
defaultProbeInterval={defaultProbeInterval}
|
||||
onCancel={() => setGroupEd(null)}
|
||||
onSave={async (g) => {
|
||||
const ok = await commitGroup(g, null)
|
||||
@@ -1066,8 +995,6 @@ export default function Targets() {
|
||||
egresses={egresses}
|
||||
taken={without(groupNames, g.Name)}
|
||||
busy={busy}
|
||||
defaultProbeURL={defaultProbeURL}
|
||||
defaultProbeInterval={defaultProbeInterval}
|
||||
onCancel={() => setGroupEd(null)}
|
||||
onSave={async (ng) => {
|
||||
const ok = await commitGroup(ng, g.Name)
|
||||
@@ -1084,8 +1011,6 @@ export default function Targets() {
|
||||
showHealth={groupHealthOn}
|
||||
health={healthByGroup.get(g.Name)}
|
||||
healthKnown={healthRead}
|
||||
sweep={health.sweep}
|
||||
onMeasure={() => void measureAll()}
|
||||
test={testByGroup.get(g.Name)}
|
||||
// The badge is this card's business only when the run names it.
|
||||
testing={gtest.running && testScope.has(g.Name)}
|
||||
@@ -1168,6 +1093,15 @@ export default function Targets() {
|
||||
key={c.Name}
|
||||
chain={c}
|
||||
busy={busy}
|
||||
showHealth={groupHealthOn}
|
||||
used={healthByChain.get(c.Name)?.used}
|
||||
test={testByGroup.get(c.Name)}
|
||||
// The badge is this card's business only when the run names it.
|
||||
testing={gtest.running && testScope.has(c.Name)}
|
||||
// …but the daemon runs one test at a time, so any run in flight
|
||||
// is what disables the button.
|
||||
testBusy={gtest.running}
|
||||
onTest={() => void runTest(c.Name)}
|
||||
onEdit={() => setChainEd({ mode: 'edit', name: c.Name })}
|
||||
onDelete={() => removeChain(c.Name)}
|
||||
/>
|
||||
@@ -1272,8 +1206,6 @@ function GroupRow({
|
||||
showHealth,
|
||||
health,
|
||||
healthKnown,
|
||||
sweep,
|
||||
onMeasure,
|
||||
test,
|
||||
testing,
|
||||
testBusy,
|
||||
@@ -1292,11 +1224,6 @@ function GroupRow({
|
||||
/** The health endpoint has answered at least once, so an absent entry really
|
||||
* does mean "the engine doesn't have this group". */
|
||||
healthKnown: boolean
|
||||
/** The daemon's background sweep, so the card can say whether untested members
|
||||
* resolve by themselves. Without it, "wait and it will fill in" is a false
|
||||
* promise. Absent on daemons that don't report it. */
|
||||
sweep?: HealthSweep
|
||||
onMeasure: () => void
|
||||
test?: GroupTestResult
|
||||
/**
|
||||
* A group exit test covering THIS group is in flight.
|
||||
@@ -1365,8 +1292,6 @@ function GroupRow({
|
||||
egress={group.Egress ?? ''}
|
||||
health={health}
|
||||
healthKnown={healthKnown}
|
||||
sweep={sweep}
|
||||
onMeasure={onMeasure}
|
||||
/>
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
</>
|
||||
@@ -1404,15 +1329,11 @@ function GroupHealthReadout({
|
||||
egress,
|
||||
health,
|
||||
healthKnown,
|
||||
sweep,
|
||||
onMeasure,
|
||||
}: {
|
||||
name: string
|
||||
egress: string
|
||||
health?: GroupHealth
|
||||
healthKnown: boolean
|
||||
sweep?: HealthSweep
|
||||
onMeasure: () => void
|
||||
}) {
|
||||
const [open, setOpen] = useState(false)
|
||||
const panelId = `gh-members-${name}`
|
||||
@@ -1438,10 +1359,27 @@ function GroupHealthReadout({
|
||||
)
|
||||
}
|
||||
|
||||
// A completed sweep pass is what makes silence meaningful; until then a group's
|
||||
// own history can only have recorded successes. See isBiasedSample.
|
||||
const sweptOnce = (sweep?.cycles ?? 0) >= 1
|
||||
const v = verdictOf(health, sweptOnce)
|
||||
// No enabled rule reaches this group, so the observatory never probes it and
|
||||
// its members would stay "untested" forever. That is a fact about the ROUTING
|
||||
// CONFIG, not about the members — so instead of counters that could only ever
|
||||
// read as a permanent unknown, the card says so, quietly: unused, not unwell.
|
||||
if (!health.used) {
|
||||
return (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this group, so its members are not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this group</span>
|
||||
</div>
|
||||
</div>
|
||||
)
|
||||
}
|
||||
|
||||
const v = verdictOf(health)
|
||||
const { total, tested, alive, dead, untested } = health
|
||||
const boundTitle = `Every member of “${name}” is a private copy dialled through ${
|
||||
egress ? `egress “${egress}”` : 'this group’s egress'
|
||||
@@ -1487,7 +1425,7 @@ function GroupHealthReadout({
|
||||
/* No fraction here, on purpose. `alive / tested` would put a denominator
|
||||
on a sample that nothing has yet been able to fail a member into, and
|
||||
it always reads 100%. Two plain counts instead: what is confirmed, and
|
||||
what is still open. When the sweep fills the gap this becomes a real
|
||||
what is still open. When the observatory fills the gap this becomes a real
|
||||
ratio — which reads as the panel getting more precise, not as the
|
||||
group getting worse. */
|
||||
<>
|
||||
@@ -1542,31 +1480,14 @@ function GroupHealthReadout({
|
||||
|
||||
{/* The three states that need a sentence rather than a number. */}
|
||||
{/* Why there is no ratio. It names the mechanism, because "we're not sure"
|
||||
without a reason reads as hedging — and because the mechanism is also the
|
||||
answer to "when will I know": either the sweep gets there, or you ask. */}
|
||||
without a reason reads as hedging — and because the mechanism is also
|
||||
the answer to "when will I know": the observatory's next pass, seconds
|
||||
away. */}
|
||||
{v === 'partial' && (
|
||||
<p className="gh-say">
|
||||
Not a proportion yet — a group records only the members that answer, so any that failed
|
||||
are still counted as unchecked.{' '}
|
||||
{sweep?.enabled ? (
|
||||
<>
|
||||
The background sweep is the only thing that confirms a member is down, and it hasn’t
|
||||
finished its first pass over this group. It gets there on its own, or{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>{' '}
|
||||
to settle it.
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
The background sweep is what confirms a member is down, and it is off — so nothing
|
||||
will settle this until you{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
)}
|
||||
are still counted as unchecked. The daemon’s background probing confirms failures too and
|
||||
settles this within seconds.
|
||||
</p>
|
||||
)}
|
||||
{v === 'down' && (
|
||||
@@ -1576,41 +1497,14 @@ function GroupHealthReadout({
|
||||
nowhere to go.
|
||||
</p>
|
||||
)}
|
||||
{/* Why nothing has been measured, and what will change that. The sweep's
|
||||
`cycles` is what separates the two honest readings of the same word:
|
||||
cycles 0 means the sweep simply hasn't arrived; cycles ≥ 1 means it HAS
|
||||
been everywhere, so these members lost the readings they once had. */}
|
||||
{/* Why nothing has been measured, and what will change that. For a used
|
||||
group the observatory gets here on its own — within seconds, not
|
||||
minutes — so the sentence promises exactly that and nothing more. */}
|
||||
{v === 'unmeasured' && (
|
||||
<p className="gh-say">
|
||||
Nothing has been probed through this group yet, so there is nothing to report — not a
|
||||
fault.{' '}
|
||||
{!sweep?.enabled ? (
|
||||
<>
|
||||
The background sweep is off, so this fills in only when you ask:{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
) : sweep.cycles === 0 ? (
|
||||
<>
|
||||
The background sweep is still on its first pass and gets to these within a few
|
||||
minutes, or{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
) : (
|
||||
<>
|
||||
The background sweep has already been everywhere, so these lost the readings they
|
||||
had — probe them to find out where they stand:{' '}
|
||||
<button type="button" className="linkish" onClick={onMeasure}>
|
||||
measure now
|
||||
</button>
|
||||
.
|
||||
</>
|
||||
)}
|
||||
fault. The observatory probes every group your rules use in the background, so these
|
||||
numbers fill in by themselves within seconds.
|
||||
</p>
|
||||
)}
|
||||
|
||||
@@ -1637,7 +1531,8 @@ function GroupHealthReadout({
|
||||
id={panelId}
|
||||
group={name}
|
||||
// Refetch when the cheap summary poll shows this group's counters moved
|
||||
// (a probe-all landing, say) — the open panel keeps no poll of its own.
|
||||
// (the observatory landing fresh verdicts, say) — the open panel keeps
|
||||
// no poll of its own.
|
||||
version={`${total}-${tested}-${alive}-${dead}`}
|
||||
/>
|
||||
)}
|
||||
@@ -1815,8 +1710,6 @@ function GroupEditor({
|
||||
egresses,
|
||||
taken,
|
||||
busy,
|
||||
defaultProbeURL,
|
||||
defaultProbeInterval,
|
||||
onCancel,
|
||||
onSave,
|
||||
}: {
|
||||
@@ -1829,8 +1722,6 @@ function GroupEditor({
|
||||
egresses: Egress[]
|
||||
taken: Set<string>
|
||||
busy: boolean
|
||||
defaultProbeURL: string
|
||||
defaultProbeInterval: string
|
||||
onCancel: () => void
|
||||
onSave: (g: Group) => Promise<boolean>
|
||||
}) {
|
||||
@@ -1848,8 +1739,6 @@ function GroupEditor({
|
||||
const [protos, setProtos] = useState<string[]>(asArray(initial?.FilterProto))
|
||||
const [country, setCountry] = useState(joinList(initial?.FilterCountry))
|
||||
const [dedup, setDedup] = useState(initial?.Dedup ?? false)
|
||||
const [probeURL, setProbeURL] = useState(initial?.ProbeURL ?? '')
|
||||
const [probeInterval, setProbeInterval] = useState(initial?.ProbeInterval ?? '')
|
||||
const [egress, setEgress] = useState(initial?.Egress ?? '')
|
||||
const [search, setSearch] = useState('')
|
||||
const [err, setErr] = useState<string | null>(null)
|
||||
@@ -1892,8 +1781,6 @@ function GroupEditor({
|
||||
Name: nm,
|
||||
Source: source,
|
||||
Strategy: strategy,
|
||||
ProbeURL: probeURL.trim() || undefined,
|
||||
ProbeInterval: probeInterval.trim() || undefined,
|
||||
// Always sent, '' for "not set" — the field name is `Egress`, matching
|
||||
// Node.Egress. PUT /api/config rejects the WHOLE model on an unknown key.
|
||||
Egress: egress,
|
||||
@@ -2195,34 +2082,6 @@ function GroupEditor({
|
||||
)}
|
||||
</div>
|
||||
|
||||
<div className="tg-ed-grid">
|
||||
<label className="tg-field">
|
||||
<span className="tg-flabel">Probe URL</span>
|
||||
<input
|
||||
className="tg-input"
|
||||
value={probeURL}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
inputMode="url"
|
||||
placeholder={defaultProbeURL || 'default'}
|
||||
onChange={(e) => setProbeURL(e.target.value)}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
<label className="tg-field">
|
||||
<span className="tg-flabel">Probe interval</span>
|
||||
<input
|
||||
className="tg-input"
|
||||
value={probeInterval}
|
||||
spellCheck={false}
|
||||
autoComplete="off"
|
||||
placeholder={defaultProbeInterval || 'default'}
|
||||
onChange={(e) => setProbeInterval(e.target.value)}
|
||||
disabled={busy}
|
||||
/>
|
||||
</label>
|
||||
</div>
|
||||
|
||||
<EditorFoot
|
||||
busy={busy}
|
||||
err={err}
|
||||
@@ -2238,11 +2097,33 @@ function GroupEditor({
|
||||
function ChainRow({
|
||||
chain,
|
||||
busy,
|
||||
showHealth,
|
||||
used,
|
||||
test,
|
||||
testing,
|
||||
testBusy,
|
||||
onTest,
|
||||
onEdit,
|
||||
onDelete,
|
||||
}: {
|
||||
chain: Chain
|
||||
busy: boolean
|
||||
/** Group health checks are on (Settings). When false, the card drops its
|
||||
* exit-test readout and Test button — it is config only. */
|
||||
showHealth: boolean
|
||||
/** This chain's reachability (GroupHealth.Used's chain analogue, plan §5.E).
|
||||
* undefined ⇒ the health endpoint hasn't reported this chain (not applied yet, or
|
||||
* a daemon version without chains): no badge. false ⇒ no enabled rule routes
|
||||
* through the chain, so the observatory never probes it and the card renders
|
||||
* "unused" instead of an exit-test readout. */
|
||||
used?: boolean
|
||||
test?: GroupTestResult
|
||||
/** An exit test covering THIS chain is in flight (the caller resolves it
|
||||
* against the run's scope, exactly as for a group card). */
|
||||
testing: boolean
|
||||
/** Any exit test is in flight; the daemon runs one at a time. */
|
||||
testBusy: boolean
|
||||
onTest: () => void
|
||||
onEdit: () => void
|
||||
onDelete: () => void
|
||||
}) {
|
||||
@@ -2281,6 +2162,30 @@ function ChainRow({
|
||||
</div>
|
||||
)}
|
||||
</div>
|
||||
{showHealth && (
|
||||
<>
|
||||
{/* A chain no enabled rule routes through is never probed (the
|
||||
observatory walks only reachable paths), so instead of an exit-test
|
||||
readout the card says so, quietly — the same "unused" pattern the
|
||||
group card uses (GroupHealthReadout), not a new design. `used` is
|
||||
undefined until the health endpoint reports this chain (or from a
|
||||
daemon version without chains): no badge then. */}
|
||||
{used === false && (
|
||||
<div className="gh gh--unused">
|
||||
<div className="gh-line">
|
||||
<span
|
||||
className="gh-unused"
|
||||
title="No enabled rule routes through this chain, so its exit is not probed. Add it to a rule to see health."
|
||||
>
|
||||
unused
|
||||
</span>
|
||||
<span className="gh-quiet">not probed — no enabled rule routes through this chain</span>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
<GroupTestReadout test={test} pending={testing && !test} />
|
||||
</>
|
||||
)}
|
||||
</div>
|
||||
<RowActions
|
||||
onEdit={onEdit}
|
||||
@@ -2288,6 +2193,9 @@ function ChainRow({
|
||||
busy={busy}
|
||||
editLabel={`Edit chain ${chain.Name}`}
|
||||
deleteLabel={`Delete chain ${chain.Name}`}
|
||||
onTest={showHealth ? onTest : undefined}
|
||||
testLabel={showHealth ? `Test the exit of chain ${chain.Name}` : undefined}
|
||||
testDisabled={testBusy}
|
||||
/>
|
||||
</li>
|
||||
)
|
||||
@@ -2767,7 +2675,7 @@ function RowActions({
|
||||
busy: boolean
|
||||
editLabel: string
|
||||
deleteLabel: string
|
||||
// Only groups can be tested today, so the control is optional and absent
|
||||
// Only groups and chains can be tested, so the control is optional and absent
|
||||
// everywhere else rather than a disabled stub on every row.
|
||||
onTest?: () => void
|
||||
testLabel?: string
|
||||
|
||||
+121
-113
@@ -93,6 +93,10 @@ func (s *URLTest) Start() error {
|
||||
}
|
||||
group.balancer = s.balancer // lx: SPEC 019 v2 — health-check drives the pool through it
|
||||
if s.balancer != nil {
|
||||
// lx: health board §5.B — slot liveness reads through the board verdict, so a
|
||||
// death recorded by any prober or a failed dial takes effect on the next pick,
|
||||
// not the next health-check tick.
|
||||
s.balancer.verdict = group.slotVerdict
|
||||
// lx: SPEC 020 — a pool rebuild changes the active routing tree; invalidate
|
||||
// the router's reachable cache. ctx captured here has the invalidator.
|
||||
ctx := s.ctx
|
||||
@@ -148,9 +152,10 @@ type PoolSlot struct {
|
||||
}
|
||||
|
||||
// Pool returns the current rotation pool (one entry per slot) for round_robin groups. For
|
||||
// least_test (nil balancer) it returns nil — "this group has no pool". Delay is read from
|
||||
// history and clamped 0->1 for live nodes so 0 in the output unambiguously means dead/untested.
|
||||
// lx: SPEC 019 v2 (exposed to clients via the GetPool RPC).
|
||||
// least_test (nil balancer) it returns nil — "this group has no pool". Delay is reported only
|
||||
// for slots with a fresh-alive board verdict, clamped 0->1, so 0 in the output unambiguously
|
||||
// means dead/untested (failures now persist in history — an entry alone no longer means alive).
|
||||
// lx: SPEC 019 v2 (exposed to clients via the GetPool RPC); health board §5.B.
|
||||
func (s *URLTest) Pool() []PoolSlot {
|
||||
if s.balancer == nil || s.group == nil {
|
||||
return nil
|
||||
@@ -167,10 +172,12 @@ func (s *URLTest) Pool() []PoolSlot {
|
||||
if node, loaded := s.outbound.Outbound(tag); loaded {
|
||||
historyTag = RealTag(node)
|
||||
}
|
||||
if history := s.group.history.LoadURLTestHistory(historyTag); history != nil {
|
||||
delay = history.Delay
|
||||
if delay == 0 {
|
||||
delay = 1 // live sub-ms node: never report 0 (0 is reserved for dead/untested)
|
||||
if s.group.history.Verdict(historyTag, s.group.healthTTL()) == urltest.VerdictAlive {
|
||||
if history := s.group.history.LoadURLTestHistory(historyTag); history != nil {
|
||||
delay = history.Delay
|
||||
if delay == 0 {
|
||||
delay = 1 // live sub-ms node: never report 0 (0 is reserved for dead/untested)
|
||||
}
|
||||
}
|
||||
}
|
||||
slots[i] = PoolSlot{Slot: i, Tag: tag, Delay: delay}
|
||||
@@ -187,10 +194,12 @@ func (s *URLTest) CheckOutbounds() {
|
||||
}
|
||||
|
||||
// selectBalanced picks an outbound per-connection in round_robin mode from the balancer's
|
||||
// fixed-size pool. lx: SPEC 019 v2. fallback (Select's outbounds[0]) covers the cold-start
|
||||
// window before the first health-check fills the pool. Returns nil only when nothing usable.
|
||||
func (s *URLTest) selectBalanced(ctx context.Context, network string, destination M.Socksaddr) adapter.Outbound {
|
||||
fallback, _ := s.group.Select(network)
|
||||
// fixed-size pool. lx: SPEC 019 v2. fallback (selectExcluding) covers the cold-start window
|
||||
// before the first health-check fills the pool and the all-slots-dead state. exclude carries
|
||||
// the members already tried by this connection's dial-retry loop (health board §5.B).
|
||||
// Returns nil only when nothing usable.
|
||||
func (s *URLTest) selectBalanced(ctx context.Context, network string, destination M.Socksaddr, exclude map[string]bool) adapter.Outbound {
|
||||
fallback, _ := s.group.selectExcluding(network, exclude)
|
||||
selected := s.balancer.pick(ctx, destination, fallback, func(tag string) adapter.Outbound {
|
||||
node, _ := s.outbound.Outbound(tag)
|
||||
if node != nil && !common.Contains(node.Network(), network) {
|
||||
@@ -206,68 +215,74 @@ func (s *URLTest) selectBalanced(ctx context.Context, network string, destinatio
|
||||
|
||||
func (s *URLTest) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
s.group.Touch()
|
||||
var outbound adapter.Outbound
|
||||
if s.balancer != nil {
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP, N.NetworkUDP:
|
||||
outbound = s.selectBalanced(ctx, network, destination)
|
||||
default:
|
||||
return nil, E.Extend(N.ErrUnknownNetwork, network)
|
||||
}
|
||||
} else {
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP:
|
||||
outbound = s.group.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
outbound = s.group.selectedOutboundUDP
|
||||
default:
|
||||
return nil, E.Extend(N.ErrUnknownNetwork, network)
|
||||
}
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP, N.NetworkUDP:
|
||||
default:
|
||||
return nil, E.Extend(N.ErrUnknownNetwork, network)
|
||||
}
|
||||
// lx: health board §5.B — a failed dial no longer fails the user's connection outright:
|
||||
// the member is marked dead on the board and the dial is retried through the next
|
||||
// candidate (at most dialAttemptsMax members, within the context deadline), for every
|
||||
// mode. The old behaviour deleted the history entry in least_test (making a dead member
|
||||
// indistinguishable from an untested one) and did nothing at all in balanced modes
|
||||
// (plan §2 Д1/Д5).
|
||||
var tried map[string]bool
|
||||
var lastErr error
|
||||
for range dialAttemptsMax {
|
||||
outbound := s.dialSelect(ctx, network, destination, tried)
|
||||
if outbound == nil {
|
||||
outbound, _ = s.group.Select(network)
|
||||
break
|
||||
}
|
||||
conn, err := outbound.DialContext(ctx, network, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
if tried == nil {
|
||||
tried = make(map[string]bool, dialAttemptsMax)
|
||||
}
|
||||
s.markDialFailure(outbound, tried, err)
|
||||
lastErr = err
|
||||
if ctx.Err() != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
if outbound == nil {
|
||||
return nil, E.New("missing supported outbound")
|
||||
if lastErr != nil {
|
||||
return nil, lastErr
|
||||
}
|
||||
conn, err := outbound.DialContext(ctx, network, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
// lx: SPEC 019 v2 — in round_robin a dial error must NOT touch the pool: the cause is
|
||||
// unknown (dead node vs. dead destination vs. local network drop). Only the health-check
|
||||
// changes pool membership. least_test keeps the upstream behaviour (drop the history).
|
||||
if s.balancer == nil {
|
||||
s.group.history.DeleteURLTestHistory(outbound.Tag())
|
||||
}
|
||||
return nil, err
|
||||
return nil, E.New("missing supported outbound")
|
||||
}
|
||||
|
||||
func (s *URLTest) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
s.group.Touch()
|
||||
var outbound adapter.Outbound
|
||||
if s.balancer != nil {
|
||||
outbound = s.selectBalanced(ctx, N.NetworkUDP, destination)
|
||||
} else {
|
||||
outbound = s.group.selectedOutboundUDP
|
||||
// lx: health board §5.B — same mark-fail + re-pick retry as DialContext, but only for
|
||||
// the ListenPacket call itself: once the packet conn is returned no retry is possible
|
||||
// (the UDP session is bound to its member from the first send).
|
||||
var tried map[string]bool
|
||||
var lastErr error
|
||||
for range dialAttemptsMax {
|
||||
outbound := s.dialSelect(ctx, N.NetworkUDP, destination, tried)
|
||||
if outbound == nil {
|
||||
outbound, _ = s.group.Select(N.NetworkUDP)
|
||||
break
|
||||
}
|
||||
conn, err := outbound.ListenPacket(ctx, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewPacketConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
if tried == nil {
|
||||
tried = make(map[string]bool, dialAttemptsMax)
|
||||
}
|
||||
s.markDialFailure(outbound, tried, err)
|
||||
lastErr = err
|
||||
if ctx.Err() != nil {
|
||||
break
|
||||
}
|
||||
}
|
||||
if outbound == nil {
|
||||
return nil, E.New("missing supported outbound")
|
||||
if lastErr != nil {
|
||||
return nil, lastErr
|
||||
}
|
||||
conn, err := outbound.ListenPacket(ctx, destination)
|
||||
if err == nil {
|
||||
return s.group.interruptGroup.NewPacketConn(conn, interrupt.IsExternalConnectionFromContext(ctx)), nil
|
||||
}
|
||||
s.logger.ErrorContext(ctx, err)
|
||||
// lx: SPEC 019 v2 — round_robin dial error leaves the pool untouched (see DialContext).
|
||||
if s.balancer == nil {
|
||||
s.group.history.DeleteURLTestHistory(outbound.Tag())
|
||||
}
|
||||
return nil, err
|
||||
return nil, E.New("missing supported outbound")
|
||||
}
|
||||
|
||||
func (s *URLTest) NewConnection(ctx context.Context, conn net.Conn, metadata adapter.InboundContext, onClose N.CloseHandlerFunc) {
|
||||
@@ -382,47 +397,12 @@ func (g *URLTestGroup) Close() error {
|
||||
}
|
||||
|
||||
func (g *URLTestGroup) Select(network string) (adapter.Outbound, bool) {
|
||||
var minDelay uint16
|
||||
var minOutbound adapter.Outbound
|
||||
switch network {
|
||||
case N.NetworkTCP:
|
||||
if g.selectedOutboundTCP != nil {
|
||||
if history := g.history.LoadURLTestHistory(RealTag(g.selectedOutboundTCP)); history != nil {
|
||||
minOutbound = g.selectedOutboundTCP
|
||||
minDelay = history.Delay
|
||||
}
|
||||
}
|
||||
case N.NetworkUDP:
|
||||
if g.selectedOutboundUDP != nil {
|
||||
if history := g.history.LoadURLTestHistory(RealTag(g.selectedOutboundUDP)); history != nil {
|
||||
minOutbound = g.selectedOutboundUDP
|
||||
minDelay = history.Delay
|
||||
}
|
||||
}
|
||||
}
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
history := g.history.LoadURLTestHistory(RealTag(detour))
|
||||
if history == nil {
|
||||
continue
|
||||
}
|
||||
if minDelay == 0 || minDelay > history.Delay+g.tolerance {
|
||||
minDelay = history.Delay
|
||||
minOutbound = detour
|
||||
}
|
||||
}
|
||||
if minOutbound == nil {
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
return detour, false
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
return minOutbound, true
|
||||
// lx: health board §5.B — selection reads the board verdict instead of "has a history
|
||||
// entry": fresh-alive members ranked by delay first, untested members in config order
|
||||
// second, first-by-config last — a dead member is never picked while a live or untested
|
||||
// one exists. The logic lives in selectExcluding (urltest_health_lx.go) so the
|
||||
// dial-retry path can re-run it minus the members that just failed.
|
||||
return g.selectExcluding(network, nil)
|
||||
}
|
||||
|
||||
func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{}) {
|
||||
@@ -451,6 +431,16 @@ func (g *URLTestGroup) loopCheck(ticker *time.Ticker, closeChan <-chan struct{})
|
||||
}
|
||||
}
|
||||
|
||||
// lx: health board §5.B/§5.C — CheckOutbounds is the FAST loop of the pair: an
|
||||
// ACTIVE group probes its own members on its own ticker (interval = the global
|
||||
// probe interval; failover keeps its 30s default), and a failed probe writes
|
||||
// MarkFailed to the shared health board (common/urltest) instead of deleting
|
||||
// the entry. The shater observatory is the complementary BACKGROUND loop: it
|
||||
// probes what no active group measures (idle groups' members, selector
|
||||
// candidates, chain exits), and its freshness gate skips any tag this loop
|
||||
// keeps current — no duplicate probes, and this loop's cadence is never
|
||||
// suppressed. Both loops write to the SAME board, so selection and the panel
|
||||
// see one source of truth no matter which prober found the death.
|
||||
func (g *URLTestGroup) CheckOutbounds(force bool) {
|
||||
_, _ = g.urlTest(g.ctx, force)
|
||||
}
|
||||
@@ -481,8 +471,10 @@ func (g *URLTestGroup) urlTest(ctx context.Context, force bool) (map[string]uint
|
||||
}
|
||||
|
||||
// testNodes runs the URL test over the given outbounds (skipping fresh history unless force),
|
||||
// stores/deletes history, and returns tag->delay for the live ones. lx: shared by least_test
|
||||
// and the round_robin force path.
|
||||
// stores successes and marks failures on the health board, and returns tag->delay for the live
|
||||
// ones. lx: shared by least_test and the round_robin force path; this is the probe
|
||||
// primitive of the fast active-group loop (see CheckOutbounds for the two-loop
|
||||
// contract with the shater observatory).
|
||||
func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbound, force bool) map[string]uint16 {
|
||||
result := make(map[string]uint16)
|
||||
b, _ := batch.New(ctx, batch.WithConcurrencyNum[any](10))
|
||||
@@ -495,7 +487,7 @@ func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbou
|
||||
continue
|
||||
}
|
||||
history := g.history.LoadURLTestHistory(realTag)
|
||||
if !force && history != nil && time.Since(history.Time) < g.interval {
|
||||
if !force && history != nil && time.Since(history.LastOK) < g.interval { // lx: health board §5.A — Time renamed to LastOK
|
||||
continue
|
||||
}
|
||||
checked[realTag] = true
|
||||
@@ -509,12 +501,18 @@ func (g *URLTestGroup) testNodes(ctx context.Context, outbounds []adapter.Outbou
|
||||
t, err := urltest.URLTest(testCtx, g.link, p)
|
||||
if err != nil {
|
||||
g.logger.Debug("outbound ", tag, " unavailable: ", err)
|
||||
g.history.DeleteURLTestHistory(realTag)
|
||||
// lx: health board §5.A — a failed check marks the entry instead of deleting
|
||||
// it (deletion stays reserved for members removed from the configuration).
|
||||
g.markFailedLogged(realTag, tag, "probe", err)
|
||||
} else {
|
||||
g.logger.Debug("outbound ", tag, " available: ", t, "ms")
|
||||
// lx: health board — log the dead -> alive flip before the success overwrites it.
|
||||
if g.history.Verdict(realTag, g.healthTTL()) == urltest.VerdictDead {
|
||||
g.logger.Info("outbound ", tag, " flipped dead -> alive (probe: ", t, "ms)")
|
||||
}
|
||||
g.history.StoreURLTestHistory(realTag, &adapter.URLTestHistory{
|
||||
Time: time.Now(),
|
||||
Delay: t,
|
||||
LastOK: time.Now(), // lx: health board §5.A — Time renamed to LastOK
|
||||
Delay: t,
|
||||
})
|
||||
resultAccess.Lock()
|
||||
result[tag] = t
|
||||
@@ -757,7 +755,11 @@ func (g *URLTestGroup) rebuildPool() {
|
||||
results := make(map[string]candidate, len(g.outbounds))
|
||||
for _, detour := range g.outbounds {
|
||||
tag := detour.Tag()
|
||||
if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
realTag := RealTag(detour)
|
||||
// lx: health board §5.B — alive = fresh board verdict, not entry presence
|
||||
// (failures persist in history now).
|
||||
if g.history.Verdict(realTag, g.healthTTL()) == urltest.VerdictAlive {
|
||||
history := g.history.LoadURLTestHistory(realTag)
|
||||
results[tag] = candidate{tag: tag, delay: history.Delay, alive: true}
|
||||
} else {
|
||||
results[tag] = candidate{tag: tag, alive: false}
|
||||
@@ -819,8 +821,10 @@ func (g *URLTestGroup) rebuildPool() {
|
||||
g.balancer.setSlots(planTolerantPool(current, results, size, g.balancer.poolTolerance), tolerantLive)
|
||||
}
|
||||
|
||||
// seedPool fills the pool before the first health-check: prefer nodes with live history (the
|
||||
// process was not unloaded), else the first `size` nodes in config order. lx: SPEC 019 v2.
|
||||
// seedPool fills the pool before the first health-check: prefer nodes the board holds a
|
||||
// fresh-alive verdict for (the process was not unloaded), else the first `size` nodes in
|
||||
// config order. lx: SPEC 019 v2; health board §5.B — a persisted failure record must not
|
||||
// look like a warm node.
|
||||
func (g *URLTestGroup) seedPool() {
|
||||
if g.balancer == nil {
|
||||
return
|
||||
@@ -829,10 +833,14 @@ func (g *URLTestGroup) seedPool() {
|
||||
if size == 0 {
|
||||
return
|
||||
}
|
||||
// Nodes with existing history first (top by delay), then config order to fill.
|
||||
// Fresh-alive nodes first (top by delay), then config order to fill.
|
||||
withHistory := make([]candidate, 0, len(g.outbounds))
|
||||
for _, detour := range g.outbounds {
|
||||
if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
realTag := RealTag(detour)
|
||||
if g.history.Verdict(realTag, g.healthTTL()) != urltest.VerdictAlive {
|
||||
continue
|
||||
}
|
||||
if history := g.history.LoadURLTestHistory(realTag); history != nil {
|
||||
withHistory = append(withHistory, candidate{tag: detour.Tag(), delay: history.Delay, alive: true})
|
||||
}
|
||||
}
|
||||
|
||||
@@ -9,6 +9,7 @@ import (
|
||||
"sync/atomic"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
E "github.com/sagernet/sing/common/exceptions"
|
||||
@@ -46,6 +47,14 @@ type balancer struct {
|
||||
random bool // mode == random: pick a uniformly-random live slot per connection
|
||||
priority bool // priority == config order is the ranking (failover); enables fail-back
|
||||
|
||||
// verdict, if set, reads the health-board verdict for a slot tag (wired to
|
||||
// URLTestGroup.slotVerdict). Slot liveness is the verdict when the board has fresh
|
||||
// data — alive forces live, dead forces dead — and the health-check flag otherwise
|
||||
// (untested keeps the optimistic seed usable on cold start). This is what makes a
|
||||
// death recorded by ANY prober (group checker, observatory, failed dial) take effect
|
||||
// on the next pick instead of the next tick. Health board plan §5.B.
|
||||
verdict func(tag string) urltest.HealthVerdict
|
||||
|
||||
access sync.Mutex // guards slots
|
||||
slots []slot // the pool; len == min(poolSize, available nodes), index = fixed slot number
|
||||
counter atomic.Uint64
|
||||
@@ -133,7 +142,7 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
|
||||
}
|
||||
liveCount := 0
|
||||
for i := range b.slots {
|
||||
if b.slots[i].live {
|
||||
if b.slotIsLive(i) {
|
||||
liveCount++
|
||||
}
|
||||
}
|
||||
@@ -155,7 +164,7 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
|
||||
// picked deterministically from the same key — the flow still lands somewhere working.
|
||||
h := hashKey(b.stickyKey(ctx, destination))
|
||||
idx := int(h % uint64(n))
|
||||
if b.slots[idx].live {
|
||||
if b.slotIsLive(idx) {
|
||||
tag = b.slots[idx].tag
|
||||
} else {
|
||||
tag = b.nthLiveTag(int(h % uint64(liveCount)))
|
||||
@@ -172,11 +181,30 @@ func (b *balancer) pick(ctx context.Context, destination M.Socksaddr, fallback a
|
||||
return fallback
|
||||
}
|
||||
|
||||
// slotIsLive reports whether slot i is selectable: the board verdict wins when fresh
|
||||
// (alive → live, dead → dead), the last health-check flag decides for untested slots.
|
||||
// Caller must hold access. Health board plan §5.B.
|
||||
func (b *balancer) slotIsLive(i int) bool {
|
||||
s := b.slots[i]
|
||||
if s.tag == "" {
|
||||
return false
|
||||
}
|
||||
if b.verdict != nil {
|
||||
switch b.verdict(s.tag) {
|
||||
case urltest.VerdictAlive:
|
||||
return true
|
||||
case urltest.VerdictDead:
|
||||
return false
|
||||
}
|
||||
}
|
||||
return s.live
|
||||
}
|
||||
|
||||
// nthLiveTag returns the tag of the k-th live slot (0-based, in slot order). Caller must hold
|
||||
// access and pass k in [0, liveCount). Walks without allocating (pools may be large for random).
|
||||
func (b *balancer) nthLiveTag(k int) string {
|
||||
for i := range b.slots {
|
||||
if !b.slots[i].live {
|
||||
if !b.slotIsLive(i) {
|
||||
continue
|
||||
}
|
||||
if k == 0 {
|
||||
|
||||
@@ -0,0 +1,173 @@
|
||||
// lx:begin health-board
|
||||
|
||||
// Health-board driven selection and dial retry for the urltest group (plan §5.B).
|
||||
//
|
||||
// The group used to equate "has a history entry" with "alive": failures deleted the
|
||||
// entry, so a dead member was indistinguishable from a never-measured one (plan §2 Д5)
|
||||
// and a stale success counted as alive forever (Д2). With failures now recorded on the
|
||||
// board (common/urltest MarkFailed), liveness is a read-time verdict with a TTL derived
|
||||
// from the group's check interval. Selection prefers fresh-alive members ranked by
|
||||
// delay, falls back to untested members in config order, and only then to the first
|
||||
// member by config; a failed user dial marks the member dead on the board and retries
|
||||
// the connection through the next candidate instead of failing outright (Д1).
|
||||
|
||||
package group
|
||||
|
||||
import (
|
||||
"context"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
"github.com/sagernet/sing/common"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
)
|
||||
|
||||
// dialAttemptsMax bounds how many distinct members one connection may try: the first
|
||||
// pick plus up to two re-picks after a mark-fail. The context deadline still applies to
|
||||
// every attempt, so a short dial timeout cuts the sequence earlier.
|
||||
const dialAttemptsMax = 3
|
||||
|
||||
// healthTTLFloor keeps verdicts meaningful for groups with short check intervals: a
|
||||
// single missed tick must not flip a member to untested.
|
||||
const healthTTLFloor = 10 * time.Minute
|
||||
|
||||
// healthTTL is the freshness window for board verdicts: three check intervals (a member
|
||||
// that missed several consecutive checks is stale), never below healthTTLFloor.
|
||||
func (g *URLTestGroup) healthTTL() time.Duration {
|
||||
ttl := 3 * g.interval
|
||||
if ttl < healthTTLFloor {
|
||||
ttl = healthTTLFloor
|
||||
}
|
||||
return ttl
|
||||
}
|
||||
|
||||
// markFailedLogged records a failure for realTag on the board (the entry is kept — plan
|
||||
// §2 Д5) and logs the alive → dead verdict flip with its cause (probe or dial), the
|
||||
// single diagnostic trail for locating dead members and spotting flapping.
|
||||
func (g *URLTestGroup) markFailedLogged(realTag string, displayTag string, reason string, cause error) {
|
||||
previous := g.history.Verdict(realTag, g.healthTTL())
|
||||
g.history.MarkFailed(realTag)
|
||||
if previous == urltest.VerdictAlive {
|
||||
g.logger.Info("outbound ", displayTag, " flipped alive -> dead (", reason, ": ", cause, ")")
|
||||
}
|
||||
}
|
||||
|
||||
// slotVerdict reports the board verdict for a balancer slot tag. Slots hold member tags
|
||||
// as configured while the board is keyed by RealTag (a nested group's live leaf), so
|
||||
// resolve through the outbound manager first — same discipline as Pool().
|
||||
func (g *URLTestGroup) slotVerdict(tag string) urltest.HealthVerdict {
|
||||
historyTag := tag
|
||||
if node, loaded := g.outbound.Outbound(tag); loaded {
|
||||
historyTag = RealTag(node)
|
||||
}
|
||||
return g.history.Verdict(historyTag, g.healthTTL())
|
||||
}
|
||||
|
||||
// selectExcluding is Select with an exclusion set (RealTag keys) for the dial-retry
|
||||
// path: a member that just failed a dial is marked dead on the board AND excluded here,
|
||||
// so even the last-resort config-order fallback cannot re-pick it.
|
||||
func (g *URLTestGroup) selectExcluding(network string, exclude map[string]bool) (adapter.Outbound, bool) {
|
||||
ttl := g.healthTTL()
|
||||
var minDelay uint16
|
||||
var minOutbound adapter.Outbound
|
||||
// Keep the upstream hysteresis: the currently selected outbound only yields to a
|
||||
// member faster by more than tolerance — but only while it is still alive itself.
|
||||
var current adapter.Outbound
|
||||
switch network {
|
||||
case N.NetworkTCP:
|
||||
current = g.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
current = g.selectedOutboundUDP
|
||||
}
|
||||
if current != nil {
|
||||
currentTag := RealTag(current)
|
||||
if !exclude[currentTag] && g.history.Verdict(currentTag, ttl) == urltest.VerdictAlive {
|
||||
if history := g.history.LoadURLTestHistory(currentTag); history != nil {
|
||||
minOutbound = current
|
||||
minDelay = history.Delay
|
||||
}
|
||||
}
|
||||
}
|
||||
var firstUntested adapter.Outbound
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
realTag := RealTag(detour)
|
||||
if exclude[realTag] {
|
||||
continue
|
||||
}
|
||||
switch g.history.Verdict(realTag, ttl) {
|
||||
case urltest.VerdictAlive:
|
||||
history := g.history.LoadURLTestHistory(realTag)
|
||||
if history == nil {
|
||||
continue
|
||||
}
|
||||
if minDelay == 0 || minDelay > history.Delay+g.tolerance {
|
||||
minDelay = history.Delay
|
||||
minOutbound = detour
|
||||
}
|
||||
case urltest.VerdictUntested:
|
||||
if firstUntested == nil {
|
||||
firstUntested = detour
|
||||
}
|
||||
}
|
||||
}
|
||||
if minOutbound != nil {
|
||||
return minOutbound, true
|
||||
}
|
||||
// No fresh-alive member: an untested one (config order) is a better bet than a
|
||||
// known-dead one. When every member is dead, fall back to config order so the group
|
||||
// still dials something — the retry loop walks further members on failure.
|
||||
if firstUntested != nil {
|
||||
return firstUntested, false
|
||||
}
|
||||
for _, detour := range g.outbounds {
|
||||
if !common.Contains(detour.Network(), network) {
|
||||
continue
|
||||
}
|
||||
if exclude[RealTag(detour)] {
|
||||
continue
|
||||
}
|
||||
return detour, false
|
||||
}
|
||||
return nil, false
|
||||
}
|
||||
|
||||
// dialSelect picks the member for one dial attempt, skipping members this connection has
|
||||
// already tried (and marked dead). In least_test mode the cached selection is used only
|
||||
// while its verdict is not dead — a member the board knows to be down is re-selected
|
||||
// around immediately instead of waiting for the next checker tick.
|
||||
func (s *URLTest) dialSelect(ctx context.Context, network string, destination M.Socksaddr, tried map[string]bool) adapter.Outbound {
|
||||
if s.balancer != nil {
|
||||
return s.selectBalanced(ctx, network, destination, tried)
|
||||
}
|
||||
var outbound adapter.Outbound
|
||||
switch N.NetworkName(network) {
|
||||
case N.NetworkTCP:
|
||||
outbound = s.group.selectedOutboundTCP
|
||||
case N.NetworkUDP:
|
||||
outbound = s.group.selectedOutboundUDP
|
||||
}
|
||||
if outbound != nil {
|
||||
realTag := RealTag(outbound)
|
||||
if !tried[realTag] && s.group.history.Verdict(realTag, s.group.healthTTL()) != urltest.VerdictDead {
|
||||
return outbound
|
||||
}
|
||||
}
|
||||
outbound, _ = s.group.selectExcluding(network, tried)
|
||||
return outbound
|
||||
}
|
||||
|
||||
// markDialFailure records a failed dial: the member is marked dead on the board — every
|
||||
// reader (this group's next pick, the balancer slots via slotVerdict, the panel) sees it
|
||||
// immediately — and added to the connection's tried set so the retry never re-picks it.
|
||||
func (s *URLTest) markDialFailure(outbound adapter.Outbound, tried map[string]bool, err error) {
|
||||
realTag := RealTag(outbound)
|
||||
s.group.markFailedLogged(realTag, outbound.Tag(), "dial", err)
|
||||
tried[realTag] = true
|
||||
}
|
||||
|
||||
// lx:end health-board
|
||||
@@ -0,0 +1,505 @@
|
||||
package group
|
||||
|
||||
// lx: health board §5.B tests — verdict-driven selection, slot liveness read-through,
|
||||
// and the dial-retry path, per mode × {alive, dead, untested, stale}.
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/interrupt"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
)
|
||||
|
||||
// --- fixtures -----------------------------------------------------------------------
|
||||
|
||||
// healthNode is a dialable fake outbound: DialContext/ListenPacket succeed or fail per
|
||||
// the fail flag and record the dial order into dialed.
|
||||
type healthNode struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
fail bool
|
||||
dialed *[]string
|
||||
}
|
||||
|
||||
func (n *healthNode) Tag() string { return n.tag }
|
||||
func (n *healthNode) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||
|
||||
func (n *healthNode) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
}
|
||||
if n.fail {
|
||||
return nil, errors.New("dial refused")
|
||||
}
|
||||
left, right := net.Pipe()
|
||||
_ = right.Close()
|
||||
return left, nil
|
||||
}
|
||||
|
||||
func (n *healthNode) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
}
|
||||
if n.fail {
|
||||
return nil, errors.New("listen refused")
|
||||
}
|
||||
return net.ListenPacket("udp", "127.0.0.1:0")
|
||||
}
|
||||
|
||||
// fakeOutboundManager resolves tags over a fixed node set (only Outbound is used).
|
||||
type fakeOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
nodes map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *fakeOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
node, ok := m.nodes[tag]
|
||||
return node, ok
|
||||
}
|
||||
|
||||
func managerOf(nodes ...adapter.Outbound) *fakeOutboundManager {
|
||||
m := &fakeOutboundManager{nodes: make(map[string]adapter.Outbound, len(nodes))}
|
||||
for _, node := range nodes {
|
||||
m.nodes[node.Tag()] = node
|
||||
}
|
||||
return m
|
||||
}
|
||||
|
||||
// healthTestGroup builds a minimal URLTestGroup (not started: no ticker, no checker).
|
||||
// interval 3m → healthTTL = 10m floor.
|
||||
func healthTestGroup(hist *urltest.HistoryStorage, manager adapter.OutboundManager, nodes ...adapter.Outbound) *URLTestGroup {
|
||||
return &URLTestGroup{
|
||||
ctx: context.Background(),
|
||||
outbound: manager,
|
||||
outbounds: nodes,
|
||||
history: hist,
|
||||
interval: 3 * time.Minute,
|
||||
tolerance: 50,
|
||||
logger: log.NewNOPFactory().Logger(),
|
||||
interruptGroup: interrupt.NewGroup(),
|
||||
}
|
||||
}
|
||||
|
||||
// healthURLTest wraps a group into a dialable *URLTest, wiring the balancer the same way
|
||||
// Start() does (slot liveness through the board verdict).
|
||||
func healthURLTest(g *URLTestGroup, bal *balancer, manager adapter.OutboundManager) *URLTest {
|
||||
g.balancer = bal
|
||||
if bal != nil {
|
||||
bal.verdict = g.slotVerdict
|
||||
}
|
||||
return &URLTest{
|
||||
group: g,
|
||||
balancer: bal,
|
||||
outbound: manager,
|
||||
logger: log.NewNOPFactory().Logger(),
|
||||
}
|
||||
}
|
||||
|
||||
// storeAlive records a fresh success 1s in the past: still well inside the TTL, but
|
||||
// strictly older than any failure the test triggers afterwards (Windows' coarse
|
||||
// monotonic clock can otherwise produce LastOK == LastFail ties within one tick).
|
||||
func storeAlive(hist *urltest.HistoryStorage, tag string, delay uint16) {
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: time.Now().Add(-time.Second), Delay: delay})
|
||||
}
|
||||
|
||||
// storeStale records a success far older than the TTL: verdict must degrade to untested.
|
||||
func storeStale(hist *urltest.HistoryStorage, tag string, delay uint16) {
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: time.Now().Add(-time.Hour), Delay: delay})
|
||||
}
|
||||
|
||||
// --- healthTTL ----------------------------------------------------------------------
|
||||
|
||||
func TestHealthTTLFloor(t *testing.T) {
|
||||
g := &URLTestGroup{interval: 3 * time.Minute}
|
||||
if ttl := g.healthTTL(); ttl != 10*time.Minute {
|
||||
t.Fatalf("healthTTL(3m) = %v, want the 10m floor", ttl)
|
||||
}
|
||||
g.interval = 5 * time.Minute
|
||||
if ttl := g.healthTTL(); ttl != 15*time.Minute {
|
||||
t.Fatalf("healthTTL(5m) = %v, want 3×interval = 15m", ttl)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Select (least_test): alive > untested > config order ---------------------------
|
||||
|
||||
func TestSelectPrefersAliveOverDeadAndUntested(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b, c := &balNode{tag: "a"}, &balNode{tag: "b"}, &balNode{tag: "c"}
|
||||
g := healthTestGroup(hist, nil, a, b, c)
|
||||
// a: dead but with a LOWER recorded delay than b (the old "has an entry" logic
|
||||
// would have picked it); b: alive; c: untested.
|
||||
storeAlive(hist, "a", 10)
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 100)
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v (exists %v), want alive b", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectRanksAliveByDelay(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
storeAlive(hist, "a", 200)
|
||||
storeAlive(hist, "b", 50) // 200 > 50+tolerance(50) → b wins
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v (exists %v), want faster alive b", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectHysteresisKeepsAliveCurrent(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
g.selectedOutboundTCP = a
|
||||
storeAlive(hist, "a", 100)
|
||||
storeAlive(hist, "b", 60) // within tolerance (100 ≤ 60+50) → keep a
|
||||
if selected, _ := g.Select(N.NetworkTCP); selected != adapter.Outbound(a) {
|
||||
t.Fatalf("Select = %v, want current a kept within tolerance", selected)
|
||||
}
|
||||
storeAlive(hist, "b", 40) // beyond tolerance (100 > 40+50) → switch to b
|
||||
if selected, _ := g.Select(N.NetworkTCP); selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v, want b beyond tolerance", selected)
|
||||
}
|
||||
}
|
||||
|
||||
// A dead current selection must NOT seed the hysteresis: any alive member wins.
|
||||
func TestSelectDeadCurrentLosesToAlive(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
g.selectedOutboundTCP = a
|
||||
storeAlive(hist, "a", 10)
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 500)
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(b) {
|
||||
t.Fatalf("Select = %v (exists %v), want alive b over dead current a", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectFallsBackToUntestedInConfigOrder(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b, c, d := &balNode{tag: "a"}, &balNode{tag: "b"}, &balNode{tag: "c"}, &balNode{tag: "d"}
|
||||
g := healthTestGroup(hist, nil, a, b, c, d)
|
||||
hist.MarkFailed("a")
|
||||
hist.MarkFailed("b")
|
||||
// c and d untested → first by config order (c), not exists.
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if exists || selected != adapter.Outbound(c) {
|
||||
t.Fatalf("Select = %v (exists %v), want first untested c", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectAllDeadFallsBackToConfigOrder(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
hist.MarkFailed("a")
|
||||
hist.MarkFailed("b")
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if exists || selected != adapter.Outbound(a) {
|
||||
t.Fatalf("Select = %v (exists %v), want first-by-config a", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectStaleSuccessCountsAsUntested(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
storeStale(hist, "a", 20) // success older than TTL → untested tier
|
||||
hist.MarkFailed("b")
|
||||
selected, exists := g.Select(N.NetworkTCP)
|
||||
if exists || selected != adapter.Outbound(a) {
|
||||
t.Fatalf("Select = %v (exists %v), want stale a as untested over dead b", selected, exists)
|
||||
}
|
||||
// A fresh-alive member still beats the stale one.
|
||||
c := &balNode{tag: "c"}
|
||||
g.outbounds = append(g.outbounds, c)
|
||||
storeAlive(hist, "c", 300)
|
||||
selected, exists = g.Select(N.NetworkTCP)
|
||||
if !exists || selected != adapter.Outbound(c) {
|
||||
t.Fatalf("Select = %v (exists %v), want fresh alive c over stale a", selected, exists)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSelectExcludingSkipsTriedMembers(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b := &balNode{tag: "a"}, &balNode{tag: "b"}
|
||||
g := healthTestGroup(hist, nil, a, b)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 20)
|
||||
selected, _ := g.selectExcluding(N.NetworkTCP, map[string]bool{"a": true})
|
||||
if selected != adapter.Outbound(b) {
|
||||
t.Fatalf("selectExcluding = %v, want b with a excluded", selected)
|
||||
}
|
||||
// Every member excluded → nothing to dial.
|
||||
selected, _ = g.selectExcluding(N.NetworkTCP, map[string]bool{"a": true, "b": true})
|
||||
if selected != nil {
|
||||
t.Fatalf("selectExcluding = %v, want nil with all excluded", selected)
|
||||
}
|
||||
}
|
||||
|
||||
// --- slot liveness reads the board verdict (RR / random / failover) -----------------
|
||||
|
||||
func TestSlotLivenessVerdictOverridesFlag(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
b := rrBalancer(t, 2, []string{"none"})
|
||||
b.verdict = func(tag string) urltest.HealthVerdict { return hist.Verdict(tag, 10*time.Minute) }
|
||||
_, resolve := resolveFrom("a", "b")
|
||||
// Both slots flagged live by the last check, but the board learned "a" died since
|
||||
// (observatory or a failed dial) — pick must skip it the very next connection.
|
||||
b.setSlots([]string{"a", "b"}, allLive("a", "b"))
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 30)
|
||||
for range 10 {
|
||||
picked := b.pick(context.Background(), destDomain("example.com"), nil, resolve)
|
||||
if picked == nil || picked.Tag() != "b" {
|
||||
t.Fatalf("pick = %v, want b (a is board-dead despite live flag)", picked)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
func TestSlotLivenessVerdictAliveRevives(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
b := rrBalancer(t, 1, []string{"none"})
|
||||
b.verdict = func(tag string) urltest.HealthVerdict { return hist.Verdict(tag, 10*time.Minute) }
|
||||
_, resolve := resolveFrom("a")
|
||||
// Slot flagged dead by the last check, but the board holds a fresh success now:
|
||||
// the verdict wins and the slot is selectable again before the next tick.
|
||||
b.setSlots([]string{"a"}, nil)
|
||||
storeAlive(hist, "a", 20)
|
||||
picked := b.pick(context.Background(), destDomain("example.com"), nil, resolve)
|
||||
if picked == nil || picked.Tag() != "a" {
|
||||
t.Fatalf("pick = %v, want board-alive a despite dead flag", picked)
|
||||
}
|
||||
}
|
||||
|
||||
func TestRandomPickSkipsBoardDeadSlot(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
b := randomBalancer(t, 3)
|
||||
b.verdict = func(tag string) urltest.HealthVerdict { return hist.Verdict(tag, 10*time.Minute) }
|
||||
_, resolve := resolveFrom("a", "b", "c")
|
||||
b.setSlots([]string{"a", "b", "c"}, allLive("a", "b", "c"))
|
||||
hist.MarkFailed("a")
|
||||
for range 50 {
|
||||
picked := b.pick(context.Background(), destDomain("example.com"), nil, resolve)
|
||||
if picked == nil || picked.Tag() == "a" {
|
||||
t.Fatalf("random pick returned board-dead a")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// failover (priority balancer, pool 1): the slot occupant went board-dead → the group
|
||||
// falls back to Select, which routes to a live member, never the dead occupant.
|
||||
func TestFailoverDeadSlotFallsBackToAliveMember(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
manager := managerOf(a, nodeB)
|
||||
g := healthTestGroup(hist, manager, a, nodeB)
|
||||
bal := &balancer{poolSize: 1, priority: true}
|
||||
s := healthURLTest(g, bal, manager)
|
||||
bal.setSlots([]string{"a"}, allLive("a"))
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "b", 40)
|
||||
selected := s.dialSelect(context.Background(), N.NetworkTCP, destDomain("example.com"), nil)
|
||||
if selected == nil || selected.Tag() != "b" {
|
||||
t.Fatalf("dialSelect = %v, want alive fallback b for a dead failover slot", selected)
|
||||
}
|
||||
}
|
||||
|
||||
// --- dial retry: mark-fail + re-pick, ≤ 3 candidates, within deadline ---------------
|
||||
|
||||
func TestDialContextRetriesThroughNextAlive(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
g := healthTestGroup(hist, nil, a, nodeB)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
g.selectedOutboundTCP = a // the checker had picked a; it dies between ticks
|
||||
conn, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("DialContext failed despite live member b: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if len(dialed) != 2 || dialed[0] != "a" || dialed[1] != "b" {
|
||||
t.Fatalf("dial order = %v, want [a b]", dialed)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed dial", v)
|
||||
}
|
||||
if entry := hist.LoadURLTestHistory("a"); entry == nil {
|
||||
t.Fatalf("history entry for a was deleted; mark-fail must keep it")
|
||||
}
|
||||
}
|
||||
|
||||
func TestDialContextStopsAfterThreeCandidates(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
nodes := make([]adapter.Outbound, 0, 4)
|
||||
for i, tag := range []string{"a", "b", "c", "d"} {
|
||||
nodes = append(nodes, &healthNode{tag: tag, fail: true, dialed: &dialed})
|
||||
storeAlive(hist, tag, uint16(10+80*i)) // spread beyond tolerance so order is a,b,c
|
||||
}
|
||||
g := healthTestGroup(hist, nil, nodes...)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
_, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err == nil {
|
||||
t.Fatalf("DialContext succeeded with every member failing")
|
||||
}
|
||||
if len(dialed) != dialAttemptsMax {
|
||||
t.Fatalf("dialed %d members (%v), want at most %d", len(dialed), dialed, dialAttemptsMax)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDialContextStopsWhenContextDone(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
g := healthTestGroup(hist, nil, a, nodeB)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
ctx, cancel := context.WithCancel(context.Background())
|
||||
cancel()
|
||||
if _, err := s.DialContext(ctx, N.NetworkTCP, destDomain("example.com")); err == nil {
|
||||
t.Fatalf("DialContext succeeded past a done context")
|
||||
}
|
||||
if len(dialed) != 1 {
|
||||
t.Fatalf("dialed %v, want a single attempt within a done context", dialed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestDialContextBalancedDemotesAndRetries(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
manager := managerOf(a, nodeB)
|
||||
g := healthTestGroup(hist, manager, a, nodeB)
|
||||
bal := rrBalancer(t, 2, []string{"none"})
|
||||
s := healthURLTest(g, bal, manager)
|
||||
bal.setSlots([]string{"a", "b"}, allLive("a", "b"))
|
||||
conn, err := s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("balanced DialContext failed despite live member b: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if len(dialed) != 2 || dialed[0] != "a" || dialed[1] != "b" {
|
||||
t.Fatalf("dial order = %v, want [a b] (a demoted, b retried)", dialed)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed dial", v)
|
||||
}
|
||||
// The demotion is visible to the NEXT connection immediately: a is never dialed again.
|
||||
conn, err = s.DialContext(context.Background(), N.NetworkTCP, destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("second balanced DialContext failed: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if dialed[len(dialed)-1] != "b" {
|
||||
t.Fatalf("dial order = %v, want the second connection to go straight to b", dialed)
|
||||
}
|
||||
}
|
||||
|
||||
func TestListenPacketRetriesBeforeFirstSend(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
dialed := []string{}
|
||||
a := &healthNode{tag: "a", fail: true, dialed: &dialed}
|
||||
nodeB := &healthNode{tag: "b", dialed: &dialed}
|
||||
g := healthTestGroup(hist, nil, a, nodeB)
|
||||
s := healthURLTest(g, nil, nil)
|
||||
storeAlive(hist, "a", 10)
|
||||
storeAlive(hist, "b", 100)
|
||||
g.selectedOutboundUDP = a
|
||||
conn, err := s.ListenPacket(context.Background(), destDomain("example.com"))
|
||||
if err != nil {
|
||||
t.Fatalf("ListenPacket failed despite live member b: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
if len(dialed) != 2 || dialed[0] != "a" || dialed[1] != "b" {
|
||||
t.Fatalf("listen order = %v, want [a b]", dialed)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed listen", v)
|
||||
}
|
||||
}
|
||||
|
||||
// --- check failures mark the board, never delete ------------------------------------
|
||||
|
||||
func TestTestNodesMarksFailedKeepsEntry(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a", fail: true}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
storeAlive(hist, "a", 42)
|
||||
result := g.testNodes(context.Background(), []adapter.Outbound{a}, true)
|
||||
if len(result) != 0 {
|
||||
t.Fatalf("testNodes result = %v, want empty for a failing node", result)
|
||||
}
|
||||
entry := hist.LoadURLTestHistory("a")
|
||||
if entry == nil {
|
||||
t.Fatalf("history entry deleted on check failure; must be marked instead")
|
||||
}
|
||||
if entry.Delay != 42 {
|
||||
t.Fatalf("entry.Delay = %d, want the last-known 42 preserved", entry.Delay)
|
||||
}
|
||||
if v := hist.Verdict("a", 10*time.Minute); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(a) = %v, want dead after failed check", v)
|
||||
}
|
||||
}
|
||||
|
||||
// --- Pool / seedPool read the verdict, not entry presence ---------------------------
|
||||
|
||||
func TestPoolDelayRequiresAliveVerdict(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a := &healthNode{tag: "a"}
|
||||
manager := managerOf(a)
|
||||
g := healthTestGroup(hist, manager, a)
|
||||
bal := rrBalancer(t, 1, []string{"none"})
|
||||
s := healthURLTest(g, bal, manager)
|
||||
bal.setSlots([]string{"a"}, allLive("a"))
|
||||
storeAlive(hist, "a", 0) // live sub-ms → clamped to 1
|
||||
pool := s.Pool()
|
||||
if len(pool) != 1 || pool[0].Delay != 1 {
|
||||
t.Fatalf("Pool = %v, want alive slot a with clamped delay 1", pool)
|
||||
}
|
||||
hist.MarkFailed("a") // entry still present, but dead → delay must read 0
|
||||
pool = s.Pool()
|
||||
if len(pool) != 1 || pool[0].Delay != 0 {
|
||||
t.Fatalf("Pool = %v, want dead slot a reported with delay 0", pool)
|
||||
}
|
||||
}
|
||||
|
||||
func TestSeedPoolSkipsBoardDeadNodes(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
a, b, c := &balNode{tag: "a"}, &balNode{tag: "b"}, &balNode{tag: "c"}
|
||||
g := healthTestGroup(hist, nil, a, b, c)
|
||||
bal := rrBalancer(t, 2, []string{"none"})
|
||||
g.balancer = bal
|
||||
// a has an entry (fast, but DEAD); c is fresh-alive. The warm seed must be c, not a.
|
||||
storeAlive(hist, "a", 10)
|
||||
hist.MarkFailed("a")
|
||||
storeAlive(hist, "c", 30)
|
||||
g.seedPool()
|
||||
tags := bal.poolTags()
|
||||
if len(tags) != 2 || tags[0] != "c" {
|
||||
t.Fatalf("seeded pool = %v, want the alive c seeded first", tags)
|
||||
}
|
||||
}
|
||||
+84
-120
@@ -31,6 +31,7 @@ import (
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/generate"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
@@ -168,106 +169,34 @@ func (a *Applier) UpdateRuleSet(tag string) error {
|
||||
return a.eng.UpdateRuleSet(tag)
|
||||
}
|
||||
|
||||
// TestAllNodes launches the engine's one-shot probe of every node (manual "Test
|
||||
// all nodes", feedback #3), returning started=false when a run is already in flight
|
||||
// or the engine is absent. It delegates straight to the engine and takes NEITHER
|
||||
// the apply mutex nor the flock, so kicking off a test never blocks behind an Apply.
|
||||
// configureObservatory (re)installs the engine's background observatory from the
|
||||
// model and the JUST-GENERATED options. It is called on every SUCCESSFUL apply,
|
||||
// which is exactly when the facts it depends on — the applied option set (the
|
||||
// reachability plan's source), the global probe URL/interval and the GroupHealth
|
||||
// master switch — can have changed.
|
||||
//
|
||||
// It reads the model itself to resolve each group's probe URL and egress binding: a
|
||||
// manual run must measure every tag with the URL of the group that owns it, or it
|
||||
// overwrites that group's entries with numbers taken against a different server (see
|
||||
// engine/probeplan.go). A UCI read failure degrades to "no group opinions, global
|
||||
// default URL" so a test can always be kicked off.
|
||||
func (a *Applier) TestAllNodes() (started bool) {
|
||||
if a.eng == nil {
|
||||
return false
|
||||
}
|
||||
specs, fallback := a.probeSpecs()
|
||||
return a.eng.TestAllNodes(specs, fallback)
|
||||
}
|
||||
|
||||
// probeSpecs resolves the per-group probe facts (egress binding + effective probe URL)
|
||||
// and the global fallback URL from UCI. A read failure yields (nil, "") — every tag is
|
||||
// then measured with urltest's built-in default, which is what the generator would have
|
||||
// used anyway when nothing is configured.
|
||||
func (a *Applier) probeSpecs() (map[string]engine.GroupProbeSpec, string) {
|
||||
m, err := model.ReadUCI()
|
||||
if err != nil || m == nil {
|
||||
return nil, ""
|
||||
}
|
||||
return groupProbeSpecs(m), strings.TrimSpace(m.Globals.ProbeURL)
|
||||
}
|
||||
|
||||
// groupProbeSpecs mirrors generate.groupProbeURL's precedence — the group's own probe
|
||||
// URL, else globals' — for every group in the model, alongside its egress binding.
|
||||
//
|
||||
// It deliberately stops at "" rather than substituting the gstatic default: "" means
|
||||
// "this group has no opinion", which is what lets the planner tell a group that shares
|
||||
// the global instrument from one that chose its own (only the latter can create the
|
||||
// ambiguity resolveProbeURL has to arbitrate).
|
||||
func groupProbeSpecs(m *model.Model) map[string]engine.GroupProbeSpec {
|
||||
if m == nil || len(m.Groups) == 0 {
|
||||
return nil
|
||||
}
|
||||
global := strings.TrimSpace(m.Globals.ProbeURL)
|
||||
out := make(map[string]engine.GroupProbeSpec, len(m.Groups))
|
||||
for _, g := range m.Groups {
|
||||
url := strings.TrimSpace(g.ProbeURL)
|
||||
if url == "" {
|
||||
url = global
|
||||
}
|
||||
out[g.Name] = engine.GroupProbeSpec{
|
||||
Egress: strings.TrimSpace(g.Egress),
|
||||
ProbeURL: url,
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// configureSweep (re)installs the engine's scheduled health sweep from the model. It is
|
||||
// called on every SUCCESSFUL apply, which is exactly when the facts it depends on — the
|
||||
// group set, their egress bindings, their probe URLs and the schedule itself — can have
|
||||
// changed.
|
||||
//
|
||||
// The schedule comes from Globals.SweepInterval via model.SweepSchedule, the same
|
||||
// function ValidateGlobals warns from, so what the operator is told and what the engine
|
||||
// does cannot diverge. An UNSET interval means ON at the engine's default tick: urltest
|
||||
// groups probe only while in use and selector groups never probe at all, so with the
|
||||
// sweep off a 376-node subscription reads almost entirely "untested". A zero Interval is
|
||||
// passed through deliberately — it tells engine.SweepConfig.withDefaults to apply its
|
||||
// own default, keeping that number in one place. See engine/sweep.go for the cost model.
|
||||
func (a *Applier) configureSweep(m *model.Model) {
|
||||
// The engine keeps its walk position when the rebuilt plan is identical
|
||||
// (engine.ConfigureObservatory), so the every-minute cron reconcile cannot
|
||||
// restart the cycle. GroupHealth=false — the operator's master switch — disables
|
||||
// background probing entirely.
|
||||
func (a *Applier) configureObservatory(m *model.Model, opts option.Options) {
|
||||
if a.eng == nil {
|
||||
return
|
||||
}
|
||||
interval, enabled, _ := m.Globals.SweepSchedule()
|
||||
// GroupHealth is the operator's master-switch for our sweep. When off, disable it
|
||||
// exactly like SweepInterval="off" — no warning, this is an explicit choice — while
|
||||
// PRESERVING the interval (so flipping the toggle back on restores the schedule).
|
||||
if !m.Globals.GroupHealth {
|
||||
enabled = false
|
||||
}
|
||||
a.eng.ConfigureSweep(engine.SweepConfig{
|
||||
Enabled: enabled,
|
||||
Interval: interval,
|
||||
Specs: groupProbeSpecs(m),
|
||||
FallbackURL: strings.TrimSpace(m.Globals.ProbeURL),
|
||||
interval, _ := model.ParseDuration(m.Globals.ProbeInterval)
|
||||
a.eng.ConfigureObservatory(engine.ObservatoryConfig{
|
||||
Enabled: m.Globals.GroupHealth,
|
||||
Options: opts,
|
||||
ProbeURL: strings.TrimSpace(m.Globals.ProbeURL),
|
||||
ProbeInterval: interval,
|
||||
})
|
||||
}
|
||||
|
||||
// NodeTestStatus reports the engine's probe-all progress (running + done/total). A
|
||||
// nil engine reads as idle. Lock-free like the engine method, safe to poll hot.
|
||||
func (a *Applier) NodeTestStatus() (running bool, done, total int) {
|
||||
if a.eng == nil {
|
||||
return false, 0, 0
|
||||
}
|
||||
return a.eng.NodeTestStatus()
|
||||
}
|
||||
|
||||
// TestGroups launches the engine's one-shot per-group test (delay + exit address,
|
||||
// F2), returning started=false when a run is already in flight or the engine is
|
||||
// absent. names empty/nil = every group. Like TestAllNodes it takes NEITHER the
|
||||
// apply mutex nor the flock, so kicking off a test never blocks behind an Apply.
|
||||
// TestGroups launches the engine's one-shot exit test (delay + exit address, F2)
|
||||
// of the named groups/chains, returning started=false when a run is already in
|
||||
// flight or the engine is absent. names empty/nil = every group and every chain.
|
||||
// It takes NEITHER the apply mutex nor the flock, so kicking off a test never
|
||||
// blocks behind an Apply.
|
||||
func (a *Applier) TestGroups(names []string, probeURL string) (started bool) {
|
||||
if a.eng == nil {
|
||||
return false
|
||||
@@ -275,8 +204,9 @@ func (a *Applier) TestGroups(names []string, probeURL string) (started bool) {
|
||||
return a.eng.TestGroups(names, probeURL)
|
||||
}
|
||||
|
||||
// GroupTestStatus reports the engine's group-test progress, the set of groups the run
|
||||
// covers, and its results. A nil engine reads as idle with empty (never nil) slices, so
|
||||
// GroupTestStatus reports the engine's exit-test progress, the set of targets
|
||||
// (groups and chains) the run covers, and its results. A nil engine reads as idle
|
||||
// with empty (never nil) slices, so
|
||||
// the panel can render it unconditionally.
|
||||
func (a *Applier) GroupTestStatus() (running bool, done, total int, scope []string, results []engine.GroupTestResult) {
|
||||
if a.eng == nil {
|
||||
@@ -306,6 +236,41 @@ func (a *Applier) GroupHealthOne(name string) (engine.GroupHealth, bool) {
|
||||
return a.eng.GroupHealthOne(name)
|
||||
}
|
||||
|
||||
// ChainHealth reports each configured chain's reachability (used/unused) for the
|
||||
// Targets page's "unused" badge (plan §5.E) — the chain analogue of GroupHealth.Used.
|
||||
// names come from the desired-state model (model.ReadUCI — the same source GET
|
||||
// /api/config lists, so every chain the panel shows gets a row, including one no
|
||||
// rule references and that the running box therefore never materialised); the engine
|
||||
// supplies the observatory's published used-set. A nil engine reads as empty.
|
||||
//
|
||||
// Takes NEITHER the apply mutex nor the flock, so it is safe to poll while an apply
|
||||
// is in flight — exactly like GroupHealth. A failed UCI read (no `uci` binary, e.g.
|
||||
// off-router) yields an empty slice rather than an error, so the endpoint stays a
|
||||
// pure read.
|
||||
func (a *Applier) ChainHealth() []engine.ChainHealth {
|
||||
if a.eng == nil {
|
||||
return []engine.ChainHealth{}
|
||||
}
|
||||
return a.eng.ChainHealth(chainNamesFromModel())
|
||||
}
|
||||
|
||||
// chainNamesFromModel reads the desired-state model best-effort and returns the
|
||||
// configured chain names in model order. A failed read (off-router, no `uci`) yields
|
||||
// nil — the engine then reports an empty chain list, never an error.
|
||||
func chainNamesFromModel() []string {
|
||||
m, err := model.ReadUCI()
|
||||
if err != nil || m == nil {
|
||||
return nil
|
||||
}
|
||||
names := make([]string, 0, len(m.Chains))
|
||||
for _, c := range m.Chains {
|
||||
if name := strings.TrimSpace(c.Name); name != "" {
|
||||
names = append(names, name)
|
||||
}
|
||||
}
|
||||
return names
|
||||
}
|
||||
|
||||
// HTTPClient returns an http.Client that dials THROUGH the running engine's
|
||||
// outbound named by `via` (feedback #1/#8). It delegates to the engine; a stopped
|
||||
// engine yields engine.ErrEngineStopped and an unknown tag engine.ErrOutboundUnknown.
|
||||
@@ -317,12 +282,14 @@ func (a *Applier) HTTPClient(via string) (*http.Client, error) {
|
||||
}
|
||||
|
||||
// UpdateSubscription fetches the named subscription, folds the parsed nodes into
|
||||
// UCI (replacing exactly that sub's cache), and reconciles so the running engine
|
||||
// picks them up (feedback #8). When the subscription's FetchVia=="proxy" the fetch
|
||||
// is routed THROUGH the engine outbound named by its FetchDetour (group:/node:/
|
||||
// egress:/direct); otherwise the fetch is DIRECT. Zero-node safety is inherited
|
||||
// from subscribe.UpdateSubscription (a bad body leaves the cache untouched); UCI is
|
||||
// only written on a successful parse. Returns the node count now cached.
|
||||
// the model (replacing exactly that sub's cache), persists them to the sub's
|
||||
// JSON cache file (model.SaveSubCache) + the userinfo counters to UCI, and
|
||||
// reconciles so the running engine picks them up (feedback #8). When the
|
||||
// subscription's FetchVia=="proxy" the fetch is routed THROUGH the engine
|
||||
// outbound named by its FetchDetour (group:/node:/egress:/direct); otherwise the
|
||||
// fetch is DIRECT. Zero-node safety is inherited from subscribe.UpdateSubscription
|
||||
// (a bad body leaves the cache untouched); nothing is written on a failed parse.
|
||||
// Returns the node count now cached.
|
||||
//
|
||||
// It does NOT take the apply mutex itself — the fetch/parse run lock-free and the
|
||||
// final Reconcile takes the flock + mutex like any apply.
|
||||
@@ -373,14 +340,20 @@ func (a *Applier) UpdateSubscription(name string) (added int, err error) {
|
||||
if err != nil {
|
||||
return 0, err
|
||||
}
|
||||
// Fold the account state into the model BEFORE WriteUCI, so quota/expiry land on
|
||||
// disk in the same write as the refreshed nodes. Failing to record statistics
|
||||
// must never fail a subscription update that actually succeeded — the nodes are
|
||||
// the payload, the counters are a bonus — so the error is logged, not returned.
|
||||
// Fold the account state in BEFORE the writes, so quota/expiry land in the
|
||||
// same pass as the refreshed nodes. Failing to record statistics must never
|
||||
// fail a subscription update that actually succeeded — the nodes are the
|
||||
// payload, the counters are a bonus — so the error is logged, not returned.
|
||||
// (It can only fire on an unknown subscription name, which we just resolved.)
|
||||
if _, serr := subscribe.StoreUserInfo(m, name, info, time.Now()); serr != nil {
|
||||
a.log.Warn("subscription ", name, ": store user info: ", serr)
|
||||
}
|
||||
// Persistence is SPLIT: the refreshed node set goes to this subscription's
|
||||
// JSON cache file; UCI only takes the userinfo counters (RenderUCIExport does
|
||||
// not emit FromSub nodes), so a refresh never rewrites the whole config.
|
||||
if err = model.SaveSubCache(name, m.NodesFromSub(name)); err != nil {
|
||||
return 0, fmt.Errorf("write subscription cache: %w", err)
|
||||
}
|
||||
if err = model.WriteUCI(m); err != nil {
|
||||
return 0, fmt.Errorf("write UCI: %w", err)
|
||||
}
|
||||
@@ -538,10 +511,10 @@ func (a *Applier) applyLocked(m *model.Model) (bool, error) {
|
||||
// hotplug event) is what buries a real warning under a thousand identical
|
||||
// lines a day and evicts incident history from the in-memory ring buffer.
|
||||
a.logWarningsIfChanged(ws)
|
||||
// (Re)arm the scheduled health sweep against the config that is now running: the
|
||||
// group set, their egress bindings and their probe URLs are exactly what the sweep
|
||||
// plan is built from, and this is the only place they can change.
|
||||
a.configureSweep(m)
|
||||
// (Re)configure the observatory against the config that is now running: the
|
||||
// applied options are exactly what its reachability plan is built from, and
|
||||
// this is the only place they can change.
|
||||
a.configureObservatory(m, opts)
|
||||
raiseActiveFlag(a.log)
|
||||
return changed, nil
|
||||
}
|
||||
@@ -697,9 +670,9 @@ func (a *Applier) Teardown() error {
|
||||
a.mu.Lock()
|
||||
defer a.mu.Unlock()
|
||||
|
||||
// Stop the background sweep BEFORE closing the engine, and wait for an in-flight
|
||||
// Stop the observatory BEFORE closing the engine, and wait for an in-flight
|
||||
// tick: a teardown must not leave probe dials racing a box that is going away.
|
||||
a.eng.StopSweep()
|
||||
a.eng.StopObservatory()
|
||||
|
||||
var firstErr error
|
||||
if err := a.eng.Close(); err != nil {
|
||||
@@ -1122,12 +1095,3 @@ func effectivePanelPort(configured int) int {
|
||||
|
||||
// MarshalJSON is a convenience so callers can json.Marshal a Status directly.
|
||||
func (s Status) JSON() ([]byte, error) { return json.MarshalIndent(s, "", " ") }
|
||||
|
||||
// SweepStatus reports the engine's scheduled health-sweep progress (see
|
||||
// engine/sweep.go). A nil engine reads as disabled and idle.
|
||||
func (a *Applier) SweepStatus() (enabled bool, cursor, total int, cycles uint64) {
|
||||
if a.eng == nil {
|
||||
return false, 0, 0, 0
|
||||
}
|
||||
return a.eng.SweepStatus()
|
||||
}
|
||||
|
||||
+18
-116
@@ -2,132 +2,34 @@ package apply
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// groupProbeSpecs must mirror generate.groupProbeURL's precedence — the group's own
|
||||
// probe URL, else globals' — because that is the URL the group ACTUALLY measures its
|
||||
// members with. If the two ever drift, a manual run or the scheduled sweep starts
|
||||
// overwriting a group's entries with numbers taken against a different server, which
|
||||
// is precisely what the probe plan exists to prevent.
|
||||
func TestGroupProbeSpecsPrecedence(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.Globals{ProbeURL: "https://global.example/204"},
|
||||
Groups: []model.Group{
|
||||
{Name: "inherits"},
|
||||
{Name: "own", ProbeURL: "https://own.example/probe"},
|
||||
{Name: "bound", Egress: " wg0 "},
|
||||
{Name: "padded", ProbeURL: " https://padded.example/probe "},
|
||||
},
|
||||
}
|
||||
got := groupProbeSpecs(m)
|
||||
if len(got) != 4 {
|
||||
t.Fatalf("specs = %d entries, want 4", len(got))
|
||||
}
|
||||
if u := got["inherits"].ProbeURL; u != "https://global.example/204" {
|
||||
t.Errorf("a group with no URL of its own = %q, want the global one", u)
|
||||
}
|
||||
if u := got["own"].ProbeURL; u != "https://own.example/probe" {
|
||||
t.Errorf("a group's own URL = %q, want it to win over the global one", u)
|
||||
}
|
||||
if u := got["padded"].ProbeURL; u != "https://padded.example/probe" {
|
||||
t.Errorf("padded URL = %q, want it trimmed (a stray space would look like a different instrument)", u)
|
||||
}
|
||||
if e := got["bound"].Egress; e != "wg0" {
|
||||
t.Errorf("egress = %q, want it trimmed — it is a dedup key component", e)
|
||||
}
|
||||
if e := got["own"].Egress; e != "" {
|
||||
t.Errorf("unbound group egress = %q, want empty", e)
|
||||
}
|
||||
}
|
||||
|
||||
// No globals URL means "no opinion" all the way down: the planner then falls back to
|
||||
// urltest's built-in default, exactly as the generator does when nothing is configured.
|
||||
func TestGroupProbeSpecsNoOpinion(t *testing.T) {
|
||||
got := groupProbeSpecs(&model.Model{Groups: []model.Group{{Name: "a"}}})
|
||||
if u := got["a"].ProbeURL; u != "" {
|
||||
t.Fatalf("ProbeURL = %q, want \"\" (no opinion) so it cannot outvote a group that has one", u)
|
||||
}
|
||||
if groupProbeSpecs(nil) != nil {
|
||||
t.Fatal("nil model must yield nil specs")
|
||||
}
|
||||
if groupProbeSpecs(&model.Model{}) != nil {
|
||||
t.Fatal("a model with no groups must yield nil specs")
|
||||
}
|
||||
}
|
||||
|
||||
// configureSweep must honour Globals.SweepInterval end to end — the knob is worthless
|
||||
// if the value parses correctly in the model and is then ignored by the consumer.
|
||||
func TestConfigureSweepHonoursInterval(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
interval string
|
||||
wantEnabled bool
|
||||
}{
|
||||
{"unset is ON at the engine default", "", true},
|
||||
{"an explicit tick is ON", "30s", true},
|
||||
{"below the floor is still ON (clamped, not refused)", "1s", true},
|
||||
{"garbage is still ON (a typo must not disable health data)", "soon", true},
|
||||
{"0 is OFF", "0", false},
|
||||
{"off is OFF", "off", false},
|
||||
{"disabled is OFF", "disabled", false},
|
||||
{"none is OFF", "none", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
eng := engine.New()
|
||||
a := New(eng, nil)
|
||||
// GroupHealth: true is the default (DefaultGlobals seed); set it explicitly here
|
||||
// because a bare model.Globals{} zeroes it, which would gate the sweep off and
|
||||
// mask what this test is actually about (the SweepInterval grammar).
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: c.interval, GroupHealth: true}})
|
||||
enabled, _, _, _ := eng.SweepStatus()
|
||||
if enabled != c.wantEnabled {
|
||||
t.Errorf("%s: sweep_interval=%q -> enabled=%v, want %v", c.name, c.interval, enabled, c.wantEnabled)
|
||||
}
|
||||
eng.StopSweep()
|
||||
}
|
||||
}
|
||||
|
||||
// The GroupHealth master-switch gates the sweep independently of SweepInterval: when it
|
||||
// is off the sweep is OFF even for a perfectly valid interval, and the interval value is
|
||||
// still passed through untouched (flipping the toggle back on restores the schedule).
|
||||
func TestConfigureSweepGroupHealthGate(t *testing.T) {
|
||||
// GroupHealth off + a valid interval -> sweep disabled.
|
||||
// The GroupHealth master-switch is the ONLY thing that gates the observatory:
|
||||
// off means off, on means background probing of everything the applied config's
|
||||
// rules reach, on the engine's own schedule.
|
||||
func TestConfigureObservatoryGroupHealthGate(t *testing.T) {
|
||||
// GroupHealth off -> observatory disabled.
|
||||
eng := engine.New()
|
||||
a := New(eng, nil)
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "30s", GroupHealth: false}})
|
||||
if enabled, _, _, _ := eng.SweepStatus(); enabled {
|
||||
t.Fatalf("GroupHealth=false with sweep_interval=30s -> enabled=true, want the sweep gated off")
|
||||
a.configureObservatory(&model.Model{Globals: model.Globals{GroupHealth: false}}, option.Options{})
|
||||
if enabled, _, _ := eng.ObservatoryStatus(); enabled {
|
||||
t.Fatalf("GroupHealth=false -> enabled=true, want the observatory gated off")
|
||||
}
|
||||
eng.StopSweep()
|
||||
eng.StopObservatory()
|
||||
|
||||
// GroupHealth on (the default) + the same interval -> sweep enabled, as before.
|
||||
// GroupHealth on (the default) -> observatory enabled.
|
||||
eng = engine.New()
|
||||
a = New(eng, nil)
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "30s", GroupHealth: true}})
|
||||
if enabled, _, _, _ := eng.SweepStatus(); !enabled {
|
||||
t.Fatalf("GroupHealth=true with sweep_interval=30s -> enabled=false, want the sweep on")
|
||||
}
|
||||
eng.StopSweep()
|
||||
}
|
||||
|
||||
// The interval actually reaches the engine (and the floor clamp with it), rather than
|
||||
// every value collapsing onto the default.
|
||||
func TestConfigureSweepPassesTickThrough(t *testing.T) {
|
||||
eng := engine.New()
|
||||
a := New(eng, nil)
|
||||
defer eng.StopSweep()
|
||||
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "45s", GroupHealth: true}})
|
||||
if got := eng.SweepTickForTest(); got != 45*time.Second {
|
||||
t.Errorf("tick = %v, want 45s", got)
|
||||
}
|
||||
|
||||
a.configureSweep(&model.Model{Globals: model.Globals{SweepInterval: "1s", GroupHealth: true}})
|
||||
if got := eng.SweepTickForTest(); got != model.SweepIntervalMin {
|
||||
t.Errorf("tick = %v, want the %v floor", got, model.SweepIntervalMin)
|
||||
a.configureObservatory(&model.Model{Globals: model.Globals{GroupHealth: true}}, option.Options{})
|
||||
if enabled, _, _ := eng.ObservatoryStatus(); !enabled {
|
||||
t.Fatalf("GroupHealth=true -> enabled=false, want the observatory on")
|
||||
}
|
||||
eng.StopObservatory()
|
||||
|
||||
// A nil engine is a no-op, not a panic (offline daemon stub).
|
||||
(&Applier{}).configureObservatory(&model.Model{}, option.Options{})
|
||||
}
|
||||
|
||||
@@ -11,10 +11,9 @@ package apply
|
||||
//
|
||||
// This file computes the difference, by walking the FULLY RESOLVED routing rules
|
||||
// that generate produced. Using generate's output rather than the raw model is
|
||||
// deliberate: preset packs (ru-bypass), WAN-profile overrides and time-scheduled
|
||||
// rules have all been folded in by then, so a preset-driven "Russian addresses go
|
||||
// direct" is accounted for exactly like a hand-written rule. The raw model would
|
||||
// have missed all three.
|
||||
// deliberate: WAN-profile overrides and time-scheduled rules have both been
|
||||
// folded in by then, so an override-driven "go direct" is accounted for exactly
|
||||
// like a hand-written rule. The raw model would have missed both.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
|
||||
@@ -104,7 +104,6 @@ var protectionSections = map[string]bool{
|
||||
"ruleset": true,
|
||||
"blocklist": true,
|
||||
"allowlist": true,
|
||||
"preset": true,
|
||||
}
|
||||
|
||||
// infoMarkers identify operational notes that are not protection gaps.
|
||||
|
||||
@@ -497,9 +497,12 @@ func watchNewDevices(n *alert.Notifier, l log.ContextLogger) {
|
||||
continue // baseline poll: seed seen[], alert nothing
|
||||
}
|
||||
for _, d := range fresh {
|
||||
label := d.Hostname
|
||||
label := strings.TrimSpace(d.Hostname)
|
||||
// No hostname: the old fallback used the IP as the "name", producing
|
||||
// "10.67.0.223 (mac) at 10.67.0.223". Lead with the MAC instead.
|
||||
body := fmt.Sprintf("%s (%s) at %s", label, d.MAC, d.IP)
|
||||
if label == "" {
|
||||
label = d.IP
|
||||
body = fmt.Sprintf("%s at %s", d.MAC, d.IP)
|
||||
}
|
||||
l.Info("new device on LAN: ", d.MAC, " ", d.IP, " ", label)
|
||||
// Key on the MAC: two different devices joining within the dedup window
|
||||
@@ -509,7 +512,7 @@ func watchNewDevices(n *alert.Notifier, l log.ContextLogger) {
|
||||
n.FireIncident(alert.Incident{
|
||||
Events: []string{"new_device"},
|
||||
Title: "New device on the network",
|
||||
Body: fmt.Sprintf("%s (%s) at %s", label, d.MAC, d.IP),
|
||||
Body: body,
|
||||
Key: "new_device:" + d.MAC,
|
||||
})
|
||||
}
|
||||
@@ -544,7 +547,7 @@ func watchActiveProfile(applier *apply.Applier, logger log.ContextLogger) {
|
||||
logger.Debug("profile watch: read UCI failed — skip: ", err)
|
||||
continue
|
||||
}
|
||||
desired := desiredActiveProfile(time.Now(), m.Profiles, m.Globals.ActiveProfile, dev, netplane.IfaceDevice)
|
||||
desired := desiredActiveProfile(m.Profiles, m.Globals.ActiveProfile, dev, netplane.IfaceDevice)
|
||||
if desired == m.Globals.ActiveProfile {
|
||||
continue // no change — anti-flap
|
||||
}
|
||||
@@ -729,12 +732,21 @@ func cmdSubUpdate(name string) int {
|
||||
failed = true
|
||||
continue
|
||||
}
|
||||
// The refreshed set is durable in the sub's own JSON cache file; UCI
|
||||
// (below) only drains legacy sections — RenderUCIExport does not emit
|
||||
// FromSub nodes, so a refresh no longer rewrites /etc/config/shater.
|
||||
if serr := model.SaveSubCache(s.Name, m.NodesFromSub(s.Name)); serr != nil {
|
||||
logger.Error("sub ", s.Name, ": ", serr)
|
||||
failed = true
|
||||
continue
|
||||
}
|
||||
updated++
|
||||
summary = append(summary, fmt.Sprintf("%s: %d nodes", s.Name, added))
|
||||
}
|
||||
|
||||
// Only persist when something actually changed — a total failure must leave the
|
||||
// on-disk config (and its existing node cache) untouched.
|
||||
// on-disk config (and the existing node caches) untouched. The write is what
|
||||
// migrates an old config: legacy from_sub sections are dropped by the render.
|
||||
if updated > 0 {
|
||||
if werr := model.WriteUCI(m); werr != nil {
|
||||
fmt.Fprintln(os.Stderr, "shaterd sub update: write UCI:", werr)
|
||||
|
||||
@@ -14,34 +14,18 @@ package main
|
||||
import (
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// pickIfaceProfile returns the highest-Priority ENABLED profile whose MatchIface
|
||||
// matches the active uplink device AND whose schedule window contains now, or nil
|
||||
// when none match. A MatchIface entry matches when it equals activeDev literally
|
||||
// (already a device name) OR when it resolves to activeDev (it was a UCI interface
|
||||
// name, e.g. "wan" -> "eth1"). resolve maps a UCI iface name to its L3 device
|
||||
// (netplane.IfaceDevice in prod; a fake map in tests). Priority ties break by
|
||||
// lowest Name, matching generate's resolveActiveProfile determinism.
|
||||
//
|
||||
// THE SCHEDULE IS A SECOND CONDITION, AND'd with the interface — which is what
|
||||
// model.Profile has always documented ("only the specified ones are checked; all
|
||||
// must hold") and what this function used to ignore. An iface profile with a
|
||||
// 22:00-06:00 window was applied around the clock: the window was accepted,
|
||||
// stored, shown in the panel and inert, which is the same silent-pretending defect
|
||||
// as any other dead knob. A profile with no schedule is unaffected —
|
||||
// model.ProfileScheduleActive returns true for an empty window — so this only
|
||||
// changes configs that asked for a window in the first place.
|
||||
//
|
||||
// The evaluator lives in model (not here, and no longer only in generate) so the
|
||||
// watcher and the code generator cannot drift on what an overnight window or a
|
||||
// weekday token means. Its warnings are dropped here: this runs on a 25s ticker,
|
||||
// and re-logging a bad HH:MM every tick would bury the log — generate reports the
|
||||
// same config on every apply, which is where the operator will see it.
|
||||
func pickIfaceProfile(now time.Time, profiles []model.Profile, activeDev string, resolve func(string) string) *model.Profile {
|
||||
// matches the active uplink device, or nil when none match. A MatchIface entry
|
||||
// matches when it equals activeDev literally (already a device name) OR when it
|
||||
// resolves to activeDev (it was a UCI interface name, e.g. "wan" -> "eth1").
|
||||
// resolve maps a UCI iface name to its L3 device (netplane.IfaceDevice in prod;
|
||||
// a fake map in tests). Priority ties break by lowest Name, matching generate's
|
||||
// resolveActiveProfile determinism.
|
||||
func pickIfaceProfile(profiles []model.Profile, activeDev string, resolve func(string) string) *model.Profile {
|
||||
activeDev = strings.TrimSpace(activeDev)
|
||||
if activeDev == "" {
|
||||
return nil
|
||||
@@ -55,9 +39,6 @@ func pickIfaceProfile(now time.Time, profiles []model.Profile, activeDev string,
|
||||
if !ifaceMatches(p.MatchIface, activeDev, resolve) {
|
||||
continue
|
||||
}
|
||||
if active, _ := model.ProfileScheduleActive(now, *p); !active {
|
||||
continue
|
||||
}
|
||||
if best == nil || p.Priority > best.Priority ||
|
||||
(p.Priority == best.Priority && p.Name < best.Name) {
|
||||
best = p
|
||||
@@ -108,18 +89,13 @@ func currentIsIfaceProfile(profiles []model.Profile, current string) bool {
|
||||
// Rules:
|
||||
// - No ENABLED iface-driven profile exists => no-op (return current unchanged);
|
||||
// the watcher never touches config that has no iface profiles to manage.
|
||||
// - An iface-driven profile matches the uplink AND its schedule window is open
|
||||
// => pin its name (highest priority).
|
||||
// - An iface-driven profile matches the uplink => pin its name (highest
|
||||
// priority).
|
||||
// - No iface-driven profile matches:
|
||||
// - current pin is itself iface-driven (a stale pin whose uplink went away, OR
|
||||
// whose time window has just closed) => release it ("" -> generate falls back
|
||||
// to schedule/auto-select).
|
||||
// - current pin is itself iface-driven (a stale pin whose uplink went away)
|
||||
// => release it ("" -> generate falls back to auto-select).
|
||||
// - otherwise (empty, or a manual pin to a NON-iface profile) => leave alone.
|
||||
//
|
||||
// The release path is what makes an iface profile's schedule work at BOTH edges:
|
||||
// the watcher ticks every 25s, so the window closing looks exactly like the uplink
|
||||
// going away and the pin is dropped within a tick.
|
||||
func desiredActiveProfile(now time.Time, profiles []model.Profile, current, activeDev string, resolve func(string) string) string {
|
||||
func desiredActiveProfile(profiles []model.Profile, current, activeDev string, resolve func(string) string) string {
|
||||
hasIface := false
|
||||
for i := range profiles {
|
||||
if profiles[i].Enabled && len(profiles[i].MatchIface) > 0 {
|
||||
@@ -130,7 +106,7 @@ func desiredActiveProfile(now time.Time, profiles []model.Profile, current, acti
|
||||
if !hasIface {
|
||||
return current
|
||||
}
|
||||
if best := pickIfaceProfile(now, profiles, activeDev, resolve); best != nil {
|
||||
if best := pickIfaceProfile(profiles, activeDev, resolve); best != nil {
|
||||
return best.Name
|
||||
}
|
||||
if currentIsIfaceProfile(profiles, current) {
|
||||
|
||||
@@ -2,17 +2,10 @@ package main
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// anytime is the clock for the cases that carry NO schedule. An empty schedule
|
||||
// window is always open, so these assertions are about interface matching alone
|
||||
// and the instant is irrelevant — it is named rather than inlined so a reader does
|
||||
// not go looking for significance in the date.
|
||||
var anytime = time.Date(2026, 7, 21, 12, 0, 0, 0, time.UTC)
|
||||
|
||||
// fakeResolve maps UCI iface names to L3 devices (stands in for netplane.IfaceDevice).
|
||||
func fakeResolve(m map[string]string) func(string) string {
|
||||
return func(name string) string {
|
||||
@@ -35,7 +28,7 @@ func TestPickIfaceProfile(t *testing.T) {
|
||||
{Name: "lte", Enabled: true, Priority: 50, MatchIface: []string{"wwan0"}}, // literal device
|
||||
{Name: "lte-hi", Enabled: true, Priority: 90, MatchIface: []string{"wwan"}}, // -> wwan0 (same uplink, higher prio)
|
||||
{Name: "usb", Enabled: false, Priority: 99, MatchIface: []string{"usb0"}}, // disabled
|
||||
{Name: "sched", Enabled: true, Priority: 99, MatchIface: nil}, // not iface-driven
|
||||
{Name: "plain", Enabled: true, Priority: 99, MatchIface: nil}, // not iface-driven
|
||||
}
|
||||
|
||||
cases := []struct {
|
||||
@@ -51,7 +44,7 @@ func TestPickIfaceProfile(t *testing.T) {
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got := pickIfaceProfile(anytime, profiles, c.dev, resolve)
|
||||
got := pickIfaceProfile(profiles, c.dev, resolve)
|
||||
gotName := ""
|
||||
if got != nil {
|
||||
gotName = got.Name
|
||||
@@ -69,7 +62,7 @@ func TestPickIfaceProfileTieBreakByName(t *testing.T) {
|
||||
{Name: "beta", Enabled: true, Priority: 10, MatchIface: []string{"eth1"}},
|
||||
{Name: "alpha", Enabled: true, Priority: 10, MatchIface: []string{"eth1"}},
|
||||
}
|
||||
got := pickIfaceProfile(anytime, profiles, "eth1", resolve)
|
||||
got := pickIfaceProfile(profiles, "eth1", resolve)
|
||||
if got == nil || got.Name != "alpha" {
|
||||
t.Fatalf("tie should break to lowest name 'alpha', got %v", got)
|
||||
}
|
||||
@@ -108,7 +101,7 @@ func TestDesiredActiveProfile(t *testing.T) {
|
||||
}
|
||||
for _, c := range cases {
|
||||
t.Run(c.name, func(t *testing.T) {
|
||||
got := desiredActiveProfile(anytime, c.profiles, c.current, c.dev, resolve)
|
||||
got := desiredActiveProfile(c.profiles, c.current, c.dev, resolve)
|
||||
if got != c.want {
|
||||
t.Fatalf("desiredActiveProfile(current=%q, dev=%q) = %q, want %q", c.current, c.dev, got, c.want)
|
||||
}
|
||||
@@ -137,92 +130,3 @@ func TestParseDefaultRouteDev(t *testing.T) {
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// --- schedule as a second condition ------------------------------------------
|
||||
|
||||
// mondayNoon / mondayNight are inside and outside a 22:00-06:00 overnight window.
|
||||
var (
|
||||
mondayNoon = time.Date(2026, 7, 20, 12, 0, 0, 0, time.UTC)
|
||||
mondayNight = time.Date(2026, 7, 20, 23, 30, 0, 0, time.UTC)
|
||||
)
|
||||
|
||||
// TestIfaceProfileHonoursSchedule is the regression: an iface-driven profile with
|
||||
// a time window used to be pinned around the clock, because the watcher matched on
|
||||
// the uplink and never looked at the schedule. The window was accepted, stored and
|
||||
// displayed while restricting nothing.
|
||||
func TestIfaceProfileHonoursSchedule(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{{
|
||||
Name: "lte-night", Enabled: true, Priority: 10,
|
||||
MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00",
|
||||
}}
|
||||
|
||||
if got := pickIfaceProfile(mondayNight, profiles, "wwan0", resolve); got == nil {
|
||||
t.Fatal("inside the window the profile must be selected")
|
||||
}
|
||||
if got := pickIfaceProfile(mondayNoon, profiles, "wwan0", resolve); got != nil {
|
||||
t.Fatalf("outside the window the profile must NOT be selected, got %q", got.Name)
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileWithoutScheduleIsUnaffected: the change must only touch configs
|
||||
// that asked for a window. An empty schedule is always open.
|
||||
func TestIfaceProfileWithoutScheduleIsUnaffected(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{{Name: "lte", Enabled: true, MatchIface: []string{"wwan0"}}}
|
||||
for _, at := range []time.Time{mondayNoon, mondayNight} {
|
||||
if got := pickIfaceProfile(at, profiles, "wwan0", resolve); got == nil {
|
||||
t.Fatalf("a schedule-less iface profile must always be selectable (at %v)", at)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileScheduleClosingReleasesThePin: the window closing must behave
|
||||
// exactly like the uplink going away, or the profile would stay pinned until the
|
||||
// next reboot — which is how a "night only" profile silently becomes permanent.
|
||||
func TestIfaceProfileScheduleClosingReleasesThePin(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{{
|
||||
Name: "lte-night", Enabled: true, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00",
|
||||
}}
|
||||
if got := desiredActiveProfile(mondayNight, profiles, "", "wwan0", resolve); got != "lte-night" {
|
||||
t.Fatalf("inside the window the pin must be taken, got %q", got)
|
||||
}
|
||||
if got := desiredActiveProfile(mondayNoon, profiles, "lte-night", "wwan0", resolve); got != "" {
|
||||
t.Fatalf("outside the window the stale pin must be released, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileScheduleDoesNotClobberAManualPin: a user's manual pin to a
|
||||
// NON-iface profile is still none of the watcher's business, window or not.
|
||||
func TestIfaceProfileScheduleDoesNotClobberAManualPin(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{
|
||||
{Name: "lte-night", Enabled: true, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00"},
|
||||
{Name: "manual", Enabled: true},
|
||||
}
|
||||
if got := desiredActiveProfile(mondayNoon, profiles, "manual", "wwan0", resolve); got != "manual" {
|
||||
t.Fatalf("a manual pin to a non-iface profile must be left alone, got %q", got)
|
||||
}
|
||||
}
|
||||
|
||||
// TestIfaceProfileScheduleTieBreak: with two matching iface profiles, only the one
|
||||
// whose window is OPEN may win — priority must not resurrect a closed profile.
|
||||
func TestIfaceProfileScheduleTieBreak(t *testing.T) {
|
||||
resolve := fakeResolve(nil)
|
||||
profiles := []model.Profile{
|
||||
{Name: "night-hi", Enabled: true, Priority: 90, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "06:00"},
|
||||
{Name: "day-lo", Enabled: true, Priority: 10, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "06:00", SchedEnd: "22:00"},
|
||||
}
|
||||
if got := pickIfaceProfile(mondayNoon, profiles, "wwan0", resolve); got == nil || got.Name != "day-lo" {
|
||||
t.Fatalf("at noon the open low-priority profile must win, got %v", got)
|
||||
}
|
||||
if got := pickIfaceProfile(mondayNight, profiles, "wwan0", resolve); got == nil || got.Name != "night-hi" {
|
||||
t.Fatalf("at night the open high-priority profile must win, got %v", got)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3,6 +3,7 @@ package devices
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
@@ -46,12 +47,13 @@ func findRows(rows []Discovered, mac string) []Discovered {
|
||||
}
|
||||
|
||||
// TestDiscoverDualStackSameMAC is the audit's "one MAC, a v4 and a v6 address"
|
||||
// combination. Discovery is keyed by IP, so a dual-stacked client legitimately
|
||||
// produces TWO rows — this test pins the CONTRACT that matters to the panel and to
|
||||
// per-device stats: both rows carry the SAME MAC, and if that MAC is a configured
|
||||
// device then BOTH rows are marked configured with the same name/blockCount. A row
|
||||
// that silently lost its configured status would show up in the UI as an unmanaged
|
||||
// device that quietly escapes its parental-control rules.
|
||||
// combination. Discovery merges by MAC, so a dual-stacked client is ONE row —
|
||||
// this test pins the CONTRACT that matters to the panel and to per-device
|
||||
// stats: the row carries BOTH addresses in ips (primary first), and if that MAC
|
||||
// is a configured device the merged row is marked configured with its
|
||||
// name/blockCount. A device that split into rows (or lost its configured
|
||||
// status) would show up in the UI as an unmanaged duplicate that quietly
|
||||
// escapes its parental-control rules.
|
||||
func TestDiscoverDualStackSameMAC(t *testing.T) {
|
||||
const mac = "aa:bb:cc:dd:ee:01"
|
||||
writeNetwork(t, lanOnlyNetwork)
|
||||
@@ -71,26 +73,34 @@ func TestDiscoverDualStackSameMAC(t *testing.T) {
|
||||
}}
|
||||
rows := Discover(configured)
|
||||
|
||||
// The configured device must NOT additionally appear as a phantom offline
|
||||
// row: it was seen, so exactly one row carries that MAC.
|
||||
got := findRows(rows, mac)
|
||||
if len(got) != 2 {
|
||||
t.Fatalf("dual-stack host produced %d rows, want 2 (one per address): %+v", len(got), rows)
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("dual-stack host produced %d rows, want 1 merged row: %+v", len(got), rows)
|
||||
}
|
||||
for _, r := range got {
|
||||
if !r.Configured {
|
||||
t.Errorf("row %s (mac %s) is not marked configured — it would escape its device rules", r.IP, r.MAC)
|
||||
}
|
||||
if r.Name != "Phone" {
|
||||
t.Errorf("row %s: Name = %q, want Phone", r.IP, r.Name)
|
||||
}
|
||||
if r.BlockCount != 2 {
|
||||
t.Errorf("row %s: BlockCount = %d, want 2", r.IP, r.BlockCount)
|
||||
}
|
||||
r := got[0]
|
||||
if r.IP != "192.168.1.50" {
|
||||
t.Fatalf("primary must be the leased v4 address, got %q", r.IP)
|
||||
}
|
||||
|
||||
// The configured device must NOT additionally appear as a phantom offline row:
|
||||
// it was seen, so exactly the two live rows carry that MAC.
|
||||
if n := len(findRows(rows, mac)); n != 2 {
|
||||
t.Fatalf("configured-but-unseen fallback added a phantom row: %d rows for mac %s", n, mac)
|
||||
want := []string{"192.168.1.50", "fe80::a8bb:ccff:fedd:ee01"}
|
||||
if !reflect.DeepEqual(r.IPs, want) {
|
||||
t.Fatalf("ips = %+v, want %+v (primary first)", r.IPs, want)
|
||||
}
|
||||
if r.State != "online" || !r.Online {
|
||||
t.Fatalf("REACHABLE v4 must win the merged state, got %+v", r)
|
||||
}
|
||||
if !r.Configured {
|
||||
t.Errorf("merged row (mac %s) is not marked configured — it would escape its device rules", r.MAC)
|
||||
}
|
||||
if r.Name != "Phone" {
|
||||
t.Errorf("Name = %q, want Phone", r.Name)
|
||||
}
|
||||
if r.BlockCount != 2 {
|
||||
t.Errorf("BlockCount = %d, want 2", r.BlockCount)
|
||||
}
|
||||
if r.Network != "lan" || r.Iface != "br-lan" {
|
||||
t.Errorf("merged row not labelled from its primary address: %+v", r)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -180,7 +190,8 @@ func TestDiscoverIsDeterministic(t *testing.T) {
|
||||
t.Fatalf("run %d returned %d rows, first run returned %d", i, len(got), len(first))
|
||||
}
|
||||
for j := range got {
|
||||
if got[j] != first[j] {
|
||||
// reflect.DeepEqual: Discovered carries a slice (IPs) so == is out.
|
||||
if !reflect.DeepEqual(got[j], first[j]) {
|
||||
t.Fatalf("run %d row %d differs from the first run:\n got %+v\nwant %+v", i, j, got[j], first[j])
|
||||
}
|
||||
}
|
||||
|
||||
+104
-21
@@ -98,7 +98,11 @@ func MACToIP(path string) map[string]string {
|
||||
// Discovered is one row of GET /api/devices: a host seen on the LAN (via lease
|
||||
// and/or neighbour table) cross-referenced with the configured devices.
|
||||
type Discovered struct {
|
||||
IP string `json:"ip"`
|
||||
IP string `json:"ip"` // primary address (most recent lease, else "most online")
|
||||
// IPs is every address currently known for this device (IPv4+IPv6, multiple
|
||||
// leases), primary first. Never nil; empty only for a configured-but-unseen
|
||||
// device whose config entry carries no IP.
|
||||
IPs []string `json:"ips"`
|
||||
MAC string `json:"mac"`
|
||||
Hostname string `json:"hostname"`
|
||||
Online bool `json:"online"` // reachable or recently-seen (state != offline)
|
||||
@@ -123,14 +127,18 @@ type neigh struct {
|
||||
State string // online|idle|offline
|
||||
}
|
||||
|
||||
// Discover merges the DHCP lease table with `ip neigh show`, cross-references the
|
||||
// configured devices, and returns one row per discovered host (union of leases +
|
||||
// neighbours), plus any configured device that was not otherwise seen (as an
|
||||
// offline row) so the panel can always render the full parental-control list.
|
||||
// Discover merges the DHCP lease table with `ip neigh show`, folds every address
|
||||
// sharing a MAC into ONE row per physical device (a dual-stacked or multi-leased
|
||||
// client is still one device), cross-references the configured devices, and
|
||||
// appends any configured device that was not otherwise seen (as an offline row)
|
||||
// so the panel can always render the full parental-control list.
|
||||
// Never returns nil.
|
||||
func Discover(configured []model.Device) []Discovered {
|
||||
// Index by IP, seeding from the lease table (ip/mac/hostname).
|
||||
// Phase 1: aggregate per-IP, seeding from the lease table (ip/mac/hostname).
|
||||
// The MAC-level merge happens only after every signal is in, so a MAC that
|
||||
// arrives late (from the neighbour table) still groups its earlier lease IPs.
|
||||
byIP := map[string]*Discovered{}
|
||||
leaseSeq := map[string]int{} // IP -> index of its LAST lease line; absent = neigh-only
|
||||
order := []string{}
|
||||
add := func(ip string) *Discovered {
|
||||
if d, ok := byIP[ip]; ok {
|
||||
@@ -141,7 +149,7 @@ func Discover(configured []model.Device) []Discovered {
|
||||
order = append(order, ip)
|
||||
return d
|
||||
}
|
||||
for _, l := range ParseLeases(LeasesPath) {
|
||||
for seq, l := range ParseLeases(LeasesPath) {
|
||||
d := add(l.IP)
|
||||
if d.MAC == "" {
|
||||
d.MAC = l.MAC
|
||||
@@ -149,6 +157,7 @@ func Discover(configured []model.Device) []Discovered {
|
||||
if d.Hostname == "" {
|
||||
d.Hostname = l.Hostname
|
||||
}
|
||||
leaseSeq[l.IP] = seq // later lease lines are more recent
|
||||
}
|
||||
|
||||
// Derive the LAN device set (and WAN devices) from the network config so the
|
||||
@@ -171,28 +180,98 @@ func Discover(configured []model.Device) []Discovered {
|
||||
if d.Iface == "" {
|
||||
d.Iface = n.Dev
|
||||
}
|
||||
// Prefer the "most online" state if a host somehow appears twice.
|
||||
// Prefer the "most online" state if an address somehow appears twice.
|
||||
if stateRank(n.State) > stateRank(d.State) {
|
||||
d.State = n.State
|
||||
}
|
||||
}
|
||||
|
||||
// Cross-reference the configured devices (match by MAC first, else IP). Track
|
||||
// which configured entries matched so unseen ones can be appended as offline.
|
||||
matched := make([]bool, len(configured))
|
||||
for _, d := range byIP {
|
||||
if i, ok := matchConfigured(configured, d.MAC, d.IP); ok {
|
||||
markConfigured(d, configured[i])
|
||||
matched[i] = true
|
||||
// Phase 2: group addresses by lower-cased MAC — one group per physical
|
||||
// device. Addresses with NO known MAC stay one-per-IP: with no hardware
|
||||
// identity there is nothing safe to merge on. Groups keep first-seen order
|
||||
// (lease-file order, then neighbour order) so output is deterministic.
|
||||
type group struct {
|
||||
primary *Discovered // the address the row's ip/iface/network come from
|
||||
members []*Discovered
|
||||
}
|
||||
groups := make([]*group, 0, len(order))
|
||||
byMAC := map[string]*group{}
|
||||
// morePrimary reports whether a should displace cur as a group's primary:
|
||||
// the most recent lease wins (mirrors MACToIP, so the panel and the generate
|
||||
// stage agree on a device's "current" IP), then the "more online" address;
|
||||
// otherwise the first-seen address keeps the slot.
|
||||
morePrimary := func(a, cur *Discovered) bool {
|
||||
as, aok := leaseSeq[a.IP]
|
||||
cs, cok := leaseSeq[cur.IP]
|
||||
if aok != cok {
|
||||
return aok
|
||||
}
|
||||
if aok && as != cs {
|
||||
return as > cs
|
||||
}
|
||||
return stateRank(a.State) > stateRank(cur.State)
|
||||
}
|
||||
for _, ip := range order {
|
||||
d := byIP[ip]
|
||||
mac := strings.ToLower(d.MAC)
|
||||
if mac == "" {
|
||||
groups = append(groups, &group{primary: d, members: []*Discovered{d}})
|
||||
continue
|
||||
}
|
||||
g, ok := byMAC[mac]
|
||||
if !ok {
|
||||
g = &group{primary: d, members: []*Discovered{d}}
|
||||
byMAC[mac] = g
|
||||
groups = append(groups, g)
|
||||
continue
|
||||
}
|
||||
g.members = append(g.members, d)
|
||||
if morePrimary(d, g.primary) {
|
||||
g.primary = d
|
||||
}
|
||||
}
|
||||
|
||||
out := make([]Discovered, 0, len(order)+len(configured))
|
||||
for _, ip := range order {
|
||||
d := byIP[ip]
|
||||
d.Online = d.State != "offline"
|
||||
labelNetwork(d, nets)
|
||||
out = append(out, *d)
|
||||
// Phase 3: flatten each group into one row and cross-reference the
|
||||
// configured devices (MAC first, else primary IP). Track which configured
|
||||
// entries matched so unseen ones can be appended as offline rows.
|
||||
matched := make([]bool, len(configured))
|
||||
out := make([]Discovered, 0, len(groups)+len(configured))
|
||||
for _, g := range groups {
|
||||
row := *g.primary
|
||||
row.IPs = make([]string, 0, len(g.members))
|
||||
row.IPs = append(row.IPs, g.primary.IP)
|
||||
for _, m := range g.members {
|
||||
if m == g.primary {
|
||||
continue
|
||||
}
|
||||
row.IPs = append(row.IPs, m.IP)
|
||||
// Best state among the device's addresses wins the merged row.
|
||||
if stateRank(m.State) > stateRank(row.State) {
|
||||
row.State = m.State
|
||||
}
|
||||
}
|
||||
// Hostname: first non-empty in first-seen order (the primary may be a
|
||||
// hostname-less neighbour row while an older lease knew the name).
|
||||
for _, m := range g.members {
|
||||
if m.Hostname != "" {
|
||||
row.Hostname = m.Hostname
|
||||
break
|
||||
}
|
||||
}
|
||||
// Iface follows the primary; fall back to any member that knows it.
|
||||
for _, m := range g.members {
|
||||
if row.Iface != "" {
|
||||
break
|
||||
}
|
||||
row.Iface = m.Iface
|
||||
}
|
||||
row.Online = row.State != "offline"
|
||||
labelNetwork(&row, nets)
|
||||
if i, ok := matchConfigured(configured, row.MAC, row.IP); ok {
|
||||
markConfigured(&row, configured[i])
|
||||
matched[i] = true
|
||||
}
|
||||
out = append(out, row)
|
||||
}
|
||||
// Append configured-but-unseen devices as offline rows (identity from config).
|
||||
for i, cd := range configured {
|
||||
@@ -201,8 +280,12 @@ func Discover(configured []model.Device) []Discovered {
|
||||
}
|
||||
row := Discovered{
|
||||
IP: strings.TrimSpace(cd.IP), MAC: strings.ToLower(strings.TrimSpace(cd.MAC)),
|
||||
IPs: []string{},
|
||||
State: "offline", Online: false,
|
||||
}
|
||||
if row.IP != "" {
|
||||
row.IPs = append(row.IPs, row.IP)
|
||||
}
|
||||
markConfigured(&row, cd)
|
||||
labelNetwork(&row, nets)
|
||||
out = append(out, row)
|
||||
|
||||
@@ -102,6 +102,9 @@ func TestDiscoverMerge(t *testing.T) {
|
||||
if c.MAC != "77:88:99:aa:bb:cc" || c.State != "online" || c.Configured {
|
||||
t.Fatalf("neigh-only host wrong: %+v", c)
|
||||
}
|
||||
if len(c.IPs) != 1 || c.IPs[0] != "192.168.1.99" {
|
||||
t.Fatalf("single-address row must carry ips=[its ip], got %+v", c.IPs)
|
||||
}
|
||||
|
||||
// The away phone (configured, unseen) must appear as an offline row.
|
||||
var away *Discovered
|
||||
@@ -118,6 +121,46 @@ func TestDiscoverMerge(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestDiscoverMergesLeasesSameMAC proves one physical device with several
|
||||
// addresses (two leases on the same MAC, e.g. after a re-lease) yields ONE row
|
||||
// whose ips carries every address, with the most recent lease as the primary.
|
||||
func TestDiscoverMergesLeasesSameMAC(t *testing.T) {
|
||||
const mac = "aa:bb:cc:dd:ee:ff"
|
||||
leases := "" +
|
||||
"1700000000 " + mac + " 192.168.1.50 kidpc 01:aa\n" +
|
||||
"1700000100 " + mac + " 192.168.1.77 kidpc 01:aa\n"
|
||||
LeasesPath = writeLeases(t, leases)
|
||||
|
||||
// Only the old address is in the neighbour table (REACHABLE): the merged
|
||||
// row must still be online even though the PRIMARY (newest lease) is not.
|
||||
defer fakeNeigh(t, "192.168.1.50 dev br-lan lladdr "+mac+" REACHABLE\n")()
|
||||
|
||||
rows := Discover(nil)
|
||||
|
||||
var got []Discovered
|
||||
for _, r := range rows {
|
||||
if r.MAC == mac {
|
||||
got = append(got, r)
|
||||
}
|
||||
}
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("same-MAC leases must merge into one row, got %d: %+v", len(got), rows)
|
||||
}
|
||||
d := got[0]
|
||||
if d.IP != "192.168.1.77" {
|
||||
t.Fatalf("primary must be the most recent lease IP, got %q", d.IP)
|
||||
}
|
||||
if len(d.IPs) != 2 || d.IPs[0] != "192.168.1.77" || d.IPs[1] != "192.168.1.50" {
|
||||
t.Fatalf("ips must list every address primary-first, got %+v", d.IPs)
|
||||
}
|
||||
if d.State != "online" || !d.Online {
|
||||
t.Fatalf("best state among addresses must win the merge, got %+v", d)
|
||||
}
|
||||
if d.Hostname != "kidpc" {
|
||||
t.Fatalf("hostname lost in merge: %+v", d)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDiscoverDropsWANNeigh proves the neighbour table is filtered to LAN
|
||||
// devices and each survivor is labelled with its OpenWrt network: a WAN-side
|
||||
// slirp row (dev eth1, 10.0.2.2) is dropped while the LAN-side row (dev br-lan,
|
||||
|
||||
+16
-27
@@ -71,33 +71,21 @@ type Engine struct {
|
||||
lastGood option.Options // last known-good options (the config running BEFORE current)
|
||||
hasLastGood bool // false until a second successful Apply gives us a predecessor
|
||||
|
||||
// Manual "Test all nodes" probe-all state (see probeall.go). These are guarded by
|
||||
// their own atomics — INDEPENDENT of mu — so a status poll never blocks behind an
|
||||
// Apply, and a run may proceed concurrently with reads. testRunning is the
|
||||
// singleton guard (only one run at a time); testDone/testTotal report progress.
|
||||
testRunning atomic.Bool
|
||||
testDone atomic.Int64
|
||||
testTotal atomic.Int64
|
||||
// log is the engine's own logger. Today its one consumer is the observatory's
|
||||
// verdict-flip trail (observatory.go probeOneInto) — the single diagnostic
|
||||
// record of a node going alive<->dead. Set once in New, never nil there; code
|
||||
// paths reached from a hand-built Engine{} must nil-check it.
|
||||
log log.ContextLogger
|
||||
|
||||
// dead is our OWN overlay of "this outbound was probed and did not answer",
|
||||
// tag -> time of that failed probe (see health.go for why this is not written into
|
||||
// the engine's urltest history). It lives on the Engine, not on a Box, so — like
|
||||
// the pre-registered history store — it survives an Apply swap instead of turning
|
||||
// every known-dead node back into "untested" on every apply. Guarded by its own
|
||||
// leaf mutex, INDEPENDENT of mu, so a health read never blocks behind an Apply.
|
||||
deadMu sync.Mutex
|
||||
dead map[string]time.Time
|
||||
// Observatory state (see observatory.go): the background prober that keeps
|
||||
// the health board fresh for everything the routing rules reach. Its own leaf
|
||||
// mutex, independent of mu, so a tick never blocks an Apply and vice versa.
|
||||
obsMu sync.Mutex
|
||||
obs observatoryState
|
||||
|
||||
// Scheduled sweep state (see sweep.go): the background observatory that keeps node
|
||||
// health fresh so every strategy has current data to select on. Its own leaf mutex,
|
||||
// independent of mu, so a tick never blocks an Apply and vice versa.
|
||||
sweepMu sync.Mutex
|
||||
sweep sweepState
|
||||
|
||||
// Per-group test state (see grouptest.go). Same shape and same reasoning as the
|
||||
// probe-all state above — independent of mu so a progress poll never blocks
|
||||
// behind an Apply — plus a short leaf lock for the results slice, which is a
|
||||
// value the atomics cannot carry.
|
||||
// Per-group test state (see grouptest.go). Guarded by its own atomics and a
|
||||
// short leaf lock (for the results slice, a value the atomics cannot carry) —
|
||||
// independent of mu so a progress poll never blocks behind an Apply.
|
||||
groupTestRunning atomic.Bool
|
||||
groupTestDone atomic.Int64
|
||||
groupTestTotal atomic.Int64
|
||||
@@ -111,7 +99,8 @@ type Engine struct {
|
||||
|
||||
// New builds the engine context (with the shater-owned slim protocol/dns/
|
||||
// service registries wired via registry.Context) and returns a stopped Engine.
|
||||
// An optional logger may be supplied for the deprecated-feature manager;
|
||||
// An optional logger may be supplied for the deprecated-feature manager and the
|
||||
// observatory's verdict-flip trail;
|
||||
// log.StdLogger() is used by default. The variadic mirrors a "logFactory"-style
|
||||
// optional argument.
|
||||
func New(logger ...log.ContextLogger) *Engine {
|
||||
@@ -151,7 +140,7 @@ func New(logger ...log.ContextLogger) *Engine {
|
||||
// URLTestHistory() below. The pointer is stable across Apply swaps, so health
|
||||
// history survives config changes instead of being reset on every apply.
|
||||
ctx = service.ContextWithPtr(ctx, urltest.NewHistoryStorage())
|
||||
return &Engine{ctx: ctx}
|
||||
return &Engine{ctx: ctx, log: l}
|
||||
}
|
||||
|
||||
// Apply installs opts as the running configuration.
|
||||
|
||||
+121
-39
@@ -25,18 +25,18 @@ import (
|
||||
//
|
||||
// # Where the truth already is (no new probing)
|
||||
//
|
||||
// Nothing here dials anything. Every number below is a projection of what the engine
|
||||
// has ALREADY collected, read through one HealthView (health.go) — the shared
|
||||
// urltest.HistoryStorage for measurements plus our own overlay of failure verdicts,
|
||||
// both process-global and stable across Apply swaps:
|
||||
// Nothing here dials anything. Every number below is a projection of the shared
|
||||
// health board (common/urltest), read through one HealthView (health.go) —
|
||||
// process-global and stable across Apply swaps:
|
||||
//
|
||||
// - a urltest-strategy group probes its OWN member outbounds on its interval and
|
||||
// writes the result under the member's tag (protocol/group/urltest.go testNodes →
|
||||
// StoreURLTestHistory(realTag)). For an egress-bound group those member tags ARE
|
||||
// the "group-…" copies, so the per-copy truth is already in the store — we simply
|
||||
// never looked at it, because we only ever projected base node tags;
|
||||
// - the manual "Test all nodes" run (probeall.go) probes every tag including those
|
||||
// copies, and is the ONLY thing that can record a FAILURE.
|
||||
// writes the result under the member's tag (protocol/group/urltest.go testNodes).
|
||||
// For an egress-bound group those member tags ARE the "group-…" copies, so the
|
||||
// per-copy truth is already in the store;
|
||||
// - the observatory (observatory.go) probes everything the routing rules reach —
|
||||
// members, egress copies, chain exits — on the global probe interval;
|
||||
// - a failure, whoever finds it (group checker, observatory, failed user dial),
|
||||
// is recorded with MarkFailed and never by deletion.
|
||||
//
|
||||
// So the entire feature is a read. The cost of one GroupHealth call is
|
||||
// len(groups) x len(members) map lookups — cheap enough to serve on a few-second panel
|
||||
@@ -44,21 +44,20 @@ import (
|
||||
//
|
||||
// # "dead" vs "untested" (the honesty requirement)
|
||||
//
|
||||
// A urltest group DELETES a member's history entry when its probe fails
|
||||
// (urltest.go:512 DeleteURLTestHistory). So from the group's own probing, "probed and
|
||||
// failed" and "never probed" are the SAME observation: no entry. We must not paper
|
||||
// over that — a group whose members were never probed must not look healthy. The one
|
||||
// thing that CAN distinguish them is TestAllNodes, whose failure verdicts live in the
|
||||
// overlay HealthView consults. decodeHealth (health.go) is the single place those
|
||||
// rules are written down; this file only counts what it returns.
|
||||
// The board keeps successes AND failures per tag, and a verdict is computed on
|
||||
// read against a TTL: dead is always a POSITIVE finding (a probe or dial ran and
|
||||
// failed, recently), untested is "nothing fresh enough is known" — including a
|
||||
// once-alive record that aged past the TTL. decodeHealth (health.go) is the
|
||||
// single place those rules are written down; this file only counts what it
|
||||
// returns.
|
||||
//
|
||||
// # Why so many members are legitimately "untested"
|
||||
// # Why "untested" still legitimately exists
|
||||
//
|
||||
// A urltest group only probes while it is BEING USED: Touch() starts the ticker and an
|
||||
// idle group stops it (urltest.go Touch/performUpdateCheck). A selector group never
|
||||
// probes at all — it dials one fixed member (selector.go). On a 376-node subscription
|
||||
// most members therefore have no history and never will. That is a true statement about
|
||||
// what the router knows, and it is what the panel must show.
|
||||
// The observatory probes only what the routing rules can reach (plan §5.C). A
|
||||
// group no enabled rule routes through is deliberately never probed — its
|
||||
// members stay untested and the group is reported Used=false, which the panel
|
||||
// renders as "unused" rather than as a health problem. Members of a USED group
|
||||
// fill in within seconds of an apply (immediate first observatory pass).
|
||||
|
||||
// GroupMemberHealth is one member of one group, as that group sees it.
|
||||
//
|
||||
@@ -102,18 +101,17 @@ type GroupMemberHealth struct {
|
||||
// fallback 8 / 8 alive tested 8 of 24
|
||||
// stealth 2 / 2 alive
|
||||
//
|
||||
// Untested is deliberately NOT a third equal column, and must NEVER be folded into
|
||||
// Dead. A urltest group probes its members LAZILY — only while it is being used — so
|
||||
// on a freshly booted router with a 376-node subscription the history is nearly empty
|
||||
// by design. Counting no-entry as dead would render "3 alive / 373 dead" at exactly
|
||||
// the moment nothing is wrong. A reading that alarms without cause teaches the
|
||||
// operator to distrust every reading, which is worse than no reading.
|
||||
// Untested is deliberately NOT a third equal column, and must NEVER be folded
|
||||
// into Dead. An UNUSED group (Used=false) is never probed at all, and a used
|
||||
// group's numbers may lag a few seconds behind an apply (the observatory's first
|
||||
// pass) or age out on the health TTL. Counting no-fresh-data as dead would raise
|
||||
// an alarm at exactly the moment nothing is wrong, and a reading that alarms
|
||||
// without cause teaches the operator to distrust every reading.
|
||||
//
|
||||
// Dead is only ever a POSITIVE finding: a probe ran and failed, recorded by the manual
|
||||
// "Test all nodes" run. Absence of a measurement proves nothing — the group deletes the
|
||||
// entry when its own probe fails, so "never probed" and "probed and failed" collapse
|
||||
// into the same absence, and only that manual run can tell them apart. Anything with
|
||||
// neither a measurement nor a verdict is Untested, full stop.
|
||||
// Dead is only ever a POSITIVE finding: a probe or dial ran and FAILED recently,
|
||||
// recorded on the shared health board by the observatory, the group's own
|
||||
// checker, or a failed user dial. Absence of fresh data proves nothing — it is
|
||||
// Untested, full stop.
|
||||
type GroupHealth struct {
|
||||
Group string `json:"group"` // the group's outbound tag == its configured name
|
||||
Type string `json:"type"` // "selector" | "urltest" (engine group type)
|
||||
@@ -122,6 +120,13 @@ type GroupHealth struct {
|
||||
// The panel should say so: these numbers are not comparable with the global
|
||||
// per-node numbers, and deliberately so.
|
||||
Bound bool `json:"bound"`
|
||||
// Used is false when the group is reachable from NO enabled routing rule
|
||||
// (nor Final, nor a DNS detour) — it is outside the observatory's plan, so
|
||||
// nothing probes it and its members stay untested by design. The panel
|
||||
// renders such a group "unused" instead of showing health counters. Reported
|
||||
// true when the observatory is disabled or not yet configured: no badge is
|
||||
// better than a wrong one.
|
||||
Used bool `json:"used"`
|
||||
// Selected is the NODE NAME the group currently routes through (adapter
|
||||
// OutboundGroup.Now(), mapped back through the copy tag). "" when the group has
|
||||
// not selected anything yet.
|
||||
@@ -176,8 +181,10 @@ func (e *Engine) GroupHealth(withMembers bool) []GroupHealth {
|
||||
}
|
||||
// ONE health view for the whole projection: every member of every group is decoded
|
||||
// against the same instant, so a 3-group / 376-member config cannot report counters
|
||||
// that disagree with each other because the store moved underneath them.
|
||||
// that disagree with each other because the store moved underneath them. The
|
||||
// used-set is the one the observatory published for the running plan.
|
||||
view := e.HealthView()
|
||||
used := e.observatoryUsed()
|
||||
for _, ob := range om.Outbounds() {
|
||||
if !isGroupOutbound(ob.Type(), ob.Tag()) {
|
||||
continue
|
||||
@@ -188,7 +195,7 @@ func (e *Engine) GroupHealth(withMembers bool) []GroupHealth {
|
||||
// asked for its members; there is nothing honest to report about it.
|
||||
continue
|
||||
}
|
||||
out = append(out, groupHealthOf(g, ob.Type(), view, withMembers))
|
||||
out = append(out, groupHealthOf(g, ob.Type(), view, used, withMembers))
|
||||
}
|
||||
return out
|
||||
}
|
||||
@@ -213,9 +220,11 @@ func (e *Engine) GroupHealthOne(name string) (GroupHealth, bool) {
|
||||
|
||||
// groupHealthOf builds one group's health row from the shared history. Pure apart
|
||||
// from the history reads, so it is unit-testable against a hand-built store.
|
||||
func groupHealthOf(g adapter.OutboundGroup, typ string, view HealthView, withMembers bool) GroupHealth {
|
||||
// used is the observatory's published used-set; nil means "unknown", reported as
|
||||
// Used=true (see GroupHealth.Used).
|
||||
func groupHealthOf(g adapter.OutboundGroup, typ string, view HealthView, used map[string]bool, withMembers bool) GroupHealth {
|
||||
name := g.Tag()
|
||||
gh := GroupHealth{Group: name, Type: typ, FreshestSeconds: -1}
|
||||
gh := GroupHealth{Group: name, Type: typ, FreshestSeconds: -1, Used: used == nil || used[name]}
|
||||
|
||||
members := g.All()
|
||||
gh.Total = len(members)
|
||||
@@ -287,7 +296,7 @@ func groupHealthOf(g adapter.OutboundGroup, typ string, view HealthView, withMem
|
||||
// whose name happens to contain "-m<digits>-" cannot confuse the split. The residual
|
||||
// edge case is a real node literally NAMED "group-<thisgroup>-m0-x" inside that same
|
||||
// group, which would be read as a copy of node "x": the "group-"/"chain-"/"egress-"
|
||||
// prefixes are documented reserved tag namespaces (see shouldProbeOutbound), and the
|
||||
// prefixes are documented reserved tag namespaces (see probeablePlanTag), and the
|
||||
// generator already refuses copies that collide with a real node.
|
||||
func parseGroupCopyTag(group, tag string) (member string, ok bool) {
|
||||
prefix := "group-" + group + "-m"
|
||||
@@ -310,3 +319,76 @@ func parseGroupCopyTag(group, tag string) (member string, ok bool) {
|
||||
}
|
||||
return member, true
|
||||
}
|
||||
|
||||
// ChainHealth is one configured chain's reachability, mirroring GroupHealth.Used
|
||||
// for the chain card (plan §5.E): a chain no enabled rule routes through is never
|
||||
// probed — the observatory walks only reachable paths — and the panel renders it
|
||||
// "unused" rather than as a health problem. A chain has no membership counters: it
|
||||
// is a fixed path, and its end-to-end health is the exit test's job, not a roll-up.
|
||||
type ChainHealth struct {
|
||||
// Name is the chain's model name (config chain "chain:<name>"), what the Targets
|
||||
// page lists and what a rule targets.
|
||||
Name string `json:"name"`
|
||||
// Used is false when the chain is reachable from NO enabled routing rule (nor
|
||||
// Final, nor a DNS detour) — it is outside the observatory's plan, so its exit
|
||||
// is never probed and its end-to-end health stays untested by design. The panel
|
||||
// renders such a chain "unused" instead of an exit-test readout. Reported true
|
||||
// when the observatory is disabled or not yet configured: no badge is better
|
||||
// than a wrong one (the same rule as GroupHealth.Used).
|
||||
Used bool `json:"used"`
|
||||
}
|
||||
|
||||
// ChainHealth reports the reachability (used/unused) of every named chain, one row
|
||||
// per name, in the order given. It is the chain analogue of GroupHealth.Used: a
|
||||
// chain the observatory's used-set does not cover is reported Used=false so the
|
||||
// panel can mark it "unused" instead of running an exit test against a path nothing
|
||||
// routes through.
|
||||
//
|
||||
// names come from the desired-state model, NOT the running box: a chain no rule
|
||||
// references is never materialised (generate/chain.go resolveChain is lazy), so it
|
||||
// is invisible to a box-only enumeration — yet the panel lists it from the config
|
||||
// and must be able to badge it. The engine supplies the only fact a box read can
|
||||
// add here, the observatory's published used-set. Pure apart from that read; nil
|
||||
// names or a stopped engine (nil used-set) yield an empty/used-everything result.
|
||||
func (e *Engine) ChainHealth(names []string) []ChainHealth {
|
||||
out := make([]ChainHealth, 0, len(names))
|
||||
used := e.observatoryUsed()
|
||||
usedChains := usedChainNames(used)
|
||||
for _, name := range names {
|
||||
name = strings.TrimSpace(name)
|
||||
if name == "" {
|
||||
continue
|
||||
}
|
||||
out = append(out, ChainHealth{Name: name, Used: used == nil || usedChains[name]})
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// usedChainNames recovers the set of chain NAMES the observatory's used-set covers.
|
||||
//
|
||||
// The used-set is keyed by outbound TAG (the generator's schema, materialised in
|
||||
// opts); a chain's exit wrapper is tagged "chain-<name>-h<digits>" (generate/chain.go
|
||||
// buildHopWrapper), so parseChainExitTag (grouptest.go) inverts that exactly. Member
|
||||
// copies ("chain-<name>-h<i>-<member>") and non-chain tags parse false and are
|
||||
// skipped. nil used-set ⇒ nil (the caller treats nil as "everything used" — no badge
|
||||
// is better than a wrong one, the same convention as GroupHealth.Used).
|
||||
//
|
||||
// A single-hop chain WITHOUT an egress entry never gets a wrapper (buildChain
|
||||
// resolves it straight to the underlying node/group tag), so no chain- tag exists
|
||||
// for it and it cannot be recovered here — it reads Used=false even when a rule
|
||||
// targets it. That mirrors the exit test (chainTargetsFrom), which likewise cannot
|
||||
// discover such a chain and reports it missing when named; a 1-hop chain is an alias
|
||||
// for the hop it names, and its reachability is that hop's, reported on the hop's
|
||||
// own card.
|
||||
func usedChainNames(used map[string]bool) map[string]bool {
|
||||
if used == nil {
|
||||
return nil
|
||||
}
|
||||
out := make(map[string]bool, len(used))
|
||||
for tag := range used {
|
||||
if name, ok := parseChainExitTag(tag); ok {
|
||||
out[name] = true
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
@@ -91,16 +91,17 @@ func TestParseGroupCopyTag(t *testing.T) {
|
||||
func TestGroupHealthUnbound(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{Time: now.Add(-time.Minute), Delay: 100})
|
||||
// n2 is DEAD: a failed probe leaves nothing in the engine's history and is recorded
|
||||
// only in our overlay (see health.go — an invented history entry would be read as
|
||||
// "alive" by the round_robin pool planner).
|
||||
dead := map[string]time.Time{"n2": now.Add(-10 * time.Second)}
|
||||
// n3: nothing at all — never probed, or its group probe failed and the engine
|
||||
// deleted the entry. Indistinguishable, therefore: untested.
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{LastOK: now.Add(-time.Minute), Delay: 100})
|
||||
// n2 is DEAD: a failed probe/dial marks LastFail on the board record — nothing
|
||||
// is deleted, so a dead member stays distinguishable from a never-measured one.
|
||||
hist.StoreURLTestHistory("n2", &adapter.URLTestHistory{LastFail: now.Add(-10 * time.Second)})
|
||||
// n3: nothing at all — never probed (outside every used path). Untested.
|
||||
|
||||
g := &fakeGroup{tag: "auto", kind: C.TypeURLTest, all: []string{"n1", "n2", "n3"}, now: "n1"}
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, dead, now), true)
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, healthTTLFloor, now), nil, true)
|
||||
if !gh.Used {
|
||||
t.Error("a nil used-set must report Used=true (no badge is better than a wrong one)")
|
||||
}
|
||||
|
||||
if gh.Bound {
|
||||
t.Error("unbound group reported bound=true")
|
||||
@@ -139,16 +140,17 @@ func TestGroupHealthBound(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
now := time.Now()
|
||||
// Base tags say everything is fine (the nodes answer on the plain WAN)...
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{Time: now, Delay: 50})
|
||||
hist.StoreURLTestHistory("n2", &adapter.URLTestHistory{Time: now, Delay: 60})
|
||||
hist.StoreURLTestHistory("n3", &adapter.URLTestHistory{Time: now, Delay: 70})
|
||||
hist.StoreURLTestHistory("n1", &adapter.URLTestHistory{LastOK: now, Delay: 50})
|
||||
hist.StoreURLTestHistory("n2", &adapter.URLTestHistory{LastOK: now, Delay: 60})
|
||||
hist.StoreURLTestHistory("n3", &adapter.URLTestHistory{LastOK: now, Delay: 70})
|
||||
// ...but THROUGH the group's egress only n1 answers, n2 is a confirmed failure and
|
||||
// n3 was never measured. Note the sparse indices: generate/group.go skips members
|
||||
// whose copy cannot be built, so the index is not a dense 0..n-1 and must not be
|
||||
// used to look members up positionally.
|
||||
hist.StoreURLTestHistory(copyTag("vpn", 0, "n1"),
|
||||
&adapter.URLTestHistory{Time: now.Add(-2 * time.Second), Delay: 480})
|
||||
dead := map[string]time.Time{copyTag("vpn", 4, "n2"): now.Add(-time.Second)}
|
||||
&adapter.URLTestHistory{LastOK: now.Add(-2 * time.Second), Delay: 480})
|
||||
hist.StoreURLTestHistory(copyTag("vpn", 4, "n2"),
|
||||
&adapter.URLTestHistory{LastFail: now.Add(-time.Second)})
|
||||
|
||||
g := &fakeGroup{
|
||||
tag: "vpn",
|
||||
@@ -160,7 +162,12 @@ func TestGroupHealthBound(t *testing.T) {
|
||||
},
|
||||
now: copyTag("vpn", 0, "n1"),
|
||||
}
|
||||
gh := groupHealthOf(g, C.TypeSelector, newHealthView(hist, dead, now), true)
|
||||
// The used-set names this group: Used=true, and an unrelated name changes nothing.
|
||||
used := map[string]bool{"vpn": true}
|
||||
gh := groupHealthOf(g, C.TypeSelector, newHealthView(hist, healthTTLFloor, now), used, true)
|
||||
if !gh.Used {
|
||||
t.Error("group named in the used-set reported Used=false")
|
||||
}
|
||||
|
||||
if !gh.Bound {
|
||||
t.Error("egress-bound group reported bound=false")
|
||||
@@ -189,7 +196,7 @@ func TestGroupHealthBound(t *testing.T) {
|
||||
func TestGroupHealthSummaryOmitsMembers(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
g := &fakeGroup{tag: "auto", kind: C.TypeURLTest, all: []string{"n1", "n2"}}
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, nil, time.Now()), false)
|
||||
gh := groupHealthOf(g, C.TypeURLTest, newHealthView(hist, 0, time.Now()), nil, false)
|
||||
if gh.Members != nil {
|
||||
t.Errorf("summary carried %d member rows, want none", len(gh.Members))
|
||||
}
|
||||
@@ -199,6 +206,21 @@ func TestGroupHealthSummaryOmitsMembers(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupHealthUnused: a group the observatory's used-set does NOT name is
|
||||
// reported Used=false — the panel renders it "unused" instead of health counters
|
||||
// (its members are deliberately never probed and stay untested by design).
|
||||
func TestGroupHealthUnused(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
g := &fakeGroup{tag: "idle", kind: C.TypeSelector, all: []string{"n1", "n2"}}
|
||||
used := map[string]bool{"someOtherGroup": true}
|
||||
gh := groupHealthOf(g, C.TypeSelector, newHealthView(hist, 0, time.Now()), used, true)
|
||||
if gh.Used {
|
||||
t.Fatal("group absent from the used-set reported Used=true")
|
||||
}
|
||||
// The counters are still honest: everything untested, nothing invented.
|
||||
assertCounters(t, gh, 2, 0, 0, 2)
|
||||
}
|
||||
|
||||
// TestGroupHealthNilEngine: the read is nil-safe end to end and always yields a
|
||||
// non-nil slice (the panel maps over it unconditionally).
|
||||
func TestGroupHealthNilEngine(t *testing.T) {
|
||||
@@ -234,3 +256,61 @@ func assertCounters(t *testing.T, gh GroupHealth, total, alive, dead, untested i
|
||||
gh.Alive+gh.Dead+gh.Untested, gh.Total)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// TestUsedChainNames pins the chain-name recovery from the observatory's tag-keyed
|
||||
// used-set: only exit-wrapper tags "chain-<name>-h<digits>" yield a name, and every
|
||||
// other tag the set can hold (group tags, per-group egress copies, chain member
|
||||
// copies, plain node tags) is left out. nil ⇒ nil (the caller's "everything used").
|
||||
func TestUsedChainNames(t *testing.T) {
|
||||
if got := usedChainNames(nil); got != nil {
|
||||
t.Fatalf("usedChainNames(nil) = %v, want nil (everything used sentinel)", got)
|
||||
}
|
||||
used := map[string]bool{
|
||||
"chain-relay-h2": true, // a 2-hop chain "relay"'s exit wrapper
|
||||
"chain-relay-h1": true, // its group-hop wrapper (used, not an exit tag — still parses to "relay")
|
||||
"chain-a-h2-h1": true, // chain literally named "a-h2" (parseChainExitTag takes the last -h)
|
||||
"chain-relay-h1-nodeA": false, // a member copy — must NOT parse as an exit, so no false "relay"/"nodeA"
|
||||
"auto": true, // a group tag — not a chain
|
||||
"group-vpn-m0-n1": true, // a per-group egress copy — not a chain
|
||||
"n1": true, // a plain node tag — not a chain
|
||||
"chain-c": true, // no hop suffix — not an exit tag
|
||||
"chain--h1": true, // empty name — not an exit tag
|
||||
}
|
||||
got := usedChainNames(used)
|
||||
// Only the three real exit/group wrapper tags parse to a chain name. A group-hop
|
||||
// wrapper ("chain-relay-h1") is NOT an exit, but parseChainExitTag recovers the
|
||||
// same name from it (it only checks the -h<digits> suffix, not topological
|
||||
// exits-ness), so "relay" is covered either way — which is what the badge needs.
|
||||
want := map[string]bool{"relay": true, "a-h2": true}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("usedChainNames = %v, want %v", got, want)
|
||||
}
|
||||
for name := range want {
|
||||
if !got[name] {
|
||||
t.Errorf("usedChainNames missing chain %q (got %v)", name, got)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainHealthNilUsedSet: a stopped engine (no observatory configured) yields a
|
||||
// nil used-set, and ChainHealth reports every named chain Used=true — no badge is
|
||||
// better than a wrong one, the same convention as GroupHealth.Used. Names are
|
||||
// echoed back in order, blanks dropped.
|
||||
func TestChainHealthNilUsedSet(t *testing.T) {
|
||||
e := New()
|
||||
got := e.ChainHealth([]string{"relay", "", " unused ", "single"})
|
||||
want := []ChainHealth{
|
||||
{Name: "relay", Used: true},
|
||||
{Name: "unused", Used: true},
|
||||
{Name: "single", Used: true},
|
||||
}
|
||||
if len(got) != len(want) {
|
||||
t.Fatalf("ChainHealth = %+v, want %+v (nil used-set ⇒ all used, blanks dropped)", got, want)
|
||||
}
|
||||
for i, w := range want {
|
||||
if got[i] != w {
|
||||
t.Errorf("ChainHealth[%d] = %+v, want %+v", i, got[i], w)
|
||||
}
|
||||
}
|
||||
}
|
||||
+244
-56
@@ -16,27 +16,29 @@ import (
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// Group test — "what am I actually going out through, and how fast" (F2).
|
||||
// Exit test — "what am I actually going out through, and how fast" (F2, plan §5.E).
|
||||
//
|
||||
// # Why this is not just another node probe
|
||||
//
|
||||
// TestAllNodes (probeall.go) answers "which nodes are alive". It cannot answer the
|
||||
// question an operator actually asks after switching a group: *through which
|
||||
// address am I leaving the country right now*. A group is an indirection — a
|
||||
// selector or urltest over many members — so the delay of the group and the public
|
||||
// address it exits from are properties of the CURRENT selection, not of any node
|
||||
// the panel can point at.
|
||||
// The observatory (observatory.go) answers "which nodes are alive". It cannot
|
||||
// answer the question an operator actually asks after switching a group: *through
|
||||
// which address am I leaving the country right now*. A group is an indirection —
|
||||
// a selector or urltest over many members — so the delay of the group and the
|
||||
// public address it exits from are properties of the CURRENT selection, not of
|
||||
// any node the panel can point at. A CHAIN is the same question over a longer
|
||||
// path: its exit wrapper "chain-<name>-hN" tunnels through every hop, so dialling
|
||||
// it measures the whole L1..Ln path end to end.
|
||||
//
|
||||
// So one test per group measures two things:
|
||||
// So one test per target (group or chain) measures two things:
|
||||
//
|
||||
// delay_ms — reusing urltest.URLTest, the SAME primitive probeall.go uses for a
|
||||
// node. There is deliberately no second latency mechanism here: the
|
||||
// group outbound is just an outbound, and probing it exercises exactly
|
||||
// the path traffic will take (selector -> selected member -> server).
|
||||
// exit_ip — a real HTTP request THROUGH the group to a service that echoes the
|
||||
// client address back. Nothing else can produce this number: the router
|
||||
// cannot know its own public address, and the proxy protocol does not
|
||||
// report it.
|
||||
// delay_ms — reusing urltest.URLTest, the SAME primitive the observatory uses
|
||||
// for a node. There is deliberately no second latency mechanism
|
||||
// here: the target is just an outbound, and probing it exercises
|
||||
// exactly the path traffic will take.
|
||||
// exit_ip — a real HTTP request THROUGH the target to a service that echoes
|
||||
// the client address back. Nothing else can produce this number: the
|
||||
// router cannot know its own public address, and the proxy protocol
|
||||
// does not report it.
|
||||
//
|
||||
// # The direct-egress trap
|
||||
//
|
||||
@@ -66,12 +68,17 @@ const (
|
||||
exitBodyLimit = 4 << 10
|
||||
)
|
||||
|
||||
// GroupTestResult is one group's test outcome. The JSON tags are the panel
|
||||
// contract — see the shater API docs for /api/groups/test.
|
||||
// GroupTestResult is one target's test outcome — a group's or a chain's (Group
|
||||
// then carries the chain's model name). The JSON tags are the panel contract —
|
||||
// see the shater API docs for /api/groups/test.
|
||||
//
|
||||
// Selected is the group's current pick (OutboundGroup.Now()); for a chain it is
|
||||
// the node NAME the chain's last group hop currently selects, "" when the chain
|
||||
// has no group hop (a fixed path selects nothing).
|
||||
//
|
||||
// OK reports whether the LATENCY measurement succeeded, which is the test's primary
|
||||
// question. A failed exit-address lookup deliberately does NOT clear it: knowing the
|
||||
// group is up and fast is useful on its own, and a probe service being unreachable
|
||||
// target is up and fast is useful on its own, and a probe service being unreachable
|
||||
// says nothing about the tunnel. In that case OK stays true and ExitIP is empty —
|
||||
// "not determined", never a guess and never somebody else's address.
|
||||
type GroupTestResult struct {
|
||||
@@ -88,7 +95,8 @@ type GroupTestResult struct {
|
||||
// isGroupOutbound reports whether an outbound is a user-facing GROUP worth testing:
|
||||
// a selector/urltest that is not one of the generator's internal copies.
|
||||
//
|
||||
// The prefix exclusions mirror shouldProbeOutbound (probeall.go): "chain-<name>-h<i>"
|
||||
// The prefix exclusions mirror the reserved tag namespaces (see probeablePlanTag):
|
||||
// "chain-<name>-h<i>"
|
||||
// is a per-chain hop wrapper and "group-<name>-m<i>-<member>" a per-group egress
|
||||
// copy — both are implementation detail of a target the operator DID configure, and
|
||||
// testing them would report tags that appear nowhere in the UI.
|
||||
@@ -101,19 +109,167 @@ func isGroupOutbound(typ, tag string) bool {
|
||||
return !strings.HasPrefix(tag, "chain-") && !strings.HasPrefix(tag, "group-")
|
||||
}
|
||||
|
||||
// TestGroups launches a one-shot background test of the named groups (empty/nil =
|
||||
// every group in the running box), returning started=false when a run is already in
|
||||
// flight.
|
||||
// groupTestTarget is one thing a run measures: a user-facing GROUP (name == the
|
||||
// group outbound's tag) or a CHAIN (name == the chain's model name, dialled via
|
||||
// its exit wrapper). sel reports the target's current selection at dial time,
|
||||
// nil when the target selects nothing.
|
||||
type groupTestTarget struct {
|
||||
name string
|
||||
ob adapter.Outbound
|
||||
sel func() string
|
||||
}
|
||||
|
||||
// chainTargetsFrom discovers each materialised chain in the running
|
||||
// outbound/endpoint set and returns one target per chain, dialled via its EXIT
|
||||
// wrapper.
|
||||
//
|
||||
// It is a SINGLETON on the same pattern as TestAllNodes: a second request while a
|
||||
// run is in flight is refused rather than queued or run in parallel, because these
|
||||
// runs open real tunnelled connections and a panel that double-fires a button must
|
||||
// not multiply the load on the uplink.
|
||||
// Discovery is topological, not name parsing: among the "chain-" tagged
|
||||
// outbounds, a chain's exit is the one no other chain outbound depends on (every
|
||||
// other hop wrapper and member copy is somebody's Detour/member dependency). The
|
||||
// chain NAME is then recovered from the exit tag "chain-<name>-h<digits>"; a
|
||||
// name crafted to collide with another chain's hop namespace is the same
|
||||
// documented edge case parseGroupCopyTag accepts.
|
||||
//
|
||||
// A single-hop chain is NOT discoverable: the generator resolves it straight to
|
||||
// the underlying node/group tag with no wrapper (generate/chain.go buildChain),
|
||||
// so nothing chain-tagged exists in the box for it. Such a chain is reported
|
||||
// missing when named — its alias (the node/group itself) is the thing to test.
|
||||
func chainTargetsFrom(pool []adapter.Outbound) []groupTestTarget {
|
||||
byTag := map[string]adapter.Outbound{}
|
||||
for _, ob := range pool {
|
||||
if strings.HasPrefix(ob.Tag(), "chain-") {
|
||||
byTag[ob.Tag()] = ob
|
||||
}
|
||||
}
|
||||
if len(byTag) == 0 {
|
||||
return nil
|
||||
}
|
||||
referenced := map[string]bool{}
|
||||
for _, ob := range byTag {
|
||||
for _, dep := range ob.Dependencies() {
|
||||
if _, ok := byTag[dep]; ok {
|
||||
referenced[dep] = true
|
||||
}
|
||||
}
|
||||
}
|
||||
var out []groupTestTarget
|
||||
for tag, ob := range byTag {
|
||||
if referenced[tag] {
|
||||
continue
|
||||
}
|
||||
name, ok := parseChainExitTag(tag)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
t := groupTestTarget{name: name, ob: ob}
|
||||
if hop := chainLastGroupHop(byTag, name); hop != nil {
|
||||
t.sel = func() string { return chainMemberName(hop.Now(), name) }
|
||||
}
|
||||
out = append(out, t)
|
||||
}
|
||||
sort.Slice(out, func(i, j int) bool { return out[i].name < out[j].name })
|
||||
return out
|
||||
}
|
||||
|
||||
// parseChainExitTag recovers the chain name from an exit wrapper tag
|
||||
// "chain-<name>-h<digits>". Member copies ("…-h<i>-<member>") and non-chain tags
|
||||
// parse false.
|
||||
func parseChainExitTag(tag string) (string, bool) {
|
||||
rest, ok := strings.CutPrefix(tag, "chain-")
|
||||
if !ok {
|
||||
return "", false
|
||||
}
|
||||
i := strings.LastIndex(rest, "-h")
|
||||
if i <= 0 {
|
||||
return "", false
|
||||
}
|
||||
if _, ok := parseAllDigits(rest[i+2:]); !ok {
|
||||
return "", false
|
||||
}
|
||||
return rest[:i], true
|
||||
}
|
||||
|
||||
// chainLastGroupHop finds the chain's LAST group hop — the wrapper selector
|
||||
// "chain-<name>-h<i>" with the largest hop index that is a group outbound. That
|
||||
// hop's Now() is the only selection a chain has (plan §5.E); nil when the chain
|
||||
// is a fixed node path.
|
||||
func chainLastGroupHop(byTag map[string]adapter.Outbound, name string) adapter.OutboundGroup {
|
||||
prefix := "chain-" + name + "-h"
|
||||
best := -1
|
||||
var bestG adapter.OutboundGroup
|
||||
for tag, ob := range byTag {
|
||||
rest, ok := strings.CutPrefix(tag, prefix)
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
idx, ok := parseAllDigits(rest)
|
||||
if !ok {
|
||||
continue // a member copy, or another chain sharing the prefix
|
||||
}
|
||||
g, isGroup := ob.(adapter.OutboundGroup)
|
||||
if !isGroup {
|
||||
continue
|
||||
}
|
||||
if idx > best {
|
||||
best, bestG = idx, g
|
||||
}
|
||||
}
|
||||
return bestG
|
||||
}
|
||||
|
||||
// chainMemberName maps a chain member copy tag "chain-<name>-h<i>-<member>" back
|
||||
// to the member's node name (the name is carried in the tag itself, exactly as
|
||||
// with per-group egress copies). Anything else is returned verbatim — a name the
|
||||
// panel can at least display.
|
||||
func chainMemberName(tag, chain string) string {
|
||||
rest, ok := strings.CutPrefix(tag, "chain-"+chain+"-h")
|
||||
if !ok {
|
||||
return tag
|
||||
}
|
||||
j := strings.IndexByte(rest, '-')
|
||||
if j <= 0 {
|
||||
return tag
|
||||
}
|
||||
if _, ok := parseAllDigits(rest[:j]); !ok {
|
||||
return tag
|
||||
}
|
||||
if member := rest[j+1:]; member != "" {
|
||||
return member
|
||||
}
|
||||
return tag
|
||||
}
|
||||
|
||||
// parseAllDigits parses a non-empty all-digit string.
|
||||
func parseAllDigits(s string) (int, bool) {
|
||||
if s == "" {
|
||||
return 0, false
|
||||
}
|
||||
n := 0
|
||||
for i := range len(s) {
|
||||
c := s[i]
|
||||
if c < '0' || c > '9' {
|
||||
return 0, false
|
||||
}
|
||||
n = n*10 + int(c-'0')
|
||||
}
|
||||
return n, true
|
||||
}
|
||||
|
||||
// TestGroups launches a one-shot background test of the named targets — groups
|
||||
// and chains (empty/nil = every group and every chain in the running box),
|
||||
// returning started=false when a run is already in flight.
|
||||
//
|
||||
// It is a SINGLETON: a second request while a run is in flight is refused rather
|
||||
// than queued or run in parallel, because these runs open real tunnelled
|
||||
// connections and a panel that double-fires a button must not multiply the load
|
||||
// on the uplink. The observatory checks the same guard and skips its tick while
|
||||
// a run is in flight (observatory.go), so a manual test never competes with
|
||||
// background probing for the uplink.
|
||||
//
|
||||
// probeURL is the latency-probe URL; "" falls back to urltest's gstatic default.
|
||||
//
|
||||
// Apply-swap safety: the target outbounds are snapshotted up front, so a config swap
|
||||
// mid-run cannot change what is being tested. A group torn down mid-run simply fails
|
||||
// mid-run cannot change what is being tested. A target torn down mid-run simply fails
|
||||
// its probe and is reported not-ok.
|
||||
func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
if !e.groupTestRunning.CompareAndSwap(false, true) {
|
||||
@@ -133,13 +289,14 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
defer e.groupTestRunning.Store(false)
|
||||
|
||||
results := make([]GroupTestResult, len(targets)+len(missing))
|
||||
// A group the caller named that the running box does not have is a RESULT,
|
||||
// not a silent omission: the operator asked about it and deserves to be told
|
||||
// it is not there (typo, not applied yet, or dropped for having no members).
|
||||
// A name the caller asked for that the running box has neither as a group
|
||||
// nor as a chain is a RESULT, not a silent omission: the operator asked
|
||||
// about it and deserves to be told it is not there (typo, not applied yet,
|
||||
// or dropped for having no usable members/hops).
|
||||
for i, name := range missing {
|
||||
results[len(targets)+i] = GroupTestResult{
|
||||
Group: name,
|
||||
Error: "no such group in the running engine",
|
||||
Error: "no such group or chain in the running engine",
|
||||
TestedUnix: time.Now().Unix(),
|
||||
}
|
||||
}
|
||||
@@ -148,27 +305,30 @@ func (e *Engine) TestGroups(names []string, probeURL string) (started bool) {
|
||||
|
||||
sem := make(chan struct{}, groupTestConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for i, ob := range targets {
|
||||
for i, tgt := range targets {
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(i int, ob adapter.Outbound) {
|
||||
go func(i int, tgt groupTestTarget) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
res := e.testOneGroup(ob, probeURL)
|
||||
res := e.testOneTarget(tgt, probeURL)
|
||||
e.storeGroupTestResult(i, res)
|
||||
e.groupTestDone.Add(1)
|
||||
}(i, ob)
|
||||
}(i, tgt)
|
||||
}
|
||||
wg.Wait()
|
||||
}()
|
||||
return true
|
||||
}
|
||||
|
||||
// groupTargets resolves the requested group names against the running box,
|
||||
// returning the outbounds to test and the names that do not exist. An empty/nil
|
||||
// names list means "every group"; a stopped engine yields no targets (and every
|
||||
// explicitly named group as missing, so the caller is told why).
|
||||
func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missing []string) {
|
||||
// groupTargets resolves the requested names against the running box — groups
|
||||
// first, then chains — returning the targets to test and the names that exist as
|
||||
// neither. An empty/nil names list means "every group and every chain"; a
|
||||
// stopped engine yields no targets (and every explicitly named target as
|
||||
// missing, so the caller is told why). A group and a chain sharing one name
|
||||
// cannot collide: the generator's chain tags live in a reserved namespace, and
|
||||
// the group match wins deterministically.
|
||||
func (e *Engine) groupTargets(names []string) (targets []groupTestTarget, missing []string) {
|
||||
var want map[string]bool
|
||||
var order []string
|
||||
for _, n := range names {
|
||||
@@ -188,7 +348,13 @@ func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missi
|
||||
found := map[string]bool{}
|
||||
if inst := e.Instance(); inst != nil {
|
||||
if om := inst.Outbound(); om != nil {
|
||||
// The pool for chain discovery must include endpoints: a chain hop
|
||||
// rebuilt from a wireguard/AWG node is an ENDPOINT copy, invisible in
|
||||
// Outbounds() — an exit materialised that way would otherwise vanish
|
||||
// from the run.
|
||||
var pool []adapter.Outbound
|
||||
for _, ob := range om.Outbounds() {
|
||||
pool = append(pool, ob)
|
||||
if !isGroupOutbound(ob.Type(), ob.Tag()) {
|
||||
continue
|
||||
}
|
||||
@@ -196,7 +362,26 @@ func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missi
|
||||
continue
|
||||
}
|
||||
found[ob.Tag()] = true
|
||||
targets = append(targets, ob)
|
||||
tgt := groupTestTarget{name: ob.Tag(), ob: ob}
|
||||
if g, ok := ob.(adapter.OutboundGroup); ok {
|
||||
tgt.sel = g.Now
|
||||
}
|
||||
targets = append(targets, tgt)
|
||||
}
|
||||
if em := inst.Endpoint(); em != nil {
|
||||
for _, ep := range em.Endpoints() {
|
||||
pool = append(pool, ep)
|
||||
}
|
||||
}
|
||||
for _, ct := range chainTargetsFrom(pool) {
|
||||
if want != nil && !want[ct.name] {
|
||||
continue
|
||||
}
|
||||
if found[ct.name] {
|
||||
continue
|
||||
}
|
||||
found[ct.name] = true
|
||||
targets = append(targets, ct)
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -208,16 +393,18 @@ func (e *Engine) groupTargets(names []string) (targets []adapter.Outbound, missi
|
||||
return targets, missing
|
||||
}
|
||||
|
||||
// testOneGroup measures one group: which member it currently selects, the latency
|
||||
// through it, and the public address it exits from.
|
||||
func (e *Engine) testOneGroup(ob adapter.Outbound, probeURL string) GroupTestResult {
|
||||
res := GroupTestResult{Group: ob.Tag(), TestedUnix: time.Now().Unix()}
|
||||
if g, ok := ob.(adapter.OutboundGroup); ok {
|
||||
res.Selected = g.Now()
|
||||
// testOneTarget measures one target: what it currently selects, the latency
|
||||
// through it, and the public address it exits from. For a chain the dialled
|
||||
// outbound is the exit wrapper, so the delay and the exit address are end-to-end
|
||||
// properties of the whole L1..Ln path.
|
||||
func (e *Engine) testOneTarget(t groupTestTarget, probeURL string) GroupTestResult {
|
||||
res := GroupTestResult{Group: t.name, TestedUnix: time.Now().Unix()}
|
||||
if t.sel != nil {
|
||||
res.Selected = t.sel()
|
||||
}
|
||||
|
||||
ctx, cancel := context.WithTimeout(context.Background(), groupTestDelayTimeout)
|
||||
delay, err := urltest.URLTest(ctx, probeURL, ob)
|
||||
delay, err := urltest.URLTest(ctx, probeURL, t.ob)
|
||||
cancel()
|
||||
if err != nil {
|
||||
res.Error = err.Error()
|
||||
@@ -227,7 +414,7 @@ func (e *Engine) testOneGroup(ob adapter.Outbound, probeURL string) GroupTestRes
|
||||
res.DelayMs = int(delay)
|
||||
|
||||
// The exit address is best-effort by design: see GroupTestResult.OK.
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(ob)
|
||||
res.ExitIP, res.ExitCountry = e.exitAddress(t.ob)
|
||||
return res
|
||||
}
|
||||
|
||||
@@ -387,13 +574,14 @@ func (e *Engine) GroupTestStatus() (running bool, done, total int, scope []strin
|
||||
e.groupTestResults()
|
||||
}
|
||||
|
||||
// scopeOf is the set of group names a run covers: everything it will produce a result
|
||||
// for, whether that result is a measurement or a "no such group". Sorted so the panel
|
||||
// sees a stable order and two polls of the same run never differ.
|
||||
func scopeOf(targets []adapter.Outbound, missing []string) []string {
|
||||
// scopeOf is the set of target names a run covers — group and chain alike:
|
||||
// everything it will produce a result for, whether that result is a measurement
|
||||
// or a "no such group or chain". Sorted so the panel sees a stable order and two
|
||||
// polls of the same run never differ.
|
||||
func scopeOf(targets []groupTestTarget, missing []string) []string {
|
||||
out := make([]string, 0, len(targets)+len(missing))
|
||||
for _, ob := range targets {
|
||||
out = append(out, ob.Tag())
|
||||
for _, t := range targets {
|
||||
out = append(out, t.name)
|
||||
}
|
||||
out = append(out, missing...)
|
||||
sort.Strings(out)
|
||||
|
||||
@@ -1,13 +1,17 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"context"
|
||||
"crypto/tls"
|
||||
"net"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
)
|
||||
|
||||
// TestIsGroupOutbound pins which outbounds the group test considers a testable
|
||||
@@ -34,6 +38,172 @@ func TestIsGroupOutbound(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// depOutbound is a failingOutbound with declared dependencies, for the chain
|
||||
// discovery tests (the generator's hop copies depend on the previous hop's tag).
|
||||
type depOutbound struct {
|
||||
failingOutbound
|
||||
deps []string
|
||||
}
|
||||
|
||||
func (d *depOutbound) Dependencies() []string { return d.deps }
|
||||
|
||||
// depGroup is a fakeGroup with declared dependencies — a chain's group-hop
|
||||
// wrapper selector.
|
||||
type depGroup struct {
|
||||
fakeGroup
|
||||
deps []string
|
||||
}
|
||||
|
||||
func (d *depGroup) Dependencies() []string { return d.deps }
|
||||
|
||||
// TestParseChainExitTag pins the exit-tag format "chain-<name>-h<digits>",
|
||||
// including a chain whose NAME itself ends in "-h<digits>" (the last "-h" wins)
|
||||
// and the member-copy shape that must NOT parse as an exit.
|
||||
func TestParseChainExitTag(t *testing.T) {
|
||||
cases := []struct {
|
||||
tag string
|
||||
name string
|
||||
ok bool
|
||||
}{
|
||||
{"chain-c-h2", "c", true},
|
||||
{"chain-my chain-h10", "my chain", true},
|
||||
{"chain-a-h2-h1", "a-h2", true}, // chain literally named "a-h2"
|
||||
{"chain-c-h2-member", "", false}, // a member copy, not an exit
|
||||
{"chain-c", "", false}, // no hop suffix
|
||||
{"auto", "", false}, // not a chain tag at all
|
||||
{"chain--h1", "", false}, // empty name
|
||||
{"group-vpn-m0-chain-x-h1", "", false}, // wrong namespace
|
||||
}
|
||||
for _, c := range cases {
|
||||
name, ok := parseChainExitTag(c.tag)
|
||||
if name != c.name || ok != c.ok {
|
||||
t.Errorf("parseChainExitTag(%q) = (%q,%v), want (%q,%v)", c.tag, name, ok, c.name, c.ok)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainTargetsFrom: discovery is topological — the exit is the chain outbound
|
||||
// nobody else in the chain namespace depends on. A chain with a group hop reports
|
||||
// that hop's current pick mapped back to the NODE NAME; a fixed chain selects
|
||||
// nothing; non-chain outbounds are ignored.
|
||||
func TestChainTargetsFrom(t *testing.T) {
|
||||
pool := []adapter.Outbound{
|
||||
// chain "c": exit h2 (plain copy) -> group hop h1 over two member copies.
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h2"}, []string{"chain-c-h1"}},
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-c-h1", kind: C.TypeSelector,
|
||||
all: []string{"chain-c-h1-nodeA", "chain-c-h1-nodeB"},
|
||||
now: "chain-c-h1-nodeA",
|
||||
}, []string{"chain-c-h1-nodeA", "chain-c-h1-nodeB"}},
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h1-nodeA"}, nil},
|
||||
&depOutbound{failingOutbound{tag: "chain-c-h1-nodeB"}, nil},
|
||||
// chain "fix": two plain node hops, no group — selects nothing.
|
||||
&depOutbound{failingOutbound{tag: "chain-fix-h2"}, []string{"chain-fix-h1"}},
|
||||
&depOutbound{failingOutbound{tag: "chain-fix-h1"}, nil},
|
||||
// Not a chain: must be ignored entirely.
|
||||
&depOutbound{failingOutbound{tag: "auto"}, []string{"n1"}},
|
||||
}
|
||||
targets := chainTargetsFrom(pool)
|
||||
if len(targets) != 2 {
|
||||
t.Fatalf("got %d chain targets, want 2: %+v", len(targets), targets)
|
||||
}
|
||||
// Sorted by name: "c" then "fix".
|
||||
c, fix := targets[0], targets[1]
|
||||
if c.name != "c" || c.ob.Tag() != "chain-c-h2" {
|
||||
t.Fatalf("chain c = (%q, %q), want dialled via its exit chain-c-h2", c.name, c.ob.Tag())
|
||||
}
|
||||
if c.sel == nil {
|
||||
t.Fatal("chain c has a group hop; sel must report its pick")
|
||||
}
|
||||
if got := c.sel(); got != "nodeA" {
|
||||
t.Errorf("chain c selected = %q, want the NODE NAME nodeA (not the copy tag)", got)
|
||||
}
|
||||
if fix.name != "fix" || fix.ob.Tag() != "chain-fix-h2" {
|
||||
t.Fatalf("chain fix = (%q, %q), want exit chain-fix-h2", fix.name, fix.ob.Tag())
|
||||
}
|
||||
if fix.sel != nil {
|
||||
t.Error("a fixed chain selects nothing; sel must be nil")
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainMemberName pins the copy-tag → node-name mapping the panel displays.
|
||||
func TestChainMemberName(t *testing.T) {
|
||||
cases := []struct {
|
||||
tag, chain, want string
|
||||
}{
|
||||
{"chain-c-h1-🇩🇪 Frankfurt-01", "c", "🇩🇪 Frankfurt-01"},
|
||||
{"chain-c-h1-007", "c", "007"}, // an all-digit node name survives
|
||||
{"chain-c-h1", "c", "chain-c-h1"},
|
||||
{"unrelated", "c", "unrelated"},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := chainMemberName(c.tag, c.chain); got != c.want {
|
||||
t.Errorf("chainMemberName(%q,%q) = %q, want %q", c.tag, c.chain, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// dialableOutbound routes every dial to a fixed local address — a stand-in for a
|
||||
// chain exit whose whole path is up. urltest.URLTest dials the probe URL's host
|
||||
// through the outbound, so pointing every dial at a local HTTP server makes the
|
||||
// latency probe succeed without any network.
|
||||
type dialableOutbound struct {
|
||||
failingOutbound
|
||||
addr string
|
||||
deps []string
|
||||
}
|
||||
|
||||
func (d *dialableOutbound) Dependencies() []string { return d.deps }
|
||||
func (d *dialableOutbound) DialContext(ctx context.Context, network string, _ M.Socksaddr) (net.Conn, error) {
|
||||
return (&net.Dialer{}).DialContext(ctx, network, d.addr)
|
||||
}
|
||||
|
||||
// TestChainExitTestMeasuresEndToEnd is the §6-S4 acceptance path for chains: the
|
||||
// exit test dials the chain's EXIT TAG, returns a measured delay, carries the
|
||||
// chain's model name (not the wrapper tag) as the result's Group, and reports the
|
||||
// last group hop's pick as Selected. The exit address is measured through the
|
||||
// same outbound (unreachable from a test => empty, never a guess).
|
||||
func TestChainExitTestMeasuresEndToEnd(t *testing.T) {
|
||||
probe := httptest.NewServer(http.HandlerFunc(func(w http.ResponseWriter, r *http.Request) {
|
||||
w.WriteHeader(http.StatusNoContent)
|
||||
}))
|
||||
defer probe.Close()
|
||||
|
||||
exit := &dialableOutbound{
|
||||
failingOutbound: failingOutbound{tag: "chain-x-h2"},
|
||||
addr: probe.Listener.Addr().String(),
|
||||
deps: []string{"chain-x-h1"},
|
||||
}
|
||||
pool := []adapter.Outbound{
|
||||
exit,
|
||||
&depGroup{fakeGroup{
|
||||
tag: "chain-x-h1", kind: C.TypeSelector,
|
||||
all: []string{"chain-x-h1-relay"},
|
||||
now: "chain-x-h1-relay",
|
||||
}, []string{"chain-x-h1-relay"}},
|
||||
&depOutbound{failingOutbound{tag: "chain-x-h1-relay"}, nil},
|
||||
}
|
||||
targets := chainTargetsFrom(pool)
|
||||
if len(targets) != 1 || targets[0].ob.Tag() != "chain-x-h2" {
|
||||
t.Fatalf("targets = %+v, want the chain dialled via its exit tag", targets)
|
||||
}
|
||||
|
||||
e := New()
|
||||
res := e.testOneTarget(targets[0], probe.URL)
|
||||
if res.Group != "x" {
|
||||
t.Errorf("Group = %q, want the chain's model name x", res.Group)
|
||||
}
|
||||
if !res.OK || res.Error != "" {
|
||||
t.Fatalf("result = %+v, want a successful measurement through the exit tag (a local roundtrip may legitimately read 0ms)", res)
|
||||
}
|
||||
if res.Selected != "relay" {
|
||||
t.Errorf("Selected = %q, want the last group hop's pick relay", res.Selected)
|
||||
}
|
||||
if res.ExitIP != "" {
|
||||
t.Errorf("ExitIP = %q, want empty — the exit services are unreachable here and must never be guessed", res.ExitIP)
|
||||
}
|
||||
}
|
||||
|
||||
// TestGroupTestSingleton is the "no parallel runs" invariant: while a run holds the
|
||||
// guard, a second TestGroups must be refused rather than starting a concurrent run
|
||||
// (each of these opens real tunnelled connections).
|
||||
|
||||
+74
-160
@@ -7,155 +7,121 @@ import (
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
)
|
||||
|
||||
// Node health = the engine's MEASUREMENTS + our own FAILURE verdicts, kept apart.
|
||||
// Node health = a read of the shared HEALTH BOARD (common/urltest), classified
|
||||
// against a TTL.
|
||||
//
|
||||
// # Why the two are separate stores
|
||||
// # One store, one truth
|
||||
//
|
||||
// The engine's urltest.HistoryStorage is the engine's own truth, and it holds exactly
|
||||
// one kind of fact: "this outbound answered a probe in N ms". A failed probe is not
|
||||
// represented there at all — protocol/group/urltest.go DELETES the entry (urltest.go
|
||||
// :512) rather than recording the failure. So "probed and dead" and "never probed"
|
||||
// look identical in it, which is precisely the distinction the panel needs.
|
||||
// The board's entry shape is {LastOK, Delay, LastFail}: a success is stored by
|
||||
// StoreURLTestHistory (preserving the last failure), a failure by MarkFailed
|
||||
// (preserving the last success for display), and NOTHING is deleted on failure —
|
||||
// deletion stays reserved for nodes removed from the config. Whoever learns
|
||||
// about a death — a group's own checker, the observatory, a failed user dial —
|
||||
// marks it in the same store the strategies select on, so the panel and the
|
||||
// selection can never disagree about who is dead. The private "dead overlay"
|
||||
// this file used to keep is gone with the problem it worked around.
|
||||
//
|
||||
// The obvious shortcut is to write our own failure INTO that store under a sentinel
|
||||
// delay. We did, and it was wrong. The engine does not treat the store as opaque data
|
||||
// it merely hands back: the round_robin balancer's pool planner reads it as a liveness
|
||||
// oracle, and its test is presence, not value —
|
||||
// # Verdicts are computed on read, against a TTL
|
||||
//
|
||||
// protocol/group/urltest.go, rebuildPool (~:709) and seedPool (~:748):
|
||||
// if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
// results[tag] = candidate{tag: tag, delay: history.Delay, alive: true}
|
||||
// A record is only as good as its age. The TTL is max(3 x global probe interval,
|
||||
// 10 minutes) — three missed refresh periods, floored so a router with a short
|
||||
// interval does not flap everything to "untested" during a brief probe outage:
|
||||
//
|
||||
// so ANY entry, whatever its delay, is a live candidate. Our `failover` strategy is
|
||||
// built on round_robin with a one-slot pool and PoolTolerance 0, whose planner fills a
|
||||
// hole with the first "live" member in order (generate/group.go failoverBalancer). A
|
||||
// sentinel-marked dead node would therefore walk straight into the slot and hold it —
|
||||
// the group would pin itself to a node we had just proved dead, which is the exact
|
||||
// failure failover exists to prevent. Bounded by one probe interval, but that interval
|
||||
// is 30s for failover, and the trigger is an operator pressing "test all".
|
||||
// alive — LastOK newer than LastFail and younger than TTL;
|
||||
// dead — LastFail newer than LastOK and younger than TTL;
|
||||
// untested — nothing fresh enough either way.
|
||||
//
|
||||
// A value chosen so one consumer reads it correctly (0xFFFF is the SLOWEST delay, so
|
||||
// least_test's minimum-delay Select never prefers it) is not safe for a consumer that
|
||||
// only asks whether the key exists. Rather than audit every present and future reader
|
||||
// of the engine's store, we keep our verdict in our own overlay: the engine's history
|
||||
// then contains only real measurements in every mode, and we do not depend on how any
|
||||
// version of the engine interprets what is in it. It also disposes of the least_test
|
||||
// caveat for free — a node with no entry was never a candidate to begin with.
|
||||
//
|
||||
// # Invalidation: timestamps, not subscriptions
|
||||
//
|
||||
// A "dead" verdict must evaporate the moment the node answers again — including when
|
||||
// the answer comes from the GROUP's own probing, which writes to the engine's store
|
||||
// and knows nothing about us. HealthView.State therefore honours our mark only while
|
||||
// it is NEWER than the stored measurement (decodeHealth): any success recorded after
|
||||
// we marked a tag dead wins, whoever recorded it.
|
||||
//
|
||||
// The alternative was urltest.HistoryStorage.AddUpdateHook. It is worse on both counts:
|
||||
// the hook emits a bare struct{} with no tag and no indication of whether it fired for
|
||||
// a store or a delete, so we would have to rescan everything on every DNS-adjacent
|
||||
// write and still could not tell a success from a deletion — and a deletion carries no
|
||||
// timestamp to compare against anyway. Comparing timestamps is strictly more precise,
|
||||
// needs no subscription, no goroutine and no teardown.
|
||||
// decodeHealth below MUST agree with urltest.HistoryStorage.VerdictAt — it is
|
||||
// the same classification plus the delay/age the panel renders. The strategies
|
||||
// consume VerdictAt directly; both read the same record with the same rules.
|
||||
//
|
||||
// # Lifetime
|
||||
//
|
||||
// The overlay lives on the Engine, not on a Box, so it survives an Apply swap exactly
|
||||
// as the pre-registered history store does (see engine.New). Without that, every apply
|
||||
// would silently turn every "dead" back into "untested". It is bounded by pruning to
|
||||
// the set of outbounds the last full run actually probed (pruneProbeFailed).
|
||||
// The board is pre-registered on the engine context (see engine.New), so it
|
||||
// survives an Apply swap: a config change never turns a known-dead node back
|
||||
// into "untested".
|
||||
|
||||
// Health states. A closed set: these three strings are the JSON values the panel
|
||||
// switches on (see GroupMemberHealth.State / NodeHealthStat.State).
|
||||
const (
|
||||
// HealthAlive: a probe SUCCEEDED and no newer failure contradicts it.
|
||||
// HealthAlive: the newest fresh observation is a SUCCESS.
|
||||
HealthAlive = "alive"
|
||||
// HealthDead: a probe RAN and FAILED, and no newer success contradicts it. Always
|
||||
// a positive finding — never inferred from missing data.
|
||||
// HealthDead: the newest fresh observation is a FAILURE — recorded by the
|
||||
// observatory, a group's own checker, or a failed user dial. Always a
|
||||
// positive finding, never inferred from missing data.
|
||||
HealthDead = "dead"
|
||||
// HealthUntested: nothing is known. Either nothing ever probed this outbound, or
|
||||
// a group probed it, failed, and deleted the entry (the engine records failures by
|
||||
// deletion). The two are indistinguishable from the engine's store alone, so this
|
||||
// is reported as "no data" and NEVER as dead — and never as healthy.
|
||||
// HealthUntested: nothing fresh enough is known. Either nothing ever probed
|
||||
// this outbound (it is reachable from no enabled rule — see GroupHealth.Used)
|
||||
// or every observation is older than the TTL. NEVER to be rendered as
|
||||
// healthy, and never as dead either.
|
||||
HealthUntested = "untested"
|
||||
)
|
||||
|
||||
// HealthView is a consistent, cheap-to-query snapshot of node health: the engine's
|
||||
// measurement store plus a copy of our failure verdicts, frozen at one instant.
|
||||
// HealthView is a consistent, cheap-to-query snapshot of node health: the shared
|
||||
// health board frozen with one clock and one TTL.
|
||||
//
|
||||
// Take one view and query it for many tags. That is what makes the per-group
|
||||
// projection affordable (a 376-member group is 376 State calls against one view) and
|
||||
// what keeps every consumer — group health and the per-node stats — on the SAME
|
||||
// decoder, so the "no data means untested, never dead" rule exists in exactly one
|
||||
// place.
|
||||
// projection affordable (a 376-member group is 376 State calls against one view)
|
||||
// and what keeps every consumer — group health and the per-node stats — on the
|
||||
// SAME decoder, so the "no fresh data means untested, never dead" rule exists in
|
||||
// exactly one place.
|
||||
//
|
||||
// The zero value is usable and reports everything as HealthUntested.
|
||||
type HealthView struct {
|
||||
hist *urltest.HistoryStorage
|
||||
dead map[string]time.Time
|
||||
ttl time.Duration
|
||||
now time.Time
|
||||
}
|
||||
|
||||
// HealthView returns a snapshot of current node health. Safe on a stopped engine (it
|
||||
// reads the process-global history store, which outlives any Box) and safe for
|
||||
// HealthView returns a snapshot of current node health. Safe on a stopped engine
|
||||
// (it reads the process-global board, which outlives any Box) and safe for
|
||||
// concurrent use; the returned view is an immutable value the caller owns.
|
||||
func (e *Engine) HealthView() HealthView {
|
||||
v := HealthView{hist: e.URLTestHistory(), now: time.Now()}
|
||||
e.deadMu.Lock()
|
||||
if len(e.dead) > 0 {
|
||||
v.dead = make(map[string]time.Time, len(e.dead))
|
||||
for tag, at := range e.dead {
|
||||
v.dead[tag] = at
|
||||
}
|
||||
}
|
||||
e.deadMu.Unlock()
|
||||
return v
|
||||
return HealthView{hist: e.URLTestHistory(), ttl: e.healthTTL(), now: time.Now()}
|
||||
}
|
||||
|
||||
// State reports one outbound tag's health as of this view: the state, the last
|
||||
// SUCCESSFUL probe's RTT in ms (0 unless alive), and how many seconds ago the reported
|
||||
// observation was made (-1 when there is none).
|
||||
// SUCCESSFUL probe's RTT in ms (0 unless alive), and how many seconds ago the
|
||||
// reported observation was made (-1 when there is none).
|
||||
//
|
||||
// tag is the tag that was actually PROBED — a node's own tag for an ordinary member,
|
||||
// and the per-group egress copy tag for a member of an egress-bound group. Those are
|
||||
// two different network paths and therefore two different, independently valid states
|
||||
// for the same node.
|
||||
// tag is the tag that was actually PROBED — a node's own tag for an ordinary
|
||||
// member, and the per-group egress copy tag for a member of an egress-bound
|
||||
// group. Those are two different network paths and therefore two different,
|
||||
// independently valid states for the same node.
|
||||
func (v HealthView) State(tag string) (state string, delayMs int, ageSeconds int64) {
|
||||
// LoadURLTestHistory is nil-receiver safe, so a view taken with no history store
|
||||
// (bare Engine{}, engine never started) degrades to "untested", not a panic.
|
||||
return decodeHealth(v.hist.LoadURLTestHistory(tag), v.dead[tag], v.now)
|
||||
// LoadURLTestHistory is nil-receiver safe, so a view taken with no history
|
||||
// store (bare Engine{}, engine never started) degrades to "untested".
|
||||
return decodeHealth(v.hist.LoadURLTestHistory(tag), v.ttl, v.now)
|
||||
}
|
||||
|
||||
// decodeHealth is the SINGLE point where a stored measurement and our failure verdict
|
||||
// become one of the three states. Everything that reports node health — per-group
|
||||
// (grouphealth.go) and per-node (shater/stats) — goes through here, so the rules below
|
||||
// cannot drift between the two surfaces.
|
||||
// decodeHealth is the SINGLE point where a board record becomes one of the three
|
||||
// states. Everything that reports node health — per-group (grouphealth.go) and
|
||||
// per-node (shater/stats) — goes through here, so the rules cannot drift between
|
||||
// the two surfaces. It classifies exactly as urltest.HistoryStorage.VerdictAt
|
||||
// does (the strategies' view of the same record), adding only the delay and age
|
||||
// the panel renders.
|
||||
//
|
||||
// deadAt is the zero Time when we hold no failure verdict for the tag.
|
||||
//
|
||||
// success present, not older than our verdict -> alive (a later success always
|
||||
// wins, whoever recorded it:
|
||||
// our own run OR the group's)
|
||||
// verdict present and newer (or no success) -> dead (a probe ran and failed)
|
||||
// neither -> untested
|
||||
//
|
||||
// An entry with a ZERO timestamp cannot be ordered against a dated verdict, so a
|
||||
// verdict wins over it; that combination only arises from a hand-built history entry.
|
||||
func decodeHealth(h *adapter.URLTestHistory, deadAt, now time.Time) (state string, delayMs int, ageSeconds int64) {
|
||||
if h != nil && !deadAt.After(h.Time) {
|
||||
// An entry is only ever written after a probe SUCCEEDED (the engine stores on
|
||||
// success and deletes on failure), so its presence is proof of reachability
|
||||
// regardless of the magnitude of the delay.
|
||||
return HealthAlive, int(h.Delay), ageSince(h.Time, now)
|
||||
// A ttl <= 0 (zero HealthView in tests) falls back to the TTL floor.
|
||||
func decodeHealth(h *adapter.URLTestHistory, ttl time.Duration, now time.Time) (state string, delayMs int, ageSeconds int64) {
|
||||
if h == nil {
|
||||
return HealthUntested, 0, -1
|
||||
}
|
||||
if !deadAt.IsZero() {
|
||||
return HealthDead, 0, ageSince(deadAt, now)
|
||||
if ttl <= 0 {
|
||||
ttl = healthTTLFloor
|
||||
}
|
||||
switch {
|
||||
case h.LastOK.After(h.LastFail) && now.Sub(h.LastOK) < ttl:
|
||||
return HealthAlive, int(h.Delay), ageSince(h.LastOK, now)
|
||||
case h.LastFail.After(h.LastOK) && now.Sub(h.LastFail) < ttl:
|
||||
return HealthDead, 0, ageSince(h.LastFail, now)
|
||||
default:
|
||||
return HealthUntested, 0, -1
|
||||
}
|
||||
return HealthUntested, 0, -1
|
||||
}
|
||||
|
||||
// ageSince is whole seconds from t to now, clamped at 0 (a clock step must not produce
|
||||
// a negative age), or -1 when t is unset — the same "-1 means unknown" convention the
|
||||
// JSON carries.
|
||||
// ageSince is whole seconds from t to now, clamped at 0 (a clock step must not
|
||||
// produce a negative age), or -1 when t is unset — the same "-1 means unknown"
|
||||
// convention the JSON carries.
|
||||
func ageSince(t, now time.Time) int64 {
|
||||
if t.IsZero() {
|
||||
return -1
|
||||
@@ -166,55 +132,3 @@ func ageSince(t, now time.Time) int64 {
|
||||
}
|
||||
return age
|
||||
}
|
||||
|
||||
// MarkProbeFailed records that a probe of tag RAN and FAILED, at this instant.
|
||||
//
|
||||
// It is the only writer of a dead verdict, and it deliberately writes nowhere near the
|
||||
// engine's own history store (see the file header). It is the public counterpart of
|
||||
// URLTestHistory(), through which a caller records a SUCCESS: successes are the
|
||||
// engine's own truth and go in the engine's store, failures are ours and go here.
|
||||
//
|
||||
// There is deliberately no public "unmark": a verdict is retired by a later SUCCESS,
|
||||
// wherever that success is recorded (decodeHealth compares timestamps), so a caller
|
||||
// cannot leave a node stuck dead by forgetting to clear it.
|
||||
func (e *Engine) MarkProbeFailed(tag string) {
|
||||
e.deadMu.Lock()
|
||||
if e.dead == nil {
|
||||
e.dead = make(map[string]time.Time)
|
||||
}
|
||||
e.dead[tag] = time.Now()
|
||||
e.deadMu.Unlock()
|
||||
}
|
||||
|
||||
// clearProbeFailed drops any dead verdict for tag, called when a probe SUCCEEDS.
|
||||
//
|
||||
// Strictly speaking this is redundant — decodeHealth already ignores a verdict older
|
||||
// than the success we just stored — but dropping it keeps the map from accumulating
|
||||
// entries for tags that have long since recovered, and keeps the map's contents
|
||||
// meaning what its name says.
|
||||
func (e *Engine) clearProbeFailed(tag string) {
|
||||
e.deadMu.Lock()
|
||||
delete(e.dead, tag)
|
||||
e.deadMu.Unlock()
|
||||
}
|
||||
|
||||
// pruneProbeFailed drops verdicts for tags outside keep — the set a full probe run
|
||||
// just covered. That is what bounds the overlay: a node removed from the config, or a
|
||||
// per-group copy that vanished when a group lost its egress binding, stops being
|
||||
// probed and would otherwise keep its verdict forever.
|
||||
//
|
||||
// keep must be a COMPLETE probe target set; an empty/nil keep is ignored rather than
|
||||
// treated as "nothing is live", so a run that found no targets at all (stopped engine)
|
||||
// cannot wipe verdicts collected while it was up.
|
||||
func (e *Engine) pruneProbeFailed(keep map[string]bool) {
|
||||
if len(keep) == 0 {
|
||||
return
|
||||
}
|
||||
e.deadMu.Lock()
|
||||
for tag := range e.dead {
|
||||
if !keep[tag] {
|
||||
delete(e.dead, tag)
|
||||
}
|
||||
}
|
||||
e.deadMu.Unlock()
|
||||
}
|
||||
|
||||
@@ -31,64 +31,75 @@ func (f *failingOutbound) ListenPacket(context.Context, M.Socksaddr) (net.Packet
|
||||
|
||||
var _ adapter.Outbound = (*failingOutbound)(nil)
|
||||
|
||||
// TestDecodeHealth pins the single decoder: the three states, and above all the two
|
||||
// rules that make the feature honest — no data is UNTESTED (never dead), and a
|
||||
// success recorded AFTER our failure verdict wins (that is how a dead mark is
|
||||
// invalidated, including by the group's own probing, which knows nothing about us).
|
||||
// TestDecodeHealth pins the single decoder: the three states, and above all the
|
||||
// rules that make the feature honest — no fresh data is UNTESTED (never dead), a
|
||||
// success recorded AFTER a failure wins (that is how a dead mark is invalidated,
|
||||
// including by the group's own probing), and every observation ages out on the
|
||||
// TTL. The classification must agree with urltest.HistoryStorage.VerdictAt — the
|
||||
// strategies' view of the same record.
|
||||
func TestDecodeHealth(t *testing.T) {
|
||||
now := time.Now()
|
||||
at := func(d time.Duration) time.Time { return now.Add(d) }
|
||||
const ttl = 10 * time.Minute
|
||||
cases := []struct {
|
||||
name string
|
||||
h *adapter.URLTestHistory
|
||||
deadAt time.Time
|
||||
wantState string
|
||||
wantDelay int
|
||||
wantAge int64
|
||||
}{
|
||||
{
|
||||
name: "nothing known is untested, never dead",
|
||||
// The whole point: a group records a failed probe by DELETING the entry, so
|
||||
// absence cannot be read as a failure.
|
||||
// No record at all: the node is outside every used path, or was never
|
||||
// probed. Absence cannot be read as a failure.
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
{
|
||||
name: "measurement only is alive",
|
||||
h: &adapter.URLTestHistory{Time: at(-30 * time.Second), Delay: 142},
|
||||
name: "fresh success is alive",
|
||||
h: &adapter.URLTestHistory{LastOK: at(-30 * time.Second), Delay: 142},
|
||||
wantState: HealthAlive, wantDelay: 142, wantAge: 30,
|
||||
},
|
||||
{
|
||||
name: "verdict only is dead",
|
||||
deadAt: at(-5 * time.Second),
|
||||
name: "fresh failure is dead",
|
||||
h: &adapter.URLTestHistory{LastFail: at(-5 * time.Second)},
|
||||
wantState: HealthDead, wantAge: 5,
|
||||
},
|
||||
{
|
||||
name: "verdict NEWER than the measurement wins: dead",
|
||||
h: &adapter.URLTestHistory{Time: at(-time.Minute), Delay: 90},
|
||||
deadAt: at(-10 * time.Second),
|
||||
name: "failure NEWER than the success wins: dead (the last success's " +
|
||||
"delay stays stored for display but the state is the failure's)",
|
||||
h: &adapter.URLTestHistory{LastOK: at(-time.Minute), Delay: 90, LastFail: at(-10 * time.Second)},
|
||||
wantState: HealthDead, wantAge: 10,
|
||||
},
|
||||
{
|
||||
name: "measurement NEWER than the verdict wins: alive (this is how a dead " +
|
||||
name: "success NEWER than the failure wins: alive (this is how a dead " +
|
||||
"mark is invalidated, e.g. by the group's own successful probe)",
|
||||
h: &adapter.URLTestHistory{Time: at(-10 * time.Second), Delay: 77},
|
||||
deadAt: at(-time.Minute),
|
||||
h: &adapter.URLTestHistory{LastOK: at(-10 * time.Second), Delay: 77, LastFail: at(-time.Minute)},
|
||||
wantState: HealthAlive, wantDelay: 77, wantAge: 10,
|
||||
},
|
||||
{
|
||||
name: "a stored zero delay is still a SUCCESS, so alive",
|
||||
h: &adapter.URLTestHistory{Time: at(-time.Second), Delay: 0},
|
||||
h: &adapter.URLTestHistory{LastOK: at(-time.Second), Delay: 0},
|
||||
wantState: HealthAlive, wantDelay: 0, wantAge: 1,
|
||||
},
|
||||
{
|
||||
name: "measurement with no timestamp cannot outrank a dated verdict",
|
||||
name: "a success older than the TTL is untested — a measurement is only " +
|
||||
"as good as its age, and 'alive forever' is the defect the TTL removes",
|
||||
h: &adapter.URLTestHistory{LastOK: at(-ttl - time.Minute), Delay: 50},
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
{
|
||||
name: "a failure older than the TTL is untested, not dead forever",
|
||||
h: &adapter.URLTestHistory{LastFail: at(-ttl - time.Minute)},
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
{
|
||||
name: "a record with no usable timestamp is untested",
|
||||
h: &adapter.URLTestHistory{Delay: 5},
|
||||
deadAt: at(-time.Second),
|
||||
wantState: HealthDead, wantAge: 1,
|
||||
wantState: HealthUntested, wantAge: -1,
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
state, delay, age := decodeHealth(c.h, c.deadAt, now)
|
||||
state, delay, age := decodeHealth(c.h, ttl, now)
|
||||
if state != c.wantState || delay != c.wantDelay || age != c.wantAge {
|
||||
t.Errorf("%s:\n got (%q,%d,%d)\n want (%q,%d,%d)",
|
||||
c.name, state, delay, age, c.wantState, c.wantDelay, c.wantAge)
|
||||
@@ -96,22 +107,12 @@ func TestDecodeHealth(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestProbeFailureNeverEntersEngineHistory is the DoD tripwire for the round_robin /
|
||||
// failover pool hazard, and the reason the dead verdict is not a sentinel delay.
|
||||
//
|
||||
// The pool planner in protocol/group/urltest.go decides liveness by the PRESENCE of a
|
||||
// history entry, not by its value:
|
||||
//
|
||||
// rebuildPool (~:709) / seedPool (~:748):
|
||||
// if history := g.history.LoadURLTestHistory(RealTag(detour)); history != nil {
|
||||
// results[tag] = candidate{tag: tag, delay: history.Delay, alive: true}
|
||||
//
|
||||
// so ANY entry we invent for a dead node makes it a live candidate — and `failover`
|
||||
// (round_robin, one-slot pool, PoolTolerance 0) would fill its single slot with a node
|
||||
// we had just proved dead and hold it for a whole probe interval. The test therefore
|
||||
// asserts the invariant that makes that impossible: after a FAILED probe the engine's
|
||||
// history has no entry for the tag at all, while our own overlay reports it dead.
|
||||
func TestProbeFailureNeverEntersEngineHistory(t *testing.T) {
|
||||
// TestProbeFailureMarksBoardDead is the observatory→selection coupling point: a
|
||||
// failed probe MUST become a same-instant dead verdict on the shared board —
|
||||
// urltest.Verdict is what the balancer slots and Select() read (protocol/group),
|
||||
// so this is exactly what makes the node unselectable in the same tick. The entry
|
||||
// must NOT be deleted (a marked tag stays distinguishable from "never measured").
|
||||
func TestProbeFailureMarksBoardDead(t *testing.T) {
|
||||
e := New()
|
||||
hist := e.URLTestHistory()
|
||||
if hist == nil {
|
||||
@@ -120,40 +121,43 @@ func TestProbeFailureNeverEntersEngineHistory(t *testing.T) {
|
||||
const deadTag = "dead-node"
|
||||
const liveTag = "live-node"
|
||||
// A live member, recorded the way a real successful probe is.
|
||||
hist.StoreURLTestHistory(liveTag, &adapter.URLTestHistory{Time: time.Now(), Delay: 120})
|
||||
hist.StoreURLTestHistory(liveTag, &adapter.URLTestHistory{LastOK: time.Now(), Delay: 120})
|
||||
|
||||
e.probeOne(&failingOutbound{tag: deadTag}, "", hist)
|
||||
|
||||
if h := hist.LoadURLTestHistory(deadTag); h != nil {
|
||||
t.Fatalf("a FAILED probe wrote %+v into the engine's history; it must write "+
|
||||
"nothing there — any entry is read as 'alive' by the round_robin pool planner", h)
|
||||
// The board holds a real record (not a deletion) whose verdict is dead.
|
||||
if h := hist.LoadURLTestHistory(deadTag); h == nil || h.LastFail.IsZero() {
|
||||
t.Fatalf("a FAILED probe must MarkFailed on the board, got %+v", h)
|
||||
}
|
||||
ttl := e.healthTTL()
|
||||
if v := hist.Verdict(deadTag, ttl); v != urltest.VerdictDead {
|
||||
t.Fatalf("board verdict after failed probe = %v, want dead — this is what selection reads", v)
|
||||
}
|
||||
if v := hist.Verdict(liveTag, ttl); v != urltest.VerdictAlive {
|
||||
t.Fatalf("live node verdict = %v, want alive (unaffected)", v)
|
||||
}
|
||||
// And the panel's view agrees — one truth, two readers.
|
||||
if state, _, _ := e.HealthView().State(deadTag); state != HealthDead {
|
||||
t.Fatalf("failed probe: state = %q, want %q from our own overlay", state, HealthDead)
|
||||
}
|
||||
|
||||
// Now replay the planner's own rule over the engine's store: the dead node must not
|
||||
// be a candidate, the live one must be.
|
||||
for tag, wantCandidate := range map[string]bool{deadTag: false, liveTag: true} {
|
||||
got := hist.LoadURLTestHistory(tag) != nil // == the planner's `history != nil` test
|
||||
if got != wantCandidate {
|
||||
t.Errorf("pool candidacy of %q = %v, want %v", tag, got, wantCandidate)
|
||||
}
|
||||
t.Fatalf("failed probe: state = %q, want %q", state, HealthDead)
|
||||
}
|
||||
}
|
||||
|
||||
// A SUCCESSFUL probe stores a real measurement in the engine's history (the groups'
|
||||
// own selection legitimately benefits from it) and drops any stale dead verdict.
|
||||
func TestProbeSuccessClearsDeadVerdict(t *testing.T) {
|
||||
// A SUCCESSFUL probe stores a real measurement, which outranks the previous
|
||||
// failure on the same board record.
|
||||
func TestProbeSuccessOutranksDeadVerdict(t *testing.T) {
|
||||
e := New()
|
||||
hist := e.URLTestHistory()
|
||||
const tag = "flaky-node"
|
||||
e.MarkProbeFailed(tag)
|
||||
hist.MarkFailed(tag)
|
||||
if state, _, _ := e.HealthView().State(tag); state != HealthDead {
|
||||
t.Fatalf("after MarkProbeFailed: state = %q, want %q", state, HealthDead)
|
||||
t.Fatalf("after MarkFailed: state = %q, want %q", state, HealthDead)
|
||||
}
|
||||
// Simulate what probeOne does on success.
|
||||
e.URLTestHistory().StoreURLTestHistory(tag, &adapter.URLTestHistory{Time: time.Now(), Delay: 33})
|
||||
e.clearProbeFailed(tag)
|
||||
// Simulate what probeOneInto does on success — strictly LATER than the
|
||||
// failure: with the wall clock, LastOK could land on the SAME tick as
|
||||
// LastFail (Windows clock granularity) and read as untested (ties are
|
||||
// deliberately not alive — the ordering is strict).
|
||||
failedAt := hist.LoadURLTestHistory(tag).LastFail
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: failedAt.Add(time.Second), Delay: 33})
|
||||
|
||||
state, delay, _ := e.HealthView().State(tag)
|
||||
if state != HealthAlive || delay != 33 {
|
||||
@@ -161,14 +165,13 @@ func TestProbeSuccessClearsDeadVerdict(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// The dead overlay must survive an Apply swap exactly as the pre-registered history
|
||||
// store does — otherwise every apply would silently turn every known-dead node back
|
||||
// into "untested". It hangs off the Engine, not off a Box, so tearing the box down
|
||||
// (the destructive half of a swap) must not touch it.
|
||||
// A dead verdict must survive an Apply swap: the board is pre-registered on the
|
||||
// engine context, not owned by a Box, so tearing the box down (the destructive
|
||||
// half of a swap) must not touch it.
|
||||
func TestDeadVerdictSurvivesEngineTeardown(t *testing.T) {
|
||||
e := New()
|
||||
const tag = "node-a"
|
||||
e.MarkProbeFailed(tag)
|
||||
e.URLTestHistory().MarkFailed(tag)
|
||||
if err := e.Close(); err != nil {
|
||||
t.Fatalf("Close: %v", err)
|
||||
}
|
||||
@@ -180,32 +183,26 @@ func TestDeadVerdictSurvivesEngineTeardown(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// pruneProbeFailed bounds the overlay to what a full run actually probed, and must
|
||||
// refuse an empty keep-set so a run against a stopped engine cannot wipe verdicts
|
||||
// collected while it was up.
|
||||
func TestPruneProbeFailed(t *testing.T) {
|
||||
// healthTTL is max(3 x global probe interval, the 10-minute floor) — plan §5.A.
|
||||
// Every board consumer must classify with this same TTL, so the formula is pinned.
|
||||
func TestHealthTTLFormula(t *testing.T) {
|
||||
e := New()
|
||||
for _, tag := range []string{"keep-me", "gone-node", "group-vpn-m0-gone"} {
|
||||
e.MarkProbeFailed(tag)
|
||||
|
||||
// Unconfigured: the engine default interval (3m) gives 9m, floored to 10m.
|
||||
if got := e.healthTTL(); got != healthTTLFloor {
|
||||
t.Fatalf("default TTL = %v, want the %v floor", got, healthTTLFloor)
|
||||
}
|
||||
|
||||
e.pruneProbeFailed(nil)
|
||||
e.pruneProbeFailed(map[string]bool{})
|
||||
for _, tag := range []string{"keep-me", "gone-node"} {
|
||||
if state, _, _ := e.HealthView().State(tag); state != HealthDead {
|
||||
t.Fatalf("an empty keep-set pruned %q; it must be ignored", tag)
|
||||
}
|
||||
// A long interval dominates the floor: 3 x 5m = 15m.
|
||||
e.ConfigureObservatory(ObservatoryConfig{ProbeInterval: 5 * time.Minute})
|
||||
if got, want := e.healthTTL(), 15*time.Minute; got != want {
|
||||
t.Fatalf("TTL for 5m interval = %v, want %v", got, want)
|
||||
}
|
||||
|
||||
e.pruneProbeFailed(map[string]bool{"keep-me": true, "never-probed": true})
|
||||
if state, _, _ := e.HealthView().State("keep-me"); state != HealthDead {
|
||||
t.Errorf("keep-me was pruned but is still a probe target")
|
||||
}
|
||||
for _, tag := range []string{"gone-node", "group-vpn-m0-gone"} {
|
||||
if state, _, _ := e.HealthView().State(tag); state != HealthUntested {
|
||||
t.Errorf("%s: state = %q after prune, want %q (it is no longer a probe target)",
|
||||
tag, state, HealthUntested)
|
||||
}
|
||||
// A short interval is floored: 3 x 30s is far below 10m.
|
||||
e.ConfigureObservatory(ObservatoryConfig{ProbeInterval: 30 * time.Second})
|
||||
if got := e.healthTTL(); got != healthTTLFloor {
|
||||
t.Fatalf("TTL for 30s interval = %v, want the %v floor", got, healthTTLFloor)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -224,6 +221,6 @@ func TestHealthViewZeroValue(t *testing.T) {
|
||||
|
||||
// newHealthView is the test constructor for a view over a hand-built store, used by
|
||||
// the group-health tests.
|
||||
func newHealthView(hist *urltest.HistoryStorage, dead map[string]time.Time, now time.Time) HealthView {
|
||||
return HealthView{hist: hist, dead: dead, now: now}
|
||||
func newHealthView(hist *urltest.HistoryStorage, ttl time.Duration, now time.Time) HealthView {
|
||||
return HealthView{hist: hist, ttl: ttl, now: now}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,415 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"context"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// The observatory — the background prober that keeps the health board fresh for
|
||||
// everything the routing rules can reach (plan §5.C; replaces the full-population
|
||||
// sweep and the manual "Test all nodes" run).
|
||||
//
|
||||
// # What it does, and what it deliberately does not
|
||||
//
|
||||
// One central walker probes the reachability plan (BuildObservatoryPlan) with the
|
||||
// global probe URL and writes every outcome into the shared health board
|
||||
// (common/urltest): a success through StoreURLTestHistory, a failure through
|
||||
// MarkFailed. The strategies read the same board, so a death found here makes the
|
||||
// node unselectable immediately — no private overlay, no panel-only truth.
|
||||
//
|
||||
// It never replaces a group's own probing. An ACTIVE urltest group is its own
|
||||
// fast failure detector (failover switches on its 30s probe tick), and the
|
||||
// freshness gate below skips any tag whose newest observation is younger than the
|
||||
// global probe interval — which is precisely the set of tags an active group is
|
||||
// already keeping current. The observatory's budget lands on what nobody else
|
||||
// probes: idle groups' members, selector candidates, chain exits.
|
||||
//
|
||||
// # Schedule and cost
|
||||
//
|
||||
// The plan is walked by a cursor across small ticks — observatoryTick apart,
|
||||
// observatoryBatch measurements started per tick, observatoryConcurrency in
|
||||
// flight, probeTimeout per probe. Worst case per tick is 24/12 x 5s = 10s, one
|
||||
// period, so a fully dead population saturates the schedule but never stacks (a
|
||||
// busy tick makes the next one a no-op). The population is only what the rules
|
||||
// reach, so a full pass is minutes even on a subscription-sized config, and the
|
||||
// first pass starts IMMEDIATELY after an apply (cold boot: used paths validated
|
||||
// in seconds).
|
||||
//
|
||||
// # Reconfiguration must not restart the walk
|
||||
//
|
||||
// ConfigureObservatory runs on every successful apply, and cron reconciles every
|
||||
// minute — almost always with an identical config. Rebuilding the plan and
|
||||
// zeroing the cursor there would keep the walk from ever completing a pass (the
|
||||
// release-blocking sweep bug). So the cursor is kept whenever the rebuilt plan is
|
||||
// measurement-for-measurement identical to the old one (plansEqual); only a
|
||||
// genuinely different plan restarts it, which is correct — the old cursor indexes
|
||||
// a list that no longer exists.
|
||||
|
||||
const (
|
||||
// observatoryTick is the schedule period. Internal constant, no knob: the
|
||||
// per-tag refresh rate is governed by the global probe interval, not by this.
|
||||
observatoryTick = 10 * time.Second
|
||||
// observatoryBatch is the number of measurements started per tick.
|
||||
observatoryBatch = 24
|
||||
// observatoryConcurrency is the number of measurements in flight at once.
|
||||
observatoryConcurrency = 12
|
||||
// probeTimeout bounds a single probe (connect + HTTP HEAD).
|
||||
probeTimeout = 5 * time.Second
|
||||
// healthTTLFloor is the minimum verdict TTL. The TTL is
|
||||
// max(3 x global probe interval, this floor) — plan §5.A: three missed
|
||||
// refresh periods demote a stale success to "untested" rather than letting
|
||||
// an old measurement read as alive forever.
|
||||
healthTTLFloor = 10 * time.Minute
|
||||
)
|
||||
|
||||
// ObservatoryConfig is what ConfigureObservatory needs from an apply. The zero
|
||||
// value is DISABLED.
|
||||
type ObservatoryConfig struct {
|
||||
Enabled bool
|
||||
// Options is the JUST-APPLIED generated config the reachability plan is
|
||||
// derived from (the generator owns the tag schema; opts is where it is
|
||||
// materialised).
|
||||
Options option.Options
|
||||
// ProbeURL is the global probe URL; "" lets urltest use its gstatic default.
|
||||
ProbeURL string
|
||||
// ProbeInterval is the global probe interval — the freshness gate and the
|
||||
// verdict-TTL base. <=0 falls back to the engine default (3 minutes).
|
||||
ProbeInterval time.Duration
|
||||
}
|
||||
|
||||
// observatoryState is the observatory's mutable state, guarded by obsMu.
|
||||
type observatoryState struct {
|
||||
enabled bool
|
||||
interval time.Duration // effective global probe interval
|
||||
// stop/done/nudge belong to the running loop; nil when no loop is running.
|
||||
// nudge wakes the loop early after a plan change (immediate first pass).
|
||||
stop chan struct{}
|
||||
done chan struct{}
|
||||
nudge chan struct{}
|
||||
// plan is the cycle being walked, cursor the position in it. The plan is
|
||||
// static between applies — it derives from the applied options only.
|
||||
plan []ProbeJob
|
||||
used map[string]bool
|
||||
cursor int
|
||||
cycles uint64
|
||||
// busy guards against a slow tick overlapping the next one.
|
||||
busy bool
|
||||
}
|
||||
|
||||
// ConfigureObservatory installs (or re-tunes, or stops) the observatory. It is
|
||||
// idempotent and called from applyLocked on every SUCCESSFUL apply — the only
|
||||
// moment the facts it depends on (the applied options, the global probe
|
||||
// settings, the GroupHealth master switch) can change. See the file header for
|
||||
// why an identical plan keeps the cursor.
|
||||
func (e *Engine) ConfigureObservatory(cfg ObservatoryConfig) {
|
||||
interval := cfg.ProbeInterval
|
||||
if interval <= 0 {
|
||||
interval = C.DefaultURLTestInterval
|
||||
}
|
||||
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
e.obs.enabled = cfg.Enabled
|
||||
e.obs.interval = interval
|
||||
|
||||
if !cfg.Enabled {
|
||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
||||
if e.obs.stop != nil {
|
||||
close(e.obs.stop)
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
||||
}
|
||||
return
|
||||
}
|
||||
|
||||
jobs, used := BuildObservatoryPlan(cfg.Options, cfg.ProbeURL)
|
||||
e.obs.used = used
|
||||
fresh := e.obs.plan == nil || !plansEqual(jobs, e.obs.plan)
|
||||
if fresh {
|
||||
e.obs.plan = jobs
|
||||
e.obs.cursor = 0
|
||||
}
|
||||
|
||||
if e.obs.stop == nil {
|
||||
stop, done := make(chan struct{}), make(chan struct{})
|
||||
nudge := make(chan struct{}, 1)
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = stop, done, nudge
|
||||
go e.observatoryLoop(stop, done, nudge)
|
||||
} else if fresh {
|
||||
// The plan really changed: start the new walk now rather than on the
|
||||
// next tick (the immediate-first-pass contract after an apply).
|
||||
select {
|
||||
case e.obs.nudge <- struct{}{}:
|
||||
default:
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// StopObservatory stops the observatory and waits for an in-flight tick to
|
||||
// finish. Idempotent; safe when nothing runs (teardown path).
|
||||
//
|
||||
// The wait is deliberate and bounded by one batch (worst case
|
||||
// Batch/Concurrency x probeTimeout = 10s): teardown closes the box immediately
|
||||
// afterwards, and probes still in flight would then fail their dials and be
|
||||
// recorded as failure verdicts about nodes that were never actually unreachable.
|
||||
func (e *Engine) StopObservatory() {
|
||||
e.obsMu.Lock()
|
||||
stop, done := e.obs.stop, e.obs.done
|
||||
e.obs.stop, e.obs.done, e.obs.nudge = nil, nil, nil
|
||||
e.obs.enabled = false
|
||||
e.obs.plan, e.obs.used, e.obs.cursor = nil, nil, 0
|
||||
if stop != nil {
|
||||
close(stop)
|
||||
}
|
||||
e.obsMu.Unlock()
|
||||
if done != nil {
|
||||
<-done
|
||||
}
|
||||
}
|
||||
|
||||
// ObservatoryStatus reports whether the observatory is on, how many measurements
|
||||
// the current plan holds, and how many full passes have completed. Cheap; used by
|
||||
// tests and available for diagnostics.
|
||||
func (e *Engine) ObservatoryStatus() (enabled bool, planned int, cycles uint64) {
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
return e.obs.enabled, len(e.obs.plan), e.obs.cycles
|
||||
}
|
||||
|
||||
// observatoryUsed returns the published used-set: every tag reachable from the
|
||||
// applied routing rules, as computed for the current plan. nil until an enabled
|
||||
// ConfigureObservatory ran (callers treat nil as "everything used" — no badge is
|
||||
// better than a wrong one). The map is replaced wholesale on reconfiguration and
|
||||
// never mutated in place, so the returned reference is safe to read.
|
||||
func (e *Engine) observatoryUsed() map[string]bool {
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
return e.obs.used
|
||||
}
|
||||
|
||||
// healthTTL is how long an observation stays authoritative:
|
||||
// max(3 x global probe interval, healthTTLFloor). It MUST be computed with the
|
||||
// same formula every consumer of the board uses (plan §5.A), because a verdict
|
||||
// is a (record, TTL) pair — see HealthView and common/urltest Verdict.
|
||||
func (e *Engine) healthTTL() time.Duration {
|
||||
e.obsMu.Lock()
|
||||
interval := e.obs.interval
|
||||
e.obsMu.Unlock()
|
||||
if interval <= 0 {
|
||||
interval = C.DefaultURLTestInterval
|
||||
}
|
||||
ttl := 3 * interval
|
||||
if ttl < healthTTLFloor {
|
||||
ttl = healthTTLFloor
|
||||
}
|
||||
return ttl
|
||||
}
|
||||
|
||||
// observatoryLoop is the ticker. It owns no state of its own — everything lives
|
||||
// under obsMu — so a re-tune from ConfigureObservatory takes effect on the next
|
||||
// tick. The first tick runs immediately (cold start: an applied config's used
|
||||
// paths get their first verdicts in seconds, not after a full tick period).
|
||||
//
|
||||
// Contract with the fork's native group checker (protocol/group CheckOutbounds):
|
||||
// this loop is the BACKGROUND half of the pair — each tick the freshness gate
|
||||
// (observatoryShouldProbe) skips tags an active group is already measuring
|
||||
// itself, so nothing is probed twice and the group's standby cadence (failover's
|
||||
// 30s interval) is never suppressed. Both halves write to the same health board.
|
||||
func (e *Engine) observatoryLoop(stop <-chan struct{}, done chan<- struct{}, nudge <-chan struct{}) {
|
||||
defer close(done)
|
||||
e.observatoryTickOnce()
|
||||
ticker := time.NewTicker(observatoryTick)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-stop:
|
||||
return
|
||||
case <-ticker.C:
|
||||
e.observatoryTickOnce()
|
||||
case <-nudge:
|
||||
e.observatoryTickOnce()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// observatoryTickOnce runs one batch. It is deliberately conservative about when
|
||||
// it does nothing:
|
||||
//
|
||||
// - a manual exit test (grouptest.go) is in flight => SKIP. That run is the
|
||||
// human's explicit request; the observatory defers rather than competing for
|
||||
// the uplink. It checks the guard but never CLAIMS it, so a pressed Test
|
||||
// button can never be answered "already running" because of background work;
|
||||
// - the previous tick has not finished => SKIP, so slow probes never stack;
|
||||
// - the engine is stopped => jobs no longer resolve and are skipped (probeJob).
|
||||
func (e *Engine) observatoryTickOnce() {
|
||||
if e.groupTestRunning.Load() {
|
||||
return
|
||||
}
|
||||
|
||||
e.obsMu.Lock()
|
||||
if e.obs.busy || !e.obs.enabled || len(e.obs.plan) == 0 {
|
||||
e.obsMu.Unlock()
|
||||
return
|
||||
}
|
||||
if e.obs.cursor >= len(e.obs.plan) {
|
||||
// Cycle complete: count it and start the next pass from the top. This is
|
||||
// the ONE place the cursor is deliberately rewound.
|
||||
e.obs.cycles++
|
||||
e.obs.cursor = 0
|
||||
}
|
||||
start := e.obs.cursor
|
||||
end := start + observatoryBatch
|
||||
if end > len(e.obs.plan) {
|
||||
end = len(e.obs.plan)
|
||||
}
|
||||
batch := append([]ProbeJob(nil), e.obs.plan[start:end]...)
|
||||
e.obs.cursor = end
|
||||
interval := e.obs.interval
|
||||
e.obs.busy = true
|
||||
e.obsMu.Unlock()
|
||||
|
||||
defer func() {
|
||||
e.obsMu.Lock()
|
||||
e.obs.busy = false
|
||||
e.obsMu.Unlock()
|
||||
}()
|
||||
|
||||
hist := e.URLTestHistory()
|
||||
if hist == nil {
|
||||
return
|
||||
}
|
||||
now := time.Now()
|
||||
sem := make(chan struct{}, observatoryConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for _, j := range batch {
|
||||
if !observatoryShouldProbe(hist, j, interval, now) {
|
||||
continue
|
||||
}
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(j ProbeJob) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
e.probeJob(j)
|
||||
}(j)
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// observatoryShouldProbe is the freshness gate: a job runs when ANY tag it
|
||||
// covers has no observation at all, or its newest observation — success OR
|
||||
// failure — is older than the global probe interval.
|
||||
//
|
||||
// This is what keeps the observatory off the tags an ACTIVE group measures
|
||||
// itself: the group probes at the same global interval (failover at 30s), so its
|
||||
// tags are always fresher than the gate and are skipped — no duplicate probes,
|
||||
// and the group's own detection cadence is never suppressed.
|
||||
//
|
||||
// "any", not "all", on purpose — the job writes to every tag it covers, so a
|
||||
// single stale tag is reason enough.
|
||||
func observatoryShouldProbe(hist *urltest.HistoryStorage, j ProbeJob, interval time.Duration, now time.Time) bool {
|
||||
for _, tag := range j.Store {
|
||||
h := hist.LoadURLTestHistory(tag)
|
||||
if h == nil {
|
||||
return true
|
||||
}
|
||||
newest := h.LastOK
|
||||
if h.LastFail.After(newest) {
|
||||
newest = h.LastFail
|
||||
}
|
||||
if now.Sub(newest) >= interval {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// probeJob resolves a job's dial tag against the CURRENT box and probes it,
|
||||
// recording the outcome under every tag the job covers.
|
||||
//
|
||||
// Resolving at dial time (rather than holding the adapter.Outbound captured when
|
||||
// the plan was built) is what makes a plan safe across an Apply swap: an
|
||||
// outbound the new config dropped simply no longer resolves, and the job is
|
||||
// SKIPPED. Recording it dead would be a fabricated verdict about a node that was
|
||||
// never dialled — and one that outlives the config change.
|
||||
//
|
||||
// Outbound(tag) falls through to the endpoint manager, so wireguard/AmneziaWG
|
||||
// nodes and their copies resolve here exactly like plain outbounds.
|
||||
func (e *Engine) probeJob(j ProbeJob) {
|
||||
inst := e.Instance()
|
||||
if inst == nil {
|
||||
return
|
||||
}
|
||||
om := inst.Outbound()
|
||||
if om == nil {
|
||||
return
|
||||
}
|
||||
ob, ok := om.Outbound(j.Dial)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
hist := e.URLTestHistory()
|
||||
if hist == nil {
|
||||
return
|
||||
}
|
||||
e.probeOneInto(ob, j.URL, j.Store, hist)
|
||||
}
|
||||
|
||||
// probeOneInto probes one outbound and records the result in the shared health
|
||||
// board, for every tag the measurement covers:
|
||||
//
|
||||
// success — StoreURLTestHistory: a real measurement, exactly as the groups'
|
||||
// own probing records one; selection legitimately benefits from it.
|
||||
// failure — MarkFailed: LastFail is stamped while any previous success is
|
||||
// preserved for display. The strategies read the same board, so the
|
||||
// node becomes unselectable immediately (plan §5.A/§5.B).
|
||||
//
|
||||
// store may name MORE tags than the one dialled. That is not extrapolation: the
|
||||
// planner only groups tags whose dial path is identical, plus the base-tag alias
|
||||
// of an egress copy (see probeplan.go), so the single measurement is literally
|
||||
// the answer for each of them.
|
||||
//
|
||||
// Every alive<->dead verdict FLIP is logged (info) — the single diagnostic trail
|
||||
// for localising a dead chain hop and for spotting a flapping node (plan §5.C).
|
||||
//
|
||||
// The delay is clamped to >=1ms because 0 is the engine's "unset" value in its
|
||||
// own selection arithmetic (protocol/group/urltest.go Select treats 0 as
|
||||
// no-value), so a genuinely sub-millisecond node must not report it.
|
||||
func (e *Engine) probeOneInto(ob adapter.Outbound, probeURL string, store []string, hist *urltest.HistoryStorage) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), probeTimeout)
|
||||
defer cancel()
|
||||
|
||||
delay, err := urltest.URLTest(ctx, probeURL, ob)
|
||||
ttl := e.healthTTL()
|
||||
if err != nil {
|
||||
for _, tag := range store {
|
||||
if e.log != nil && hist.Verdict(tag, ttl) == urltest.VerdictAlive {
|
||||
e.log.Info("observatory: ", tag, ": alive -> dead (probe: ", err, ")")
|
||||
}
|
||||
hist.MarkFailed(tag)
|
||||
}
|
||||
return
|
||||
}
|
||||
if delay == 0 {
|
||||
delay = 1
|
||||
}
|
||||
now := time.Now()
|
||||
for _, tag := range store {
|
||||
if e.log != nil && hist.Verdict(tag, ttl) == urltest.VerdictDead {
|
||||
e.log.Info("observatory: ", tag, ": dead -> alive (probe)")
|
||||
}
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: now, Delay: delay})
|
||||
}
|
||||
}
|
||||
|
||||
// probeOne probes a single outbound and records it under its own tag only. It is
|
||||
// probeOneInto's one-tag form, kept for callers (and tests) that have an
|
||||
// outbound rather than a plan.
|
||||
func (e *Engine) probeOne(ob adapter.Outbound, probeURL string, hist *urltest.HistoryStorage) {
|
||||
e.probeOneInto(ob, probeURL, []string{ob.Tag()}, hist)
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// obsFixture is a small applied config: one used urltest group over two nodes,
|
||||
// one unreferenced selector.
|
||||
func obsFixture() option.Options {
|
||||
return option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixNode("n1", ""),
|
||||
fixNode("n2", ""),
|
||||
fixGroup(C.TypeURLTest, "auto", "n1", "n2"),
|
||||
fixNode("idle1", ""),
|
||||
fixGroup(C.TypeSelector, "idle", "idle1"),
|
||||
},
|
||||
Route: fixRoute("auto"),
|
||||
}
|
||||
}
|
||||
|
||||
func obsState(e *Engine) (plan []ProbeJob, cursor int, enabled bool) {
|
||||
e.obsMu.Lock()
|
||||
defer e.obsMu.Unlock()
|
||||
return e.obs.plan, e.obs.cursor, e.obs.enabled
|
||||
}
|
||||
|
||||
// ConfigureObservatory is called on every apply: it must be idempotent, publish
|
||||
// the used-set, keep the cursor across a reconfiguration with an identical plan
|
||||
// (the release-blocking sweep bug), reset it on a genuinely different plan, and
|
||||
// stop cleanly on Enabled=false.
|
||||
func TestConfigureObservatoryLifecycle(t *testing.T) {
|
||||
e := New()
|
||||
defer e.StopObservatory()
|
||||
|
||||
if _, _, enabled := obsState(e); enabled {
|
||||
t.Fatal("a fresh engine must not be observing")
|
||||
}
|
||||
if e.observatoryUsed() != nil {
|
||||
t.Fatal("used-set must be nil before the first enabled configure")
|
||||
}
|
||||
|
||||
cfg := ObservatoryConfig{Enabled: true, Options: obsFixture(), ProbeURL: "u", ProbeInterval: time.Minute}
|
||||
e.ConfigureObservatory(cfg)
|
||||
plan, _, enabled := obsState(e)
|
||||
if !enabled || len(plan) != 2 {
|
||||
t.Fatalf("after enable: enabled=%v plan=%d jobs, want true/2 (n1, n2)", enabled, len(plan))
|
||||
}
|
||||
used := e.observatoryUsed()
|
||||
if used == nil || !used["auto"] || !used["n1"] || used["idle"] {
|
||||
t.Fatalf("used-set = %v, want auto/n1 in, idle out", used)
|
||||
}
|
||||
|
||||
// Simulate a walk in progress, then a no-op reconcile: the cursor must survive.
|
||||
e.obsMu.Lock()
|
||||
e.obs.cursor = 1
|
||||
e.obsMu.Unlock()
|
||||
e.ConfigureObservatory(cfg)
|
||||
if _, cursor, _ := obsState(e); cursor != 1 {
|
||||
t.Fatalf("identical-plan reconfigure reset the cursor to %d, want 1 kept", cursor)
|
||||
}
|
||||
|
||||
// A genuinely different plan restarts the walk.
|
||||
changed := cfg
|
||||
changed.ProbeURL = "other"
|
||||
e.ConfigureObservatory(changed)
|
||||
if _, cursor, _ := obsState(e); cursor != 0 {
|
||||
t.Fatalf("changed-plan reconfigure kept the cursor at %d, want 0", cursor)
|
||||
}
|
||||
|
||||
// The master switch stops it and clears the plan and the used-set.
|
||||
e.ConfigureObservatory(ObservatoryConfig{Enabled: false})
|
||||
if plan, _, enabled := obsState(e); enabled || plan != nil {
|
||||
t.Fatalf("after disable: enabled=%v plan=%v, want stopped and empty", enabled, plan)
|
||||
}
|
||||
if e.observatoryUsed() != nil {
|
||||
t.Fatal("used-set must clear on disable")
|
||||
}
|
||||
e.StopObservatory()
|
||||
e.StopObservatory() // idempotent
|
||||
}
|
||||
|
||||
// A tick against a stopped engine walks the plan without dialling anything (jobs
|
||||
// no longer resolve) and without panicking; the cursor still advances, so a dead
|
||||
// box cannot wedge the schedule.
|
||||
func TestObservatoryTickStoppedEngine(t *testing.T) {
|
||||
e := New()
|
||||
defer e.StopObservatory()
|
||||
e.ConfigureObservatory(ObservatoryConfig{Enabled: true, Options: obsFixture(), ProbeInterval: time.Minute})
|
||||
|
||||
e.observatoryTickOnce()
|
||||
if _, cursor, _ := obsState(e); cursor == 0 {
|
||||
t.Fatal("tick did not advance the cursor")
|
||||
}
|
||||
}
|
||||
|
||||
// The exit-test gate: while a manual group/chain test is in flight the tick is a
|
||||
// no-op — the observatory checks the guard but never claims it, so a pressed
|
||||
// Test button is never refused because of background work.
|
||||
func TestObservatoryDefersToExitTest(t *testing.T) {
|
||||
e := New()
|
||||
defer e.StopObservatory()
|
||||
e.ConfigureObservatory(ObservatoryConfig{Enabled: true, Options: obsFixture(), ProbeInterval: time.Minute})
|
||||
|
||||
e.groupTestRunning.Store(true)
|
||||
e.observatoryTickOnce()
|
||||
if _, cursor, _ := obsState(e); cursor != 0 {
|
||||
t.Fatal("tick ran while an exit test was in flight")
|
||||
}
|
||||
e.groupTestRunning.Store(false)
|
||||
|
||||
// And the guard is free: a manual run can start immediately.
|
||||
if !e.TestGroups(nil, "") {
|
||||
t.Fatal("a manual exit test was refused; the observatory must never hold groupTestRunning")
|
||||
}
|
||||
waitGroupTestIdle(t, e)
|
||||
}
|
||||
|
||||
// The freshness gate: a job whose every covered tag has an observation younger
|
||||
// than the global probe interval is skipped — that set is exactly what an ACTIVE
|
||||
// group is measuring itself, and the observatory must not duplicate or suppress
|
||||
// it. Anything stale or unknown runs.
|
||||
func TestObservatoryFreshnessGate(t *testing.T) {
|
||||
hist := urltest.NewHistoryStorage()
|
||||
defer hist.Close()
|
||||
now := time.Now()
|
||||
interval := time.Minute
|
||||
|
||||
hist.StoreURLTestHistory("fresh", &adapter.URLTestHistory{LastOK: now.Add(-10 * time.Second), Delay: 5})
|
||||
hist.StoreURLTestHistory("stale", &adapter.URLTestHistory{LastOK: now.Add(-2 * interval), Delay: 5})
|
||||
hist.StoreURLTestHistory("fresh-dead", &adapter.URLTestHistory{LastFail: now.Add(-10 * time.Second)})
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
job ProbeJob
|
||||
want bool
|
||||
}{
|
||||
{"never measured runs", ProbeJob{Store: []string{"unknown"}}, true},
|
||||
{"fresh success skips", ProbeJob{Store: []string{"fresh"}}, false},
|
||||
{"stale success runs", ProbeJob{Store: []string{"stale"}}, true},
|
||||
{"fresh FAILURE skips too — a dead verdict is information, not a retry invitation",
|
||||
ProbeJob{Store: []string{"fresh-dead"}}, false},
|
||||
{"one stale tag among fresh is reason enough", ProbeJob{Store: []string{"fresh", "stale"}}, true},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := observatoryShouldProbe(hist, c.job, interval, now); got != c.want {
|
||||
t.Errorf("%s: got %v, want %v", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The §6-S4 acceptance point: a death the observatory finds is visible on the
|
||||
// BOARD in the same tick — urltest.Verdict flips to dead immediately, which is
|
||||
// exactly what the balancer slots and Select() read (protocol/group). No overlay,
|
||||
// no panel-only truth, no waiting for a next cycle.
|
||||
func TestObservatoryDeathVisibleToSelectionSameTick(t *testing.T) {
|
||||
e := New()
|
||||
hist := e.URLTestHistory()
|
||||
const tag = "doomed"
|
||||
// The node was alive a moment ago…
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: time.Now().Add(-30 * time.Second), Delay: 40})
|
||||
if v := hist.Verdict(tag, e.healthTTL()); v != urltest.VerdictAlive {
|
||||
t.Fatalf("precondition: verdict = %v, want alive", v)
|
||||
}
|
||||
|
||||
// …and this probe finds it dead.
|
||||
e.probeOne(&failingOutbound{tag: tag}, "", hist)
|
||||
|
||||
if v := hist.Verdict(tag, e.healthTTL()); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict after observed failure = %v, want dead — selection reads this same board", v)
|
||||
}
|
||||
// The panel agrees, and the last-known delay is preserved on the record for display.
|
||||
if state, _, _ := e.HealthView().State(tag); state != HealthDead {
|
||||
t.Fatalf("panel state = %q, want %q", state, HealthDead)
|
||||
}
|
||||
if h := hist.LoadURLTestHistory(tag); h == nil || h.Delay != 40 {
|
||||
t.Fatalf("record = %+v, want the last success's delay preserved", h)
|
||||
}
|
||||
}
|
||||
@@ -1,249 +0,0 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"context"
|
||||
"strings"
|
||||
"sync"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// Manual "Test all nodes" — a one-shot, on-demand health probe of every dialable node
|
||||
// in the running box: node outbounds, node ENDPOINTS (wireguard/AmneziaWG), and the
|
||||
// per-group egress COPIES an egress-bound group actually dials (feedback #3).
|
||||
//
|
||||
// # Why this exists
|
||||
//
|
||||
// Per-node health normally comes only from urltest-strategy groups, which probe
|
||||
// LAZILY (only enough members to keep their pool live) and go to sleep when idle.
|
||||
// On a big subscription that leaves the overwhelming majority of nodes "untested".
|
||||
// TestAllNodes forces a real probe of every node and writes each result into the
|
||||
// SAME shared urltest.HistoryStorage the groups use (engine.URLTestHistory()), so
|
||||
// the stats aggregator — and thus the panel — surfaces real alive+latency for all
|
||||
// nodes with no other plumbing (collectNodeHealth already reads that store).
|
||||
//
|
||||
// # Failure encoding
|
||||
//
|
||||
// A group DELETES history on a failed probe, which makes a dead node indistinguishable
|
||||
// from a never-probed one (both = "untested"). For the manual "test all" we want a
|
||||
// failed node to read as DOWN, so this run records the failure — but NOT in the
|
||||
// engine's history store. A success is stored there (it is a real measurement, and the
|
||||
// groups' own routing should benefit from it); a failure goes into our own overlay via
|
||||
// MarkProbeFailed. health.go carries the full rationale; the short version is that the
|
||||
// round_robin pool planner treats the mere PRESENCE of a history entry as "this member
|
||||
// is alive", so any sentinel we invented would walk a known-dead node into a failover
|
||||
// group's single pool slot.
|
||||
//
|
||||
// This is also why the run covers per-group egress copies (see shouldProbeOutbound):
|
||||
// for a group bound to an egress, this run is the ONLY thing that can ever mark one of
|
||||
// its members dead — the group's own probing records failure by deletion, so without
|
||||
// it grouphealth.go could report such a member only as "untested".
|
||||
const (
|
||||
// probeAllConcurrency bounds how many nodes are probed in parallel.
|
||||
probeAllConcurrency = 16
|
||||
// probeAllTimeout bounds a single node's probe (connect + HTTP HEAD).
|
||||
probeAllTimeout = 5 * time.Second
|
||||
)
|
||||
|
||||
// shouldProbeOutbound reports whether an outbound with the given type and tag is
|
||||
// worth health-probing in TestAllNodes — as opposed to something that is NOT a
|
||||
// dialable node and would only pollute the run:
|
||||
//
|
||||
// - a GROUP (selector / urltest): a virtual outbound over other nodes;
|
||||
// - a UTILITY (direct / block / dns): not a proxy hop at all;
|
||||
// - an interface EGRESS (tag "egress-<name>", netplane.EgressOutboundTag);
|
||||
// - a per-chain HOP COPY (tag "chain-<name>-h<i>" or "…-<member>",
|
||||
// generate/chain.go) — the same underlying node is already probed under its own
|
||||
// base tag, and a hop copy's health is keyed by the copy tag, not a node name,
|
||||
// so the panel would never display it anyway.
|
||||
//
|
||||
// A per-group EGRESS COPY (tag "group-<name>-m<i>-<member>", generate/group.go) IS
|
||||
// probed, and that is a deliberate reversal of the original rule. The copy is the
|
||||
// outbound an egress-bound group ACTUALLY dials; its base twin is a different network
|
||||
// path (plain WAN vs the group's tunnel) and says nothing about it. Skipping copies
|
||||
// meant that for such a group "Test all nodes" measured objects the group never uses,
|
||||
// and — because a urltest group DELETES history on a failed probe — a dead member of a
|
||||
// bound group could never be reported dead at all, only "untested". Probing the copies
|
||||
// is the only thing that can record a failure verdict for them.
|
||||
//
|
||||
// The cost is bounded and pays only where it buys something: copies exist ONLY for
|
||||
// groups with Group.Egress set (generate/group.go builds none otherwise, precisely
|
||||
// because they are not free), so an installation with no egress binding probes exactly
|
||||
// what it did before. A 331-member group bound to an egress adds 331 probes, all
|
||||
// through that egress — that roughly doubles a one-shot, singleton-guarded, manually
|
||||
// triggered run with a progress counter, which is the right price for the run telling
|
||||
// the truth about the group.
|
||||
//
|
||||
// Everything else is a node outbound whose tag == the node name (generate/outbound.go
|
||||
// emits Tag: node.Name), which is exactly the key the stats aggregator joins health
|
||||
// onto. A user node literally named "egress-…"/"chain-…" would be skipped — an
|
||||
// accepted, documented edge case (those prefixes are reserved for generated tags).
|
||||
//
|
||||
// Pure (no engine state) so it is unit-tested directly.
|
||||
func shouldProbeOutbound(typ, tag string) bool {
|
||||
switch typ {
|
||||
case C.TypeSelector, C.TypeURLTest, C.TypeDirect, C.TypeBlock, C.TypeDNS:
|
||||
return false
|
||||
}
|
||||
if strings.HasPrefix(tag, "egress-") || strings.HasPrefix(tag, "chain-") {
|
||||
return false
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// TestAllNodes launches a one-shot background probe of EVERY probeable outbound in the
|
||||
// running box — node outbounds, node endpoints (wireguard/AmneziaWG) and the per-group
|
||||
// egress copies a bound group actually dials — so the stats aggregator and the panel
|
||||
// reflect real health for all of them, not just the few a lazy urltest group has probed.
|
||||
//
|
||||
// It runs a BuildProbePlan, so it inherits the two properties that plan exists for:
|
||||
// every tag is measured with the probe URL of the group that owns it (T1 — a manual run
|
||||
// must not overwrite a group's entries with numbers taken against a different server),
|
||||
// and tags that share one dial path are measured once and recorded for all of them.
|
||||
//
|
||||
// It is a SINGLETON: if a run is already in flight it returns started=false and leaves
|
||||
// that run untouched; otherwise it claims the guard, snapshots the plan, launches
|
||||
// exactly one worker-pool goroutine (concurrency probeAllConcurrency, per-probe timeout
|
||||
// probeAllTimeout), and returns started=true.
|
||||
//
|
||||
// specs/fallbackURL are the model facts BuildProbePlan needs; see there.
|
||||
//
|
||||
// Apply-swap safety: the plan is snapshotted up front, so a config swap mid-run cannot
|
||||
// change what is being probed, and each job re-resolves its tag against the CURRENT box
|
||||
// at dial time — a tag the swap removed is skipped rather than dialled and recorded
|
||||
// dead (see probeJob).
|
||||
func (e *Engine) TestAllNodes(specs map[string]GroupProbeSpec, fallbackURL string) (started bool) {
|
||||
// Singleton guard: only one probe-all at a time.
|
||||
if !e.testRunning.CompareAndSwap(false, true) {
|
||||
return false
|
||||
}
|
||||
|
||||
jobs := e.BuildProbePlan(specs, fallbackURL)
|
||||
e.testTotal.Store(int64(len(jobs)))
|
||||
e.testDone.Store(0)
|
||||
|
||||
// This run covers EVERY probeable outbound, so its tag set is exactly the set a dead
|
||||
// verdict can still be about. Anything else in the overlay refers to an outbound that
|
||||
// no longer exists (a node deleted from the config, a per-group copy that vanished
|
||||
// when a group lost its egress binding) — drop it now, which is what keeps the
|
||||
// overlay bounded. Skipped for an empty run, so a probe-all against a stopped engine
|
||||
// cannot wipe verdicts collected while it was up.
|
||||
keep := make(map[string]bool, len(jobs))
|
||||
for _, j := range jobs {
|
||||
for _, tag := range j.Store {
|
||||
keep[tag] = true
|
||||
}
|
||||
}
|
||||
e.pruneProbeFailed(keep)
|
||||
|
||||
go func() {
|
||||
defer e.testRunning.Store(false)
|
||||
if len(jobs) == 0 {
|
||||
return
|
||||
}
|
||||
sem := make(chan struct{}, probeAllConcurrency)
|
||||
var wg sync.WaitGroup
|
||||
for _, j := range jobs {
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(j ProbeJob) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
e.probeJob(j)
|
||||
e.testDone.Add(1)
|
||||
}(j)
|
||||
}
|
||||
wg.Wait()
|
||||
}()
|
||||
return true
|
||||
}
|
||||
|
||||
// probeJob resolves a job's dial tag against the CURRENT box and probes it, recording
|
||||
// the outcome under every tag the job covers.
|
||||
//
|
||||
// Resolving at dial time (rather than holding the adapter.Outbound captured when the
|
||||
// plan was built) is what makes a plan safe across an Apply swap: an outbound the new
|
||||
// config dropped simply no longer resolves, and the job is SKIPPED. Recording it dead
|
||||
// would be a fabricated verdict about a node that was never dialled — and, worse, one
|
||||
// that outlives the config change.
|
||||
//
|
||||
// Outbound(tag) falls through to the endpoint manager, so wireguard/AmneziaWG nodes and
|
||||
// their per-group copies resolve here exactly like plain outbounds.
|
||||
func (e *Engine) probeJob(j ProbeJob) {
|
||||
inst := e.Instance()
|
||||
if inst == nil {
|
||||
return
|
||||
}
|
||||
om := inst.Outbound()
|
||||
if om == nil {
|
||||
return
|
||||
}
|
||||
ob, ok := om.Outbound(j.Dial)
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
hist := e.URLTestHistory()
|
||||
if hist == nil {
|
||||
return
|
||||
}
|
||||
e.probeOneInto(ob, j.URL, j.Store, hist)
|
||||
}
|
||||
|
||||
// probeOneInto probes one outbound and records the result in the right place, for every
|
||||
// tag the measurement covers:
|
||||
//
|
||||
// success — the measured delay into the engine's shared history, exactly as the
|
||||
// groups' own probing does. It is a real measurement, so it belongs there
|
||||
// and the groups' selection legitimately benefits from it. Any stale dead
|
||||
// verdict for the tag is dropped.
|
||||
// failure — into OUR overlay (MarkProbeFailed), never into the engine's history.
|
||||
// See health.go: an invented entry there is read as "alive" by the
|
||||
// round_robin pool planner.
|
||||
//
|
||||
// store may name MORE tags than the one dialled. That is not extrapolation: the planner
|
||||
// only groups tags whose dial path and probe URL are identical (per-group copies of one
|
||||
// node through one egress), so the single measurement is literally the answer for each
|
||||
// of them. See probeplan.go.
|
||||
//
|
||||
// The delay is clamped to >=1ms because 0 is the engine's "unset" value in its own
|
||||
// selection arithmetic (protocol/group/urltest.go Select: `minDelay == 0 ||` treats 0
|
||||
// as no-value), so a genuinely sub-millisecond node must not report it.
|
||||
//
|
||||
// The probe is bounded by probeAllTimeout; a closed box just fails the dial.
|
||||
func (e *Engine) probeOneInto(ob adapter.Outbound, probeURL string, store []string, hist *urltest.HistoryStorage) {
|
||||
ctx, cancel := context.WithTimeout(context.Background(), probeAllTimeout)
|
||||
defer cancel()
|
||||
|
||||
delay, err := urltest.URLTest(ctx, probeURL, ob)
|
||||
if err != nil {
|
||||
for _, tag := range store {
|
||||
e.MarkProbeFailed(tag)
|
||||
}
|
||||
return
|
||||
}
|
||||
if delay == 0 {
|
||||
delay = 1
|
||||
}
|
||||
now := time.Now()
|
||||
for _, tag := range store {
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{Time: now, Delay: delay})
|
||||
e.clearProbeFailed(tag)
|
||||
}
|
||||
}
|
||||
|
||||
// probeOne probes a single outbound and records it under its own tag only. It is
|
||||
// probeOneInto's one-tag form, kept for callers that have an outbound rather than a
|
||||
// plan.
|
||||
func (e *Engine) probeOne(ob adapter.Outbound, probeURL string, hist *urltest.HistoryStorage) {
|
||||
e.probeOneInto(ob, probeURL, []string{ob.Tag()}, hist)
|
||||
}
|
||||
|
||||
// NodeTestStatus reports the manual probe-all progress: whether a run is in flight,
|
||||
// how many nodes have finished, and the run's total. When idle it returns the last
|
||||
// finished run's counters (running=false). Lock-free (atomics), safe to poll hot.
|
||||
func (e *Engine) NodeTestStatus() (running bool, done, total int) {
|
||||
return e.testRunning.Load(), int(e.testDone.Load()), int(e.testTotal.Load())
|
||||
}
|
||||
@@ -1,93 +0,0 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// TestShouldProbeOutbound pins the "which outbounds are real nodes" filter used by
|
||||
// TestAllNodes: real proxy nodes are probed; groups, utilities, interface egresses
|
||||
// and per-chain hop copies are skipped.
|
||||
func TestShouldProbeOutbound(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
typ string
|
||||
tag string
|
||||
want bool
|
||||
}{
|
||||
// Real proxy nodes (tag == node name) — probe these.
|
||||
{"vless node", C.TypeVLESS, "🇩🇪 Frankfurt-01", true},
|
||||
{"trojan node", C.TypeTrojan, "premium-node", true},
|
||||
{"wireguard node", C.TypeWireGuard, "wg-home", true},
|
||||
{"shadowsocks node", C.TypeShadowsocks, "ss-jp", true},
|
||||
|
||||
// Groups — virtual outbounds over other nodes, never a node.
|
||||
{"selector group", C.TypeSelector, "auto", false},
|
||||
{"urltest group", C.TypeURLTest, "best", false},
|
||||
|
||||
// Utilities — not proxy hops.
|
||||
{"direct", C.TypeDirect, "direct", false},
|
||||
{"block", C.TypeBlock, "block", false},
|
||||
{"dns", C.TypeDNS, "dns-out", false},
|
||||
|
||||
// Interface egress (netplane.EgressOutboundTag == "egress-"+name).
|
||||
{"egress", C.TypeDirect, "egress-wan", false},
|
||||
{"egress-shaped proxy tag", C.TypeVLESS, "egress-vpn", false},
|
||||
|
||||
// Per-chain hop copies (generate/chain.go): chain-<name>-h<i>[-<member>].
|
||||
{"chain hop", C.TypeVLESS, "chain-office-h1", false},
|
||||
{"chain group member", C.TypeTrojan, "chain-office-h2-node-a", false},
|
||||
|
||||
// Per-group EGRESS copies (generate/group.go): group-<name>-m<i>-<member>.
|
||||
// These ARE probed — they are the outbounds an egress-bound group actually
|
||||
// dials, and this run is the only writer that can ever mark one of them dead
|
||||
// (the group deletes history on failure instead of marking it). Their base
|
||||
// twins measure a different path and say nothing about the group.
|
||||
{"group egress copy", C.TypeVLESS, "group-vpn-m0-node-a", true},
|
||||
{"group egress copy (endpoint)", C.TypeWireGuard, "group-vpn-m7-wg-jp", true},
|
||||
// A GROUP itself is still skipped — it is a virtual outbound over the copies,
|
||||
// caught by the type switch before any prefix is considered.
|
||||
{"group outbound named like a copy", C.TypeURLTest, "group-vpn-m0-node-a", false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := shouldProbeOutbound(c.typ, c.tag); got != c.want {
|
||||
t.Errorf("%s: shouldProbeOutbound(%q, %q) = %v, want %v", c.name, c.typ, c.tag, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestNodeTestStatusIdle a bare (never-started) engine reports an idle, zeroed run.
|
||||
func TestNodeTestStatusIdle(t *testing.T) {
|
||||
e := New()
|
||||
running, done, total := e.NodeTestStatus()
|
||||
if running || done != 0 || total != 0 {
|
||||
t.Fatalf("idle NodeTestStatus = (%v,%d,%d), want (false,0,0)", running, done, total)
|
||||
}
|
||||
}
|
||||
|
||||
// TestTestAllNodesStoppedEngine kicking off a run against a stopped engine (no box,
|
||||
// no outbounds) is a valid no-op run: it starts (claims the guard), finds nothing to
|
||||
// probe, and settles back to not-running with a zero total.
|
||||
func TestTestAllNodesStoppedEngine(t *testing.T) {
|
||||
e := New()
|
||||
if !e.TestAllNodes(nil, "") {
|
||||
t.Fatal("TestAllNodes on a stopped engine: started=false, want true (no-op run)")
|
||||
}
|
||||
// The empty run completes essentially immediately; give the goroutine a beat.
|
||||
deadline := time.Now().Add(2 * time.Second)
|
||||
for {
|
||||
running, _, total := e.NodeTestStatus()
|
||||
if !running {
|
||||
if total != 0 {
|
||||
t.Fatalf("stopped-engine run total = %d, want 0", total)
|
||||
}
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
t.Fatal("probe-all run never finished on a stopped engine")
|
||||
}
|
||||
time.Sleep(10 * time.Millisecond)
|
||||
}
|
||||
}
|
||||
+311
-197
@@ -3,216 +3,337 @@ package engine
|
||||
import (
|
||||
"fmt"
|
||||
"sort"
|
||||
"strings"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// Probe planning: WHAT to dial, WITH WHICH probe URL, and WHOSE tags a single
|
||||
// measurement is allowed to fill in.
|
||||
// Probe planning for the observatory (plan §5.C): WHAT to dial and WHOSE tags a
|
||||
// single measurement fills in, derived from the APPLIED configuration.
|
||||
//
|
||||
// # Why a plan instead of "walk the outbounds and probe them"
|
||||
// # Reachability, not population
|
||||
//
|
||||
// The shared urltest history is keyed by outbound tag, and the generator emits one
|
||||
// outbound per (node, egress) pair, so the key granularity already matches the dial
|
||||
// path: a node dialled directly, the same node inside a group bound to AmneziaWG, and
|
||||
// the same node inside another group bound to a second WAN are three different tags
|
||||
// carrying three independent measurements. Nothing is mixed, and nothing needs fixing
|
||||
// there.
|
||||
// The old sweep walked every outbound in the box. The observatory instead starts
|
||||
// from what the routing configuration can actually REACH — the targets of the
|
||||
// emitted route rules, route.Final, and the DNS servers' detours (device targets
|
||||
// are materialised into rules by the generator, so they are already covered) —
|
||||
// and expands each reference along the real dial path:
|
||||
//
|
||||
// What the store has no concept of is WHAT THE MEASUREMENT WAS TAKEN WITH. Every group
|
||||
// may carry its own probe URL (generate.groupProbeURL: the group's, else globals',
|
||||
// else gstatic), and least_test ranks members by comparing stored delays as if they
|
||||
// were homogeneous. Within one group they are. But a writer that probes with a
|
||||
// DIFFERENT URL and stores under the same tag silently makes them incomparable — and
|
||||
// that is exactly what a naive "test all nodes" used to do: take the global URL and
|
||||
// overwrite entries belonging to a group that deliberately set its own. The operator
|
||||
// who configured a group-specific probe then got member selection driven by numbers
|
||||
// measured against somebody else's server.
|
||||
// - a GROUP tag (any strategy, selector included) expands to its members: the
|
||||
// candidates selection can pick — for an egress-bound group those are the
|
||||
// per-group copies "group-<g>-m<i>-<member>", the outbounds the group
|
||||
// actually dials;
|
||||
// - a CHAIN entry tag "chain-<n>-hN" (the exit wrapper, generate/chain.go) is
|
||||
// probed ITSELF: dialling the copy pulls the whole L1→…→Ln path through its
|
||||
// Detour links, so one probe is the chain's end-to-end health. Intermediate
|
||||
// NODE hops are not probed separately — their death is visible in the exit
|
||||
// probe and no selection depends on them;
|
||||
// - a GROUP hop inside a chain ("chain-<n>-h<i>", a wrapper selector reached by
|
||||
// walking the exit's Detour links) additionally has its member copies
|
||||
// "chain-<n>-h<i>-<member>" probed: they are the wrapper's selection
|
||||
// candidates, each dialling through its own prefix of the chain;
|
||||
// - a base node tag / endpoint (wireguard/AWG) is probed as itself.
|
||||
//
|
||||
// So every writer here measures a tag with the probe URL of the group that OWNS it.
|
||||
// That is the whole reason this file exists.
|
||||
// Everything the walk never reaches is deliberately left alone: it is not used
|
||||
// by any enabled rule, stays honestly "untested", and the panel marks its group
|
||||
// "unused" (the used-set this planner returns).
|
||||
//
|
||||
// # The unavoidable limitation (read this before "fixing" it)
|
||||
// # Store aliases
|
||||
//
|
||||
// One base tag can belong to two UNBOUND groups that set DIFFERENT probe URLs. There is
|
||||
// exactly one history record for that tag, so no writer can make it simultaneously
|
||||
// comparable inside both groups — the ambiguity is in the engine's keying, not in our
|
||||
// scheduling, and it cannot be resolved without a per-(tag,URL) store, which is the
|
||||
// engine's data structure and not ours to change.
|
||||
//
|
||||
// We therefore do NOT pick a winner and pretend. Such a tag is measured with the
|
||||
// FALLBACK (global) URL: a neutral instrument belonging to neither group, so both are
|
||||
// equally and visibly approximate, rather than one of them being silently
|
||||
// authoritative. See resolveProbeURL.
|
||||
//
|
||||
// Note also that this only bounds OUR contribution. The groups themselves still write
|
||||
// to that same shared key with their own URLs (protocol/group/urltest.go testNodes), so
|
||||
// its value keeps alternating between instruments no matter what we do. Bound groups
|
||||
// are free of this by construction: their members are per-group copies, so each group
|
||||
// owns its tags outright.
|
||||
// An egress copy's measurement is recorded under the copy AND under the base
|
||||
// node tag (parseGroupCopyTag recovers the node name), so the per-node stats
|
||||
// surface keeps filling in without the base outbound — a DIFFERENT dial path —
|
||||
// ever being dialled by the observatory. Chain copies are recorded only under
|
||||
// themselves: a chain prefix is nobody else's path.
|
||||
//
|
||||
// # Dedup
|
||||
//
|
||||
// Two groups bound to the SAME egress and probing with the SAME URL produce two
|
||||
// different member tags for one node — but the dial path is byte-for-byte identical
|
||||
// (generate/group.go rebuilds the node with the same detour). Probing both is pure
|
||||
// waste, and across three groups it triples a 376-node sweep. A job therefore carries
|
||||
// one tag to Dial and the full list of tags to Store the result under.
|
||||
|
||||
// GroupProbeSpec is what the planner needs to know about one group, sourced from the
|
||||
// MODEL (the running box exposes neither a group's probe URL nor its egress binding —
|
||||
// the latter is visible only as a tag convention). The caller resolves it; see
|
||||
// apply.groupProbeSpecs.
|
||||
type GroupProbeSpec struct {
|
||||
// Egress is the group's Group.Egress, trimmed. "" means no binding, i.e. the
|
||||
// group's members are the nodes' own base outbounds.
|
||||
Egress string
|
||||
// ProbeURL is the group's EFFECTIVE probe URL — its own, else globals'. "" means
|
||||
// "no opinion", and the planner uses the fallback.
|
||||
ProbeURL string
|
||||
}
|
||||
// The probe URL is global (plan §5.D), so the dedup key is the dial path alone:
|
||||
// two groups bound to the same egress rebuild byte-identical copies of one node
|
||||
// (generate/group.go), and one measurement is literally the answer for each of
|
||||
// them. A job therefore carries one tag to Dial and the full list of tags to
|
||||
// Store the result under.
|
||||
|
||||
// ProbeJob is one measurement: dial Dial, using URL, and record the outcome under
|
||||
// every tag in Store (which always contains Dial).
|
||||
//
|
||||
// Store having more than one element is never an approximation: those tags are copies
|
||||
// of the same node through the same egress measured against the same URL — the same
|
||||
// dial path — so one measurement is literally the answer for each of them.
|
||||
// Store having more than one element is never an approximation: those tags are
|
||||
// either copies of the same node through the same egress (the same dial path) or
|
||||
// the base-tag alias of such a copy (see the file header).
|
||||
type ProbeJob struct {
|
||||
Dial string
|
||||
URL string
|
||||
Store []string
|
||||
}
|
||||
|
||||
// BuildProbePlan enumerates every probeable outbound in the running box and returns the
|
||||
// deduplicated set of measurements that covers it, each bound to the right probe URL.
|
||||
// BuildObservatoryPlan derives the observatory's probe plan from the applied
|
||||
// configuration: the deduplicated measurements covering everything reachable from
|
||||
// the routing rules, plus the set of tags that walk visited (the "used" set the
|
||||
// panel's unused badge is built from).
|
||||
//
|
||||
// specs maps group name -> its model facts (a missing group is treated as unbound with
|
||||
// no URL opinion). fallbackURL is the global probe URL, used for tags no group claims
|
||||
// and for tags whose ownership is ambiguous; "" lets urltest fall back to its gstatic
|
||||
// It is a PURE function of opts — no box, no model — because the generator is the
|
||||
// single owner of the tag schema and opts is where that schema is materialised.
|
||||
// probeURL is the global probe URL; "" lets urltest fall back to its gstatic
|
||||
// default, exactly as the generator does.
|
||||
//
|
||||
// The result is deterministic (sorted) so a cursor walking it across ticks is stable.
|
||||
// A stopped engine yields nil.
|
||||
func (e *Engine) BuildProbePlan(specs map[string]GroupProbeSpec, fallbackURL string) []ProbeJob {
|
||||
inst := e.Instance()
|
||||
if inst == nil {
|
||||
return nil
|
||||
// The result is deterministic (sorted by Dial) so a cursor walking it across
|
||||
// ticks is stable, and plansEqual can tell a no-op reconcile from real change.
|
||||
func BuildObservatoryPlan(opts option.Options, probeURL string) ([]ProbeJob, map[string]bool) {
|
||||
w := &planWalk{
|
||||
byTag: planIndex(opts),
|
||||
used: map[string]bool{},
|
||||
visited: map[string]bool{},
|
||||
walked: map[string]bool{},
|
||||
targets: map[string]*planTarget{},
|
||||
}
|
||||
om := inst.Outbound()
|
||||
if om == nil {
|
||||
return nil
|
||||
for _, root := range planRoots(opts) {
|
||||
w.visit(root)
|
||||
}
|
||||
|
||||
// (1) Every tag worth probing. Endpoints are a SEPARATE manager and absent from
|
||||
// Outbounds() — omitting them is what used to leave every wireguard/AmneziaWG node
|
||||
// permanently "untested".
|
||||
probeable := map[string]bool{}
|
||||
var order []string
|
||||
claim := func(typ, tag string) {
|
||||
if probeable[tag] || !shouldProbeOutbound(typ, tag) {
|
||||
return
|
||||
jobs := make([]ProbeJob, 0, len(w.targets))
|
||||
for _, t := range w.targets {
|
||||
store := make([]string, 0, len(t.store))
|
||||
for tag := range t.store {
|
||||
store = append(store, tag)
|
||||
}
|
||||
probeable[tag] = true
|
||||
order = append(order, tag)
|
||||
}
|
||||
for _, ob := range om.Outbounds() {
|
||||
claim(ob.Type(), ob.Tag())
|
||||
}
|
||||
if em := inst.Endpoint(); em != nil {
|
||||
for _, ep := range em.Endpoints() {
|
||||
claim(ep.Type(), ep.Tag())
|
||||
}
|
||||
}
|
||||
|
||||
// (2) The groups, as (name, member tags) pairs.
|
||||
var groups []planGroup
|
||||
for _, ob := range om.Outbounds() {
|
||||
if !isGroupOutbound(ob.Type(), ob.Tag()) {
|
||||
continue
|
||||
}
|
||||
if g, ok := ob.(adapter.OutboundGroup); ok {
|
||||
groups = append(groups, planGroup{name: ob.Tag(), members: g.All()})
|
||||
}
|
||||
}
|
||||
|
||||
return planProbes(order, probeable, groups, specs, fallbackURL)
|
||||
}
|
||||
|
||||
// planGroup is the slice of a group the planner needs: its name (== its outbound tag,
|
||||
// and the key into specs) and the member tags it balances over.
|
||||
type planGroup struct {
|
||||
name string
|
||||
members []string
|
||||
}
|
||||
|
||||
// planProbes is BuildProbePlan without the box — the whole planning decision, kept pure
|
||||
// so the dedup and probe-URL rules can be tested directly rather than through a
|
||||
// stood-up engine.
|
||||
//
|
||||
// order/probeable are the probeable tags (order fixes iteration, probeable is the
|
||||
// membership test). See BuildProbePlan for the semantics of specs/fallbackURL.
|
||||
func planProbes(order []string, probeable map[string]bool, groups []planGroup, specs map[string]GroupProbeSpec, fallbackURL string) []ProbeJob {
|
||||
// (a) Ask every group which tags it owns, with which URL, and along which dial path.
|
||||
// dialKey is what dedup collapses on: a plain tag is its own path, while a per-group
|
||||
// egress copy's path is fully described by (node, egress) — which is exactly what
|
||||
// makes two groups' copies of one node interchangeable.
|
||||
dialKeyOf := map[string]string{}
|
||||
urlsOf := map[string]map[string]bool{} // tag -> distinct URLs claimed for it
|
||||
for _, g := range groups {
|
||||
spec := specs[g.name]
|
||||
for _, tag := range g.members {
|
||||
if !probeable[tag] {
|
||||
// A member that is not a probeable outbound (a nested group, or an
|
||||
// outbound this box no longer has) is not a measurement target.
|
||||
continue
|
||||
}
|
||||
if node, isCopy := parseGroupCopyTag(g.name, tag); isCopy {
|
||||
dialKeyOf[tag] = dialKeyCopy(node, spec.Egress)
|
||||
} else {
|
||||
dialKeyOf[tag] = dialKeyTag(tag)
|
||||
}
|
||||
if urlsOf[tag] == nil {
|
||||
urlsOf[tag] = map[string]bool{}
|
||||
}
|
||||
urlsOf[tag][spec.ProbeURL] = true
|
||||
}
|
||||
}
|
||||
|
||||
// (b) Resolve each tag's URL, then bucket by (dial path, URL).
|
||||
urlFor := make(map[string]string, len(order))
|
||||
buckets := map[string][]string{}
|
||||
for _, tag := range order {
|
||||
url := resolveProbeURL(urlsOf[tag], fallbackURL)
|
||||
urlFor[tag] = url
|
||||
key, ok := dialKeyOf[tag]
|
||||
if !ok {
|
||||
key = dialKeyTag(tag) // claimed by no group: its own path, global URL
|
||||
}
|
||||
bucket := key + lenPrefixed(url)
|
||||
buckets[bucket] = append(buckets[bucket], tag)
|
||||
}
|
||||
|
||||
// (c) Emit one job per bucket, deterministically.
|
||||
jobs := make([]ProbeJob, 0, len(buckets))
|
||||
for _, tags := range buckets {
|
||||
sort.Strings(tags)
|
||||
jobs = append(jobs, ProbeJob{Dial: tags[0], URL: urlFor[tags[0]], Store: tags})
|
||||
sort.Strings(store)
|
||||
jobs = append(jobs, ProbeJob{Dial: t.dial, URL: probeURL, Store: store})
|
||||
}
|
||||
sort.Slice(jobs, func(i, j int) bool { return jobs[i].Dial < jobs[j].Dial })
|
||||
return jobs
|
||||
return jobs, w.used
|
||||
}
|
||||
|
||||
// dialKeyTag / dialKeyCopy build the dedup key for one member.
|
||||
// planEntry is the slice of one configured outbound/endpoint the planner needs.
|
||||
type planEntry struct {
|
||||
typ string
|
||||
group bool // selector/urltest — expanded, never dialled
|
||||
members []string // group members, in config order
|
||||
detour string // DialerOptions.Detour ("" when none)
|
||||
}
|
||||
|
||||
// planIndex flattens opts into tag -> planEntry. Endpoints (wireguard/AmneziaWG)
|
||||
// are a separate list in the config and would otherwise be invisible — the same
|
||||
// omission that once left every endpoint permanently "untested".
|
||||
func planIndex(opts option.Options) map[string]planEntry {
|
||||
byTag := make(map[string]planEntry, len(opts.Outbounds)+len(opts.Endpoints))
|
||||
for _, ob := range opts.Outbounds {
|
||||
byTag[ob.Tag] = planEntry{
|
||||
typ: ob.Type,
|
||||
group: ob.Type == C.TypeSelector || ob.Type == C.TypeURLTest,
|
||||
members: groupMembersOf(ob.Options),
|
||||
detour: detourOf(ob.Options),
|
||||
}
|
||||
}
|
||||
for _, ep := range opts.Endpoints {
|
||||
byTag[ep.Tag] = planEntry{typ: ep.Type, detour: detourOf(ep.Options)}
|
||||
}
|
||||
return byTag
|
||||
}
|
||||
|
||||
// planRoots is every outbound reference the applied config makes: emitted rule
|
||||
// targets, route.Final, and each DNS server's detour. This is exactly the set
|
||||
// that passed through generate.resolveTarget — the single resolution point for
|
||||
// rule/device/dns-detour/egress targets — after generation materialised it.
|
||||
func planRoots(opts option.Options) []string {
|
||||
var roots []string
|
||||
if opts.Route != nil {
|
||||
for _, r := range opts.Route.Rules {
|
||||
switch r.Type {
|
||||
case C.RuleTypeDefault:
|
||||
if tag := ruleActionOutbound(r.DefaultOptions.RuleAction); tag != "" {
|
||||
roots = append(roots, tag)
|
||||
}
|
||||
case C.RuleTypeLogical:
|
||||
if tag := ruleActionOutbound(r.LogicalOptions.RuleAction); tag != "" {
|
||||
roots = append(roots, tag)
|
||||
}
|
||||
}
|
||||
}
|
||||
if opts.Route.Final != "" {
|
||||
roots = append(roots, opts.Route.Final)
|
||||
}
|
||||
}
|
||||
if opts.DNS != nil {
|
||||
for _, srv := range opts.DNS.Servers {
|
||||
if d := detourOf(srv.Options); d != "" {
|
||||
roots = append(roots, d)
|
||||
}
|
||||
}
|
||||
}
|
||||
return roots
|
||||
}
|
||||
|
||||
// ruleActionOutbound extracts the outbound tag of a route action; "" for every
|
||||
// other action (reject, hijack-dns, sniff, route-options, …).
|
||||
func ruleActionOutbound(a option.RuleAction) string {
|
||||
if a.Action == C.RuleActionTypeRoute {
|
||||
return a.RouteOptions.Outbound
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// groupMembersOf returns a selector/urltest option struct's member list.
|
||||
func groupMembersOf(options any) []string {
|
||||
switch o := options.(type) {
|
||||
case *option.SelectorOutboundOptions:
|
||||
return o.Outbounds
|
||||
case option.SelectorOutboundOptions:
|
||||
return o.Outbounds
|
||||
case *option.URLTestOutboundOptions:
|
||||
return o.Outbounds
|
||||
case option.URLTestOutboundOptions:
|
||||
return o.Outbounds
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// detourOf reads DialerOptions.Detour out of any outbound/endpoint/dns-server
|
||||
// option struct that carries dialer options (they all embed option.DialerOptions
|
||||
// and therefore satisfy DialerOptionsWrapper); "" for everything else.
|
||||
func detourOf(options any) string {
|
||||
if w, ok := options.(option.DialerOptionsWrapper); ok {
|
||||
return w.TakeDialerOptions().Detour
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// probeablePlanTag reports whether a non-group tag is a dialable measurement
|
||||
// target: everything except the direct/block/dns utilities and the interface
|
||||
// egress outbounds ("egress-<name>", netplane.EgressOutboundTag) — an egress is
|
||||
// a WAN, not a proxy hop, and probing it would only pollute the store. Chain and
|
||||
// group copies ARE probeable: they are the real dial paths this plan exists for.
|
||||
func probeablePlanTag(typ, tag string) bool {
|
||||
switch typ {
|
||||
case C.TypeSelector, C.TypeURLTest, C.TypeDirect, C.TypeBlock, C.TypeDNS:
|
||||
return false
|
||||
}
|
||||
return !strings.HasPrefix(tag, "egress-")
|
||||
}
|
||||
|
||||
// planTarget accumulates one dial path's job: the tag to dial (the smallest
|
||||
// dialable tag, for determinism) and every tag the measurement is recorded under.
|
||||
type planTarget struct {
|
||||
dial string
|
||||
store map[string]bool
|
||||
}
|
||||
|
||||
// planWalk is the reachability expansion state.
|
||||
type planWalk struct {
|
||||
byTag map[string]planEntry
|
||||
used map[string]bool
|
||||
visited map[string]bool
|
||||
walked map[string]bool
|
||||
targets map[string]*planTarget // dial-path key -> job accumulator
|
||||
}
|
||||
|
||||
// addTarget registers tag as a measurement under the dial-path key, recording it
|
||||
// under itself plus aliases. When several dialable tags share one path (copies of
|
||||
// one node through one egress across groups) the smallest dials, deterministically.
|
||||
func (w *planWalk) addTarget(key, tag string, aliases ...string) {
|
||||
t := w.targets[key]
|
||||
if t == nil {
|
||||
t = &planTarget{dial: tag, store: map[string]bool{}}
|
||||
w.targets[key] = t
|
||||
} else if tag < t.dial {
|
||||
t.dial = tag
|
||||
}
|
||||
t.store[tag] = true
|
||||
for _, a := range aliases {
|
||||
t.store[a] = true
|
||||
}
|
||||
}
|
||||
|
||||
// visit expands one root/member reference. Groups recurse into their members;
|
||||
// leaves become measurements; a leaf's Detour link is then walked to surface the
|
||||
// group hops of a chain (walkDetour).
|
||||
func (w *planWalk) visit(tag string) {
|
||||
if tag == "" || w.visited[tag] {
|
||||
return
|
||||
}
|
||||
w.visited[tag] = true
|
||||
ent, ok := w.byTag[tag]
|
||||
if !ok {
|
||||
return // dangling reference; box.New would have refused it anyway
|
||||
}
|
||||
w.used[tag] = true
|
||||
if ent.group {
|
||||
for _, m := range ent.members {
|
||||
w.visitMember(tag, m)
|
||||
}
|
||||
return
|
||||
}
|
||||
if probeablePlanTag(ent.typ, tag) {
|
||||
w.addTarget(dialKeyTag(tag), tag)
|
||||
}
|
||||
w.walkDetour(ent.detour)
|
||||
}
|
||||
|
||||
// visitMember expands one member of a used group. An egress copy
|
||||
// ("group-<group>-m<i>-<member>") is keyed by its (node, egress) dial path — so
|
||||
// two groups bound to one egress share the measurement — and recorded under the
|
||||
// copy AND the base node tag (the store alias). Every other member is expanded
|
||||
// exactly like a root.
|
||||
func (w *planWalk) visitMember(group, member string) {
|
||||
ent, ok := w.byTag[member]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
if node, isCopy := parseGroupCopyTag(group, member); isCopy && !ent.group {
|
||||
w.used[member] = true
|
||||
w.addTarget(dialKeyCopy(node, ent.detour), member, node)
|
||||
w.walkDetour(ent.detour)
|
||||
return
|
||||
}
|
||||
w.visit(member)
|
||||
}
|
||||
|
||||
// walkDetour follows a probed tag's Detour links toward the router. A group met
|
||||
// on the way is a chain's GROUP HOP wrapper: its member copies are selection
|
||||
// candidates and are probed, each through its own prefix of the path. A plain
|
||||
// outbound met on the way is an intermediate node hop — marked used, never probed
|
||||
// on its own — and the walk continues through it.
|
||||
func (w *planWalk) walkDetour(tag string) {
|
||||
if tag == "" || w.walked[tag] {
|
||||
return
|
||||
}
|
||||
w.walked[tag] = true
|
||||
ent, ok := w.byTag[tag]
|
||||
if !ok {
|
||||
return
|
||||
}
|
||||
w.used[tag] = true
|
||||
if ent.group {
|
||||
next := ""
|
||||
for _, m := range ent.members {
|
||||
ment, ok := w.byTag[m]
|
||||
if !ok {
|
||||
continue
|
||||
}
|
||||
w.used[m] = true
|
||||
if !ment.group && probeablePlanTag(ment.typ, m) {
|
||||
w.addTarget(dialKeyTag(m), m)
|
||||
}
|
||||
if next == "" {
|
||||
next = ment.detour // every member detours into the same previous hop
|
||||
}
|
||||
}
|
||||
w.walkDetour(next)
|
||||
return
|
||||
}
|
||||
w.walkDetour(ent.detour)
|
||||
}
|
||||
|
||||
// dialKeyTag / dialKeyCopy build the dedup key for one measurement target.
|
||||
//
|
||||
// A plain member tag IS its own dial path (one tag, one outbound object). A per-group
|
||||
// egress copy's path is (node, egress): two groups bound to the same egress rebuild the
|
||||
// same node with the same detour, so their copies are interchangeable.
|
||||
// A plain tag IS its own dial path (one tag, one outbound object). A per-group
|
||||
// egress copy's path is (node, egress): two groups bound to the same egress
|
||||
// rebuild the same node with the same detour, so their copies are interchangeable.
|
||||
//
|
||||
// The parts are LENGTH-PREFIXED rather than joined by a separator. Node names and
|
||||
// egress names are free-form operator text, so no byte is reserved and any separator
|
||||
// could be forged — ("a", "b-c") and ("a-b", "c") must never collide into one key, or
|
||||
// two genuinely different dial paths would share a measurement.
|
||||
// egress names are free-form operator text, so no byte is reserved and any
|
||||
// separator could be forged — ("a", "b-c") and ("a-b", "c") must never collide
|
||||
// into one key, or two genuinely different dial paths would share a measurement.
|
||||
func dialKeyTag(tag string) string { return "t" + lenPrefixed(tag) }
|
||||
func dialKeyCopy(node, egress string) string {
|
||||
return "c" + lenPrefixed(node) + lenPrefixed(egress)
|
||||
@@ -220,29 +341,22 @@ func dialKeyCopy(node, egress string) string {
|
||||
|
||||
func lenPrefixed(s string) string { return fmt.Sprintf("%d:%s", len(s), s) }
|
||||
|
||||
// resolveProbeURL turns the set of probe URLs claimed for one tag into the URL that tag
|
||||
// will actually be measured with.
|
||||
//
|
||||
// no claim / only "" claims -> fallback (the global URL)
|
||||
// exactly one real claim -> that group's URL. This is the point: measure a tag
|
||||
// the way its owner measures it, so the numbers stay
|
||||
// comparable inside that group.
|
||||
// two or more distinct claims -> fallback. The documented ambiguity from the file
|
||||
// header: one record cannot be comparable in two groups
|
||||
// at once, so rather than make one of them silently
|
||||
// authoritative we use an instrument neither chose.
|
||||
func resolveProbeURL(claimed map[string]bool, fallback string) string {
|
||||
var only string
|
||||
var n int
|
||||
for u := range claimed {
|
||||
if u == "" {
|
||||
continue // "no opinion" is not a competing instrument
|
||||
// plansEqual reports whether two plans describe exactly the same measurements —
|
||||
// same jobs, same order, same dial tag, same URL, same store set. This is what
|
||||
// lets a no-op reconcile keep the observatory's cursor (see ConfigureObservatory).
|
||||
func plansEqual(a, b []ProbeJob) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
for i := range a {
|
||||
if a[i].Dial != b[i].Dial || a[i].URL != b[i].URL || len(a[i].Store) != len(b[i].Store) {
|
||||
return false
|
||||
}
|
||||
for j := range a[i].Store {
|
||||
if a[i].Store[j] != b[i].Store[j] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
only = u
|
||||
n++
|
||||
}
|
||||
if n == 1 {
|
||||
return only
|
||||
}
|
||||
return fallback
|
||||
return true
|
||||
}
|
||||
|
||||
+269
-171
@@ -1,208 +1,306 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"testing"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
)
|
||||
|
||||
// planFor is a small harness: every tag in probeable order, the groups, the specs.
|
||||
func planFor(t *testing.T, tags []string, groups []planGroup, specs map[string]GroupProbeSpec, fallback string) map[string]ProbeJob {
|
||||
// Fixture builders: a minimal applied config in the generator's tag schema.
|
||||
|
||||
func fixNode(tag, detour string) option.Outbound {
|
||||
return option.Outbound{
|
||||
Type: C.TypeVLESS,
|
||||
Tag: tag,
|
||||
Options: &option.VLESSOutboundOptions{
|
||||
DialerOptions: option.DialerOptions{Detour: detour},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
func fixGroup(typ, tag string, members ...string) option.Outbound {
|
||||
switch typ {
|
||||
case C.TypeSelector:
|
||||
return option.Outbound{Type: typ, Tag: tag, Options: &option.SelectorOutboundOptions{Outbounds: members}}
|
||||
default:
|
||||
return option.Outbound{Type: typ, Tag: tag, Options: &option.URLTestOutboundOptions{Outbounds: members}}
|
||||
}
|
||||
}
|
||||
|
||||
func fixUtility(typ, tag string) option.Outbound {
|
||||
return option.Outbound{Type: typ, Tag: tag, Options: &option.DirectOutboundOptions{}}
|
||||
}
|
||||
|
||||
func fixRoute(final string, targets ...string) *option.RouteOptions {
|
||||
var rules []option.Rule
|
||||
for _, tgt := range targets {
|
||||
rules = append(rules, option.Rule{
|
||||
Type: C.RuleTypeDefault,
|
||||
DefaultOptions: option.DefaultRule{
|
||||
RuleAction: option.RuleAction{
|
||||
Action: C.RuleActionTypeRoute,
|
||||
RouteOptions: option.RouteActionOptions{Outbound: tgt},
|
||||
},
|
||||
},
|
||||
})
|
||||
}
|
||||
return &option.RouteOptions{Rules: rules, Final: final}
|
||||
}
|
||||
|
||||
// planOf runs the planner and indexes the jobs by dial tag.
|
||||
func planOf(t *testing.T, opts option.Options) (map[string]ProbeJob, map[string]bool) {
|
||||
t.Helper()
|
||||
probeable := map[string]bool{}
|
||||
for _, tag := range tags {
|
||||
probeable[tag] = true
|
||||
}
|
||||
jobs := planProbes(tags, probeable, groups, specs, fallback)
|
||||
|
||||
// Every probeable tag must be covered exactly once across all jobs — a planner that
|
||||
// loses a tag silently stops reporting that node, and one that covers it twice
|
||||
// spends the budget twice and races itself into the same history key.
|
||||
seen := map[string]int{}
|
||||
byDial := map[string]ProbeJob{}
|
||||
jobs, used := BuildObservatoryPlan(opts, "https://probe.example/204")
|
||||
byDial := make(map[string]ProbeJob, len(jobs))
|
||||
for _, j := range jobs {
|
||||
if _, dup := byDial[j.Dial]; dup {
|
||||
t.Fatalf("plan dials %q twice", j.Dial)
|
||||
}
|
||||
byDial[j.Dial] = j
|
||||
found := false
|
||||
for _, tag := range j.Store {
|
||||
seen[tag]++
|
||||
if tag == j.Dial {
|
||||
found = true
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Errorf("job %+v: Store must always contain Dial", j)
|
||||
}
|
||||
}
|
||||
for _, tag := range tags {
|
||||
if seen[tag] != 1 {
|
||||
t.Errorf("tag %q covered %d times, want exactly 1 (jobs=%+v)", tag, seen[tag], jobs)
|
||||
}
|
||||
}
|
||||
return byDial
|
||||
return byDial, used
|
||||
}
|
||||
|
||||
// A tag owned by a group is measured with THAT GROUP's probe URL, and a tag no group
|
||||
// claims with the global one. This is the T1 contract: a run must never overwrite a
|
||||
// group's entries with numbers taken against a different server.
|
||||
func TestPlanProbeURLFollowsOwningGroup(t *testing.T) {
|
||||
tags := []string{"n1", "n2", "loner"}
|
||||
groups := []planGroup{
|
||||
{name: "fast", members: []string{"n1", "n2"}},
|
||||
func assertPlanHas(t *testing.T, byDial map[string]ProbeJob, dial string) ProbeJob {
|
||||
t.Helper()
|
||||
j, ok := byDial[dial]
|
||||
if !ok {
|
||||
t.Fatalf("plan is missing a probe of %q; plan = %v", dial, dialsOf(byDial))
|
||||
}
|
||||
specs := map[string]GroupProbeSpec{
|
||||
"fast": {ProbeURL: "https://fast.example/probe"},
|
||||
}
|
||||
byDial := planFor(t, tags, groups, specs, "https://global.example/204")
|
||||
return j
|
||||
}
|
||||
|
||||
for _, tag := range []string{"n1", "n2"} {
|
||||
if got := byDial[tag].URL; got != "https://fast.example/probe" {
|
||||
t.Errorf("%s measured with %q, want the OWNING group's URL", tag, got)
|
||||
func dialsOf(byDial map[string]ProbeJob) []string {
|
||||
out := make([]string, 0, len(byDial))
|
||||
for d := range byDial {
|
||||
out = append(out, d)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// TestObservatoryPlanReachability is the §6-S4 acceptance table: the plan holds
|
||||
// exactly the reachable tags — an unreferenced group/node is absent, a chain exit
|
||||
// present, a chain group-hop's members present, egress copies present (their
|
||||
// bases only as store aliases, never dialled), and a USED selector group's
|
||||
// members present.
|
||||
func TestObservatoryPlanReachability(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixUtility(C.TypeDirect, "direct"),
|
||||
fixUtility(C.TypeBlock, "block"),
|
||||
// Used urltest group over plain members.
|
||||
fixNode("n1", ""),
|
||||
fixNode("n2", ""),
|
||||
fixGroup(C.TypeURLTest, "auto", "n1", "n2"),
|
||||
// Used SELECTOR group: its members are pin candidates and must be
|
||||
// probed too (the panel shows their health for manual selection).
|
||||
fixNode("m1", ""),
|
||||
fixGroup(C.TypeSelector, "manual", "m1"),
|
||||
// Unreferenced group + its member: reachable from NO rule.
|
||||
fixNode("idle1", ""),
|
||||
fixGroup(C.TypeSelector, "idle", "idle1"),
|
||||
// A free-floating node no rule targets.
|
||||
fixNode("stray", ""),
|
||||
},
|
||||
Route: fixRoute("auto", "manual"),
|
||||
}
|
||||
byDial, used := planOf(t, opts)
|
||||
|
||||
for _, want := range []string{"n1", "n2", "m1"} {
|
||||
j := assertPlanHas(t, byDial, want)
|
||||
if len(j.Store) != 1 || j.Store[0] != want {
|
||||
t.Errorf("%s: store = %v, want just itself", want, j.Store)
|
||||
}
|
||||
}
|
||||
if got := byDial["loner"].URL; got != "https://global.example/204" {
|
||||
t.Errorf("unclaimed tag measured with %q, want the global URL", got)
|
||||
for _, absent := range []string{"idle", "idle1", "stray", "auto", "manual", "direct", "block"} {
|
||||
if _, ok := byDial[absent]; ok {
|
||||
t.Errorf("plan dials %q — unreachable or not a dialable node", absent)
|
||||
}
|
||||
}
|
||||
for _, u := range []string{"auto", "manual", "n1", "n2", "m1"} {
|
||||
if !used[u] {
|
||||
t.Errorf("used-set is missing %q", u)
|
||||
}
|
||||
}
|
||||
for _, nu := range []string{"idle", "idle1", "stray"} {
|
||||
if used[nu] {
|
||||
t.Errorf("used-set claims unreachable %q", nu)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A group with no URL opinion inherits the global one, and — importantly — does not
|
||||
// count as a competing instrument when another group DOES have an opinion.
|
||||
func TestPlanProbeURLNoOpinion(t *testing.T) {
|
||||
tags := []string{"n1"}
|
||||
groups := []planGroup{
|
||||
{name: "quiet", members: []string{"n1"}},
|
||||
{name: "picky", members: []string{"n1"}},
|
||||
// An egress-bound group's members are per-group COPIES; each copy is probed under
|
||||
// its (node, egress) dial path and recorded under the copy AND the base node tag,
|
||||
// while the base outbound itself — a different network path — is never dialled.
|
||||
// Two groups bound to the same egress share one measurement per node.
|
||||
func TestObservatoryPlanEgressCopies(t *testing.T) {
|
||||
copyA := "group-vpn-m0-n1"
|
||||
copyB := "group-vpn2-m0-n1"
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixNode("n1", ""), // base: emitted, but not a dial target here
|
||||
fixNode(copyA, "egress-wg0"),
|
||||
fixNode(copyB, "egress-wg0"),
|
||||
fixUtility(C.TypeDirect, "egress-wg0"),
|
||||
fixGroup(C.TypeSelector, "vpn", copyA),
|
||||
fixGroup(C.TypeSelector, "vpn2", copyB),
|
||||
},
|
||||
Route: fixRoute("vpn", "vpn2"),
|
||||
}
|
||||
specs := map[string]GroupProbeSpec{
|
||||
"quiet": {ProbeURL: ""}, // no opinion
|
||||
"picky": {ProbeURL: "https://picky.example/probe"},
|
||||
}
|
||||
byDial := planFor(t, tags, groups, specs, "https://global.example/204")
|
||||
if got := byDial["n1"].URL; got != "https://picky.example/probe" {
|
||||
t.Errorf("n1 measured with %q; a group with no opinion must not outvote one that has one", got)
|
||||
}
|
||||
}
|
||||
byDial, used := planOf(t, opts)
|
||||
|
||||
// THE DOCUMENTED AMBIGUITY: one base tag inside two unbound groups with DIFFERENT probe
|
||||
// URLs. There is a single history record, so it cannot be comparable in both. The
|
||||
// planner must fall back to the neutral global URL rather than silently making one
|
||||
// group authoritative.
|
||||
func TestPlanProbeURLConflictFallsBackToGlobal(t *testing.T) {
|
||||
tags := []string{"shared"}
|
||||
groups := []planGroup{
|
||||
{name: "a", members: []string{"shared"}},
|
||||
{name: "b", members: []string{"shared"}},
|
||||
}
|
||||
specs := map[string]GroupProbeSpec{
|
||||
"a": {ProbeURL: "https://a.example/probe"},
|
||||
"b": {ProbeURL: "https://b.example/probe"},
|
||||
}
|
||||
byDial := planFor(t, tags, groups, specs, "https://global.example/204")
|
||||
if got := byDial["shared"].URL; got != "https://global.example/204" {
|
||||
t.Errorf("conflicting owners: measured with %q, want the neutral global URL", got)
|
||||
}
|
||||
}
|
||||
|
||||
// Dedup: two groups bound to the SAME egress with the SAME probe URL produce different
|
||||
// copy tags for one node, but a single dial path. One measurement, recorded for both.
|
||||
func TestPlanDedupsIdenticalDialPaths(t *testing.T) {
|
||||
a0 := "group-a-m0-n1"
|
||||
b7 := "group-b-m7-n1"
|
||||
tags := []string{a0, b7}
|
||||
groups := []planGroup{
|
||||
{name: "a", members: []string{a0}},
|
||||
{name: "b", members: []string{b7}},
|
||||
}
|
||||
specs := map[string]GroupProbeSpec{
|
||||
"a": {Egress: "wg0", ProbeURL: "https://p.example/204"},
|
||||
"b": {Egress: "wg0", ProbeURL: "https://p.example/204"},
|
||||
}
|
||||
byDial := planFor(t, tags, groups, specs, "")
|
||||
if len(byDial) != 1 {
|
||||
t.Fatalf("got %d jobs, want 1 (same node, same egress, same URL = one dial path)", len(byDial))
|
||||
t.Fatalf("got %d jobs, want 1 — both copies share one (node, egress) dial path: %v",
|
||||
len(byDial), dialsOf(byDial))
|
||||
}
|
||||
j := byDial[a0]
|
||||
if !reflect.DeepEqual(j.Store, []string{a0, b7}) {
|
||||
t.Errorf("Store = %v, want both copy tags recorded from one measurement", j.Store)
|
||||
j := assertPlanHas(t, byDial, copyA) // smallest tag dials
|
||||
wantStore := map[string]bool{copyA: true, copyB: true, "n1": true}
|
||||
if len(j.Store) != len(wantStore) {
|
||||
t.Fatalf("store = %v, want copies + base alias", j.Store)
|
||||
}
|
||||
for _, tag := range j.Store {
|
||||
if !wantStore[tag] {
|
||||
t.Errorf("unexpected store tag %q", tag)
|
||||
}
|
||||
}
|
||||
if used["n1"] {
|
||||
// The BASE outbound is not on the group's dial path; only its alias is
|
||||
// written. It must not flip any usage badge.
|
||||
t.Error("base node tag must not be in the used-set for a bound group")
|
||||
}
|
||||
}
|
||||
|
||||
// Dedup must NOT collapse copies that differ in the thing that makes them different:
|
||||
// a different egress is a different network path, and a different probe URL is a
|
||||
// different instrument. Either one earns its own measurement.
|
||||
func TestPlanKeepsDistinctDialPathsApart(t *testing.T) {
|
||||
t.Run("different egress", func(t *testing.T) {
|
||||
a0, b0 := "group-a-m0-n1", "group-b-m0-n1"
|
||||
byDial := planFor(t, []string{a0, b0},
|
||||
[]planGroup{{name: "a", members: []string{a0}}, {name: "b", members: []string{b0}}},
|
||||
map[string]GroupProbeSpec{
|
||||
"a": {Egress: "wg0", ProbeURL: "https://p.example/204"},
|
||||
"b": {Egress: "wan2", ProbeURL: "https://p.example/204"},
|
||||
}, "")
|
||||
if len(byDial) != 2 {
|
||||
t.Fatalf("got %d jobs, want 2 — two egresses are two network paths", len(byDial))
|
||||
}
|
||||
})
|
||||
t.Run("different probe URL", func(t *testing.T) {
|
||||
a0, b0 := "group-a-m0-n1", "group-b-m0-n1"
|
||||
byDial := planFor(t, []string{a0, b0},
|
||||
[]planGroup{{name: "a", members: []string{a0}}, {name: "b", members: []string{b0}}},
|
||||
map[string]GroupProbeSpec{
|
||||
"a": {Egress: "wg0", ProbeURL: "https://a.example/204"},
|
||||
"b": {Egress: "wg0", ProbeURL: "https://b.example/204"},
|
||||
}, "")
|
||||
if len(byDial) != 2 {
|
||||
t.Fatalf("got %d jobs, want 2 — two instruments are two measurements", len(byDial))
|
||||
}
|
||||
for dial, j := range byDial {
|
||||
if len(j.Store) != 1 {
|
||||
t.Errorf("job %s stores %v; a measurement must not leak across instruments", dial, j.Store)
|
||||
}
|
||||
}
|
||||
})
|
||||
// A bound group's copy and the node's own base tag are never the same path either.
|
||||
t.Run("copy vs base tag", func(t *testing.T) {
|
||||
byDial := planFor(t, []string{"n1", "group-a-m0-n1"},
|
||||
[]planGroup{{name: "a", members: []string{"group-a-m0-n1"}}},
|
||||
map[string]GroupProbeSpec{"a": {Egress: "wg0"}}, "")
|
||||
if len(byDial) != 2 {
|
||||
t.Fatalf("got %d jobs, want 2 — through the tunnel and direct are different things", len(byDial))
|
||||
}
|
||||
})
|
||||
}
|
||||
// A chain: rule -> chain-c-h2 (exit, plain node copy) detouring into chain-c-h1
|
||||
// (a GROUP hop wrapper) whose member copies detour into nothing (L1). The exit is
|
||||
// probed end-to-end; the group hop's member copies are probed each through its
|
||||
// own prefix; the wrapper itself and the intermediate hop are used-but-not-dialled.
|
||||
func TestObservatoryPlanChain(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixNode("chain-c-h1-nodeA", ""),
|
||||
fixNode("chain-c-h1-nodeB", ""),
|
||||
fixGroup(C.TypeSelector, "chain-c-h1", "chain-c-h1-nodeA", "chain-c-h1-nodeB"),
|
||||
fixNode("chain-c-h2", "chain-c-h1"),
|
||||
// The chain's underlying base nodes exist standalone too, unreferenced.
|
||||
fixNode("nodeA", ""),
|
||||
fixNode("nodeB", ""),
|
||||
fixNode("exitnode", ""),
|
||||
},
|
||||
Route: fixRoute("", "chain-c-h2"),
|
||||
}
|
||||
byDial, used := planOf(t, opts)
|
||||
|
||||
// The length-prefixed dial key must not let two different (node, egress) pairs collide
|
||||
// into one measurement — node and egress names are free-form operator text, so a plain
|
||||
// separator could be forged.
|
||||
func TestDialKeyNoForgedCollision(t *testing.T) {
|
||||
if dialKeyCopy("a", "b-c") == dialKeyCopy("a-b", "c") {
|
||||
t.Fatal("dial keys collide across a forged separator: two paths would share one measurement")
|
||||
// Chain exit: probed itself (end-to-end), stored only under itself.
|
||||
j := assertPlanHas(t, byDial, "chain-c-h2")
|
||||
if len(j.Store) != 1 || j.Store[0] != "chain-c-h2" {
|
||||
t.Errorf("chain exit store = %v, want only itself (a chain prefix is nobody else's path)", j.Store)
|
||||
}
|
||||
if dialKeyCopy("n1", "") == dialKeyTag("n1") {
|
||||
t.Fatal("an unbound copy key must not collide with the plain tag key")
|
||||
// Group-hop members: probed, each under itself only.
|
||||
for _, m := range []string{"chain-c-h1-nodeA", "chain-c-h1-nodeB"} {
|
||||
j := assertPlanHas(t, byDial, m)
|
||||
if len(j.Store) != 1 || j.Store[0] != m {
|
||||
t.Errorf("chain member %s store = %v, want only itself", m, j.Store)
|
||||
}
|
||||
}
|
||||
if dialKeyCopy("x", "y") != dialKeyCopy("x", "y") {
|
||||
t.Fatal("dial key is not deterministic")
|
||||
// The wrapper selector is used but never dialled; the standalone bases are
|
||||
// neither.
|
||||
if _, ok := byDial["chain-c-h1"]; ok {
|
||||
t.Error("plan dials the chain group-hop wrapper; only its members are measurements")
|
||||
}
|
||||
for _, absent := range []string{"nodeA", "nodeB", "exitnode"} {
|
||||
if _, ok := byDial[absent]; ok {
|
||||
t.Errorf("plan dials unreferenced base node %q", absent)
|
||||
}
|
||||
}
|
||||
if !used["chain-c-h1"] || !used["chain-c-h2"] {
|
||||
t.Error("chain hop tags missing from the used-set")
|
||||
}
|
||||
}
|
||||
|
||||
// A member tag that is not a probeable outbound (a nested group, or one a config swap
|
||||
// removed) is not a measurement target and must not appear in the plan.
|
||||
func TestPlanSkipsUnprobeableMembers(t *testing.T) {
|
||||
jobs := planProbes(
|
||||
[]string{"n1"}, map[string]bool{"n1": true},
|
||||
[]planGroup{{name: "a", members: []string{"n1", "vanished"}}},
|
||||
nil, "")
|
||||
for _, j := range jobs {
|
||||
for _, tag := range j.Store {
|
||||
if tag == "vanished" {
|
||||
t.Fatal("plan targets a tag the box does not have")
|
||||
}
|
||||
}
|
||||
// DNS server detours and route.Final are roots too — a group referenced only as a
|
||||
// resolver's detour is still a used, probed path.
|
||||
func TestObservatoryPlanDNSDetourRoot(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixNode("n1", ""),
|
||||
fixGroup(C.TypeURLTest, "dnsproxy", "n1"),
|
||||
},
|
||||
DNS: &option.DNSOptions{RawDNSOptions: option.RawDNSOptions{
|
||||
Servers: []option.DNSServerOptions{{
|
||||
Type: C.DNSTypeUDP,
|
||||
Tag: "dns-remote",
|
||||
Options: &option.RemoteDNSServerOptions{
|
||||
RawLocalDNSServerOptions: option.RawLocalDNSServerOptions{
|
||||
DialerOptions: option.DialerOptions{Detour: "dnsproxy"},
|
||||
},
|
||||
},
|
||||
}},
|
||||
}},
|
||||
}
|
||||
byDial, used := planOf(t, opts)
|
||||
assertPlanHas(t, byDial, "n1")
|
||||
if !used["dnsproxy"] {
|
||||
t.Error("group referenced by a DNS detour missing from the used-set")
|
||||
}
|
||||
}
|
||||
|
||||
// A stopped engine plans nothing (and does not panic).
|
||||
func TestBuildProbePlanStoppedEngine(t *testing.T) {
|
||||
if jobs := New().BuildProbePlan(nil, ""); len(jobs) != 0 {
|
||||
t.Fatalf("stopped engine planned %d jobs, want 0", len(jobs))
|
||||
// Endpoints (wireguard/AmneziaWG) live in a separate config list and must be
|
||||
// reachable both as roots and as group members.
|
||||
func TestObservatoryPlanEndpoints(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixGroup(C.TypeSelector, "wg", "awg1"),
|
||||
},
|
||||
Endpoints: []option.Endpoint{{
|
||||
Type: C.TypeWireGuard,
|
||||
Tag: "awg1",
|
||||
Options: &option.WireGuardEndpointOptions{
|
||||
DialerOptions: option.DialerOptions{},
|
||||
},
|
||||
}},
|
||||
Route: fixRoute("wg"),
|
||||
}
|
||||
byDial, used := planOf(t, opts)
|
||||
assertPlanHas(t, byDial, "awg1")
|
||||
if !used["awg1"] {
|
||||
t.Error("endpoint member missing from the used-set")
|
||||
}
|
||||
}
|
||||
|
||||
// The plan is deterministic and self-equal, and a reordered store or different
|
||||
// dial breaks equality — the property ConfigureObservatory's cursor keeping
|
||||
// depends on.
|
||||
func TestObservatoryPlansEqual(t *testing.T) {
|
||||
opts := option.Options{
|
||||
Outbounds: []option.Outbound{
|
||||
fixNode("n1", ""),
|
||||
fixNode("n2", ""),
|
||||
fixGroup(C.TypeURLTest, "auto", "n1", "n2"),
|
||||
},
|
||||
Route: fixRoute("auto"),
|
||||
}
|
||||
a, _ := BuildObservatoryPlan(opts, "u")
|
||||
b, _ := BuildObservatoryPlan(opts, "u")
|
||||
if !plansEqual(a, b) {
|
||||
t.Fatal("two builds of one config must be measurement-for-measurement identical")
|
||||
}
|
||||
c, _ := BuildObservatoryPlan(opts, "other-url")
|
||||
if plansEqual(a, c) {
|
||||
t.Fatal("a different probe URL is a different plan")
|
||||
}
|
||||
if plansEqual(a, a[:len(a)-1]) {
|
||||
t.Fatal("a shorter plan must not compare equal")
|
||||
}
|
||||
}
|
||||
|
||||
// An empty config plans nothing (and does not panic).
|
||||
func TestObservatoryPlanEmpty(t *testing.T) {
|
||||
jobs, used := BuildObservatoryPlan(option.Options{}, "")
|
||||
if len(jobs) != 0 {
|
||||
t.Fatalf("empty config planned %d jobs, want 0", len(jobs))
|
||||
}
|
||||
if len(used) != 0 {
|
||||
t.Fatalf("empty config used-set = %v, want empty", used)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,379 +0,0 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"sync"
|
||||
"time"
|
||||
)
|
||||
|
||||
// The scheduled sweep — a background "observatory" that keeps node health fresh so the
|
||||
// strategies have something to choose from.
|
||||
//
|
||||
// # Why this is needed, and why it does NOT change any strategy
|
||||
//
|
||||
// The decoupling the owner asked for already exists in the engine: URLTestGroup.Select
|
||||
// reads the SHARED history store (g.history.LoadURLTestHistory), not measurements it
|
||||
// took itself. So anyone who writes fresh, comparable data into that store improves
|
||||
// every strategy's decisions without touching a single strategy. This file is only
|
||||
// that writer.
|
||||
//
|
||||
// What it fixes: a urltest group probes LAZILY — Touch() starts its ticker and an idle
|
||||
// group stops it — and a `single`/selector group never probes at all (selector.go dials
|
||||
// one fixed member). On a 376-node subscription that leaves nearly everything
|
||||
// permanently "untested", which means the per-group health numbers are empty until a
|
||||
// human presses "Test all nodes". That is the whole reason the sweep defaults to ON:
|
||||
// a knob that ships disabled would leave the problem it exists to solve in place.
|
||||
//
|
||||
// # What it must NOT do: replace a group's own probing
|
||||
//
|
||||
// A `failover` group's probe IS its failure detector — it switches on the probe tick,
|
||||
// not on a failed dial, so its 30s interval is the outage window (generate/group.go
|
||||
// failoverProbeInterval). A full sweep can never run that often, so the sweep is an
|
||||
// ADDITIONAL layer, never a substitute.
|
||||
//
|
||||
// It must also not SUPPRESS that probing, and it very nearly would: testNodes skips a
|
||||
// member whose history is younger than the group's interval
|
||||
// (`!force && history != nil && time.Since(history.Time) < g.interval`). If the sweep
|
||||
// kept refreshing a failover member, the group would keep skipping its own 30s check
|
||||
// and detection would slip. sweepFreshness is what prevents that: the sweep only
|
||||
// touches a tag whose newest observation is OLDER than 5 minutes, which is an order of
|
||||
// magnitude beyond the failover interval and beyond the engine's 3-minute default, so a
|
||||
// tag any group is actively probing is never eligible. The budget lands exactly where
|
||||
// it is meant to — on the nodes nobody is probing at all.
|
||||
//
|
||||
// # Cost, and the smearing that makes it affordable
|
||||
//
|
||||
// A full pass is not something a router can do on a short period. On the production
|
||||
// config — 376 nodes plus ~331 per-group egress copies ≈ 707 measurements — a single
|
||||
// batch at concurrency 16 with a 5s timeout is ~220s of worst-case wall time and 16
|
||||
// concurrent tunnel dials for the whole of it. Repeating that every few minutes is out
|
||||
// of the question on a quad-A53 router sharing its uplink with the user's traffic.
|
||||
//
|
||||
// So the pass is SMEARED: a cursor walks one plan across many small ticks.
|
||||
//
|
||||
// sweepInterval 10s tick period
|
||||
// sweepBatch 24 measurements started per tick
|
||||
// sweepConcurrency 12 in flight at once
|
||||
//
|
||||
// Worst case per tick is 24/12 x 5s = 10s — exactly one period, so a completely dead
|
||||
// fleet saturates the schedule but never stacks (a tick that finds the previous one
|
||||
// still running is skipped). Steady-state load is bounded at 12 concurrent probes and
|
||||
// ~2.4 dials/sec; against a HEAD to generate_204 that is a few KB/s and a handful of
|
||||
// TLS handshakes, which is noise next to the traffic the router is already proxying.
|
||||
// A healthy fleet answers in tens of milliseconds, so a tick's work finishes in about a
|
||||
// second and the router idles for the other nine.
|
||||
//
|
||||
// Cycle time: 707 measurements / 24 per tick = 30 ticks = ~5 minutes for the full
|
||||
// production config, ~2.7 minutes for 376 nodes with no bound group. Combined with the
|
||||
// 5-minute freshness gate that settles into a steady state where everything nobody else
|
||||
// probes is refreshed roughly every 5-10 minutes — fresh enough for a health display
|
||||
// and for least_test to rank on, without ever pretending to be a failure detector.
|
||||
const (
|
||||
// sweepInterval is the default tick period.
|
||||
sweepInterval = 10 * time.Second
|
||||
// sweepBatch is the default number of measurements started per tick.
|
||||
sweepBatch = 24
|
||||
// sweepConcurrency is the default number of measurements in flight at once. Half
|
||||
// of probeAllConcurrency: the manual run is a bounded burst the operator asked
|
||||
// for, this one runs forever.
|
||||
sweepConcurrency = 12
|
||||
// sweepFreshness is how recent an observation must be for the sweep to leave a tag
|
||||
// alone. Deliberately far above every group probe interval (failover 30s, engine
|
||||
// default 3m) so the sweep can never suppress a group's own health checking — see
|
||||
// the file header.
|
||||
sweepFreshness = 5 * time.Minute
|
||||
)
|
||||
|
||||
// SweepConfig is the scheduled sweep's configuration. The zero value is DISABLED; all
|
||||
// other fields fall back to the constants above when non-positive, so a caller only has
|
||||
// to set what it wants to change.
|
||||
type SweepConfig struct {
|
||||
Enabled bool
|
||||
Interval time.Duration
|
||||
Batch int
|
||||
Concurrency int
|
||||
// Specs/FallbackURL are the model facts BuildProbePlan needs — which group binds to
|
||||
// which egress and probes with which URL. They are refreshed on every
|
||||
// ConfigureSweep call, i.e. on every successful apply.
|
||||
Specs map[string]GroupProbeSpec
|
||||
FallbackURL string
|
||||
}
|
||||
|
||||
func (c SweepConfig) withDefaults() SweepConfig {
|
||||
if c.Interval <= 0 {
|
||||
c.Interval = sweepInterval
|
||||
}
|
||||
if c.Batch <= 0 {
|
||||
c.Batch = sweepBatch
|
||||
}
|
||||
if c.Concurrency <= 0 {
|
||||
c.Concurrency = sweepConcurrency
|
||||
}
|
||||
return c
|
||||
}
|
||||
|
||||
// sweepState is the sweeper's own mutable state, guarded by sweepMu.
|
||||
type sweepState struct {
|
||||
cfg SweepConfig
|
||||
// stop closes the running loop; nil when no loop is running.
|
||||
stop chan struct{}
|
||||
done chan struct{}
|
||||
// plan is the cycle currently being walked, cursor the position in it, and
|
||||
// planHash the engine config hash the plan was built from (a swap invalidates it).
|
||||
plan []ProbeJob
|
||||
cursor int
|
||||
planHash string
|
||||
cycles uint64
|
||||
// busy guards against a slow tick overlapping the next one.
|
||||
busy bool
|
||||
// planFn is the plan source. nil means Engine.BuildProbePlan, which is what
|
||||
// production always uses; a test substitutes a stub so the cursor-preservation
|
||||
// contract can be exercised without standing up a real box (the bug it guards
|
||||
// against only shows up across repeated reconfigurations of a NON-empty plan).
|
||||
planFn func(map[string]GroupProbeSpec, string) []ProbeJob
|
||||
}
|
||||
|
||||
// buildPlan is the single plan-source indirection. Caller holds sweepMu.
|
||||
func (e *Engine) buildPlan(cfg SweepConfig) []ProbeJob {
|
||||
if e.sweep.planFn != nil {
|
||||
return e.sweep.planFn(cfg.Specs, cfg.FallbackURL)
|
||||
}
|
||||
return e.BuildProbePlan(cfg.Specs, cfg.FallbackURL)
|
||||
}
|
||||
|
||||
// ConfigureSweep installs (or updates, or stops) the scheduled sweep. It is idempotent
|
||||
// and safe to call on every apply: the loop is started on the first enabled call,
|
||||
// re-tuned in place afterwards, and stopped by an Enabled=false call.
|
||||
//
|
||||
// # It must NOT restart the cycle (this was a release-blocking bug)
|
||||
//
|
||||
// It is called from applyLocked on every SUCCESSFUL apply — and on this router cron
|
||||
// reconciles EVERY MINUTE, so that is once a minute forever, almost always with an
|
||||
// identical config. The first version rebuilt the plan and zeroed the cursor here. A
|
||||
// full cycle is ~6 minutes (896 measurements at 24 per tick), so the cursor was reset
|
||||
// five minutes before it could ever finish: observed live walking 96 -> 24 -> 72 with
|
||||
// cycles stuck at 0. The sweep therefore never completed a pass, which in turn left the
|
||||
// nodes past the cursor permanently "untested" — the exact condition the whole feature
|
||||
// exists to remove, reintroduced by its own reconfiguration path.
|
||||
//
|
||||
// So progress is preserved across a reconfiguration that does not actually change the
|
||||
// work: syncPlanLocked rebuilds the plan and keeps the cursor when the new plan is
|
||||
// measurement-for-measurement identical to the old one. Only a genuinely different plan
|
||||
// restarts the walk, which is correct — the old cursor indexes a list that no longer
|
||||
// exists.
|
||||
func (e *Engine) ConfigureSweep(cfg SweepConfig) {
|
||||
cfg = cfg.withDefaults()
|
||||
e.sweepMu.Lock()
|
||||
defer e.sweepMu.Unlock()
|
||||
|
||||
e.sweep.cfg = cfg
|
||||
if cfg.Enabled {
|
||||
// Rebuild against the new specs/URLs, keeping our place if nothing moved.
|
||||
e.syncPlanLocked(cfg)
|
||||
} else {
|
||||
e.sweep.plan = nil
|
||||
e.sweep.cursor = 0
|
||||
e.sweep.planHash = ""
|
||||
}
|
||||
|
||||
switch {
|
||||
case cfg.Enabled && e.sweep.stop == nil:
|
||||
stop := make(chan struct{})
|
||||
done := make(chan struct{})
|
||||
e.sweep.stop, e.sweep.done = stop, done
|
||||
go e.sweepLoop(stop, done, cfg.Interval)
|
||||
case !cfg.Enabled && e.sweep.stop != nil:
|
||||
close(e.sweep.stop)
|
||||
e.sweep.stop, e.sweep.done = nil, nil
|
||||
}
|
||||
}
|
||||
|
||||
// StopSweep stops the scheduled sweep and waits for an in-flight tick to finish. It is
|
||||
// idempotent and safe to call when nothing is running (teardown path).
|
||||
//
|
||||
// The wait is deliberate and bounded by one batch (worst case Batch/Concurrency x
|
||||
// probeAllTimeout = 10s with the defaults): teardown closes the box immediately
|
||||
// afterwards, and probes still in flight would then fail their dials and be recorded as
|
||||
// failure verdicts about nodes that were never actually unreachable.
|
||||
func (e *Engine) StopSweep() {
|
||||
e.sweepMu.Lock()
|
||||
stop, done := e.sweep.stop, e.sweep.done
|
||||
e.sweep.stop, e.sweep.done = nil, nil
|
||||
e.sweep.cfg.Enabled = false
|
||||
e.sweep.plan = nil
|
||||
e.sweep.cursor = 0
|
||||
if stop != nil {
|
||||
close(stop)
|
||||
}
|
||||
e.sweepMu.Unlock()
|
||||
if done != nil {
|
||||
<-done
|
||||
}
|
||||
}
|
||||
|
||||
// SweepStatus reports the scheduled sweep's state for the panel: whether it is on, how
|
||||
// far the cursor has walked into the current cycle, how big that cycle is, and how many
|
||||
// full cycles have completed. Cheap; safe to poll.
|
||||
func (e *Engine) SweepStatus() (enabled bool, cursor, total int, cycles uint64) {
|
||||
e.sweepMu.Lock()
|
||||
defer e.sweepMu.Unlock()
|
||||
return e.sweep.cfg.Enabled, e.sweep.cursor, len(e.sweep.plan), e.sweep.cycles
|
||||
}
|
||||
|
||||
// sweepLoop is the ticker. It owns no state of its own — everything lives under
|
||||
// sweepMu — so a re-tune from ConfigureSweep takes effect on the next tick.
|
||||
func (e *Engine) sweepLoop(stop <-chan struct{}, done chan<- struct{}, interval time.Duration) {
|
||||
defer close(done)
|
||||
ticker := time.NewTicker(interval)
|
||||
defer ticker.Stop()
|
||||
for {
|
||||
select {
|
||||
case <-stop:
|
||||
return
|
||||
case <-ticker.C:
|
||||
e.sweepTick()
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// sweepTick runs one batch. It is deliberately conservative about when it does nothing:
|
||||
//
|
||||
// - a manual "Test all nodes" run is in flight => SKIP. The manual run is the human's
|
||||
// explicit request and covers everything anyway; the sweep defers to it rather than
|
||||
// competing for the uplink. This is why the sweep reuses testRunning as a check
|
||||
// instead of inventing a second scheme — and why it never CLAIMS that guard, which
|
||||
// would make a pressed button answer "already running" for a background task the
|
||||
// operator cannot see.
|
||||
// - the previous tick has not finished => SKIP, so slow probes can never stack.
|
||||
// - the engine is stopped => nothing resolves, the plan comes back empty.
|
||||
func (e *Engine) sweepTick() {
|
||||
if e.testRunning.Load() {
|
||||
return
|
||||
}
|
||||
|
||||
e.sweepMu.Lock()
|
||||
if e.sweep.busy || !e.sweep.cfg.Enabled {
|
||||
e.sweepMu.Unlock()
|
||||
return
|
||||
}
|
||||
cfg := e.sweep.cfg
|
||||
switch {
|
||||
case e.sweep.plan != nil && e.sweep.cursor >= len(e.sweep.plan):
|
||||
// Cycle complete: count it and start the next pass from the top. This is the
|
||||
// ONE place the cursor is deliberately rewound.
|
||||
e.sweep.cycles++
|
||||
e.sweep.plan = e.buildPlan(cfg)
|
||||
e.sweep.cursor = 0
|
||||
e.sweep.planHash = e.Hash()
|
||||
case e.sweep.plan == nil || e.sweep.planHash != e.Hash():
|
||||
// No plan yet, or an Apply swap changed the running config out from under it
|
||||
// (tags may have appeared or vanished). Rebuild — but keep our place if the
|
||||
// resulting work is the same, so a no-op apply cannot restart the walk.
|
||||
e.syncPlanLocked(cfg)
|
||||
}
|
||||
// Take the next slice of the cycle.
|
||||
start := e.sweep.cursor
|
||||
end := start + cfg.Batch
|
||||
if end > len(e.sweep.plan) {
|
||||
end = len(e.sweep.plan)
|
||||
}
|
||||
batch := append([]ProbeJob(nil), e.sweep.plan[start:end]...)
|
||||
e.sweep.cursor = end
|
||||
e.sweep.busy = true
|
||||
e.sweepMu.Unlock()
|
||||
|
||||
defer func() {
|
||||
e.sweepMu.Lock()
|
||||
e.sweep.busy = false
|
||||
e.sweepMu.Unlock()
|
||||
}()
|
||||
|
||||
// Freshness gate: skip anything somebody else is already keeping current. See the
|
||||
// file header — this is what stops the sweep from suppressing a failover group's own
|
||||
// 30s health check, and it is also what keeps the steady-state cost low.
|
||||
view := e.HealthView()
|
||||
sem := make(chan struct{}, cfg.Concurrency)
|
||||
var wg sync.WaitGroup
|
||||
for _, j := range batch {
|
||||
if !sweepShouldProbe(view, j, sweepFreshness) {
|
||||
continue
|
||||
}
|
||||
wg.Add(1)
|
||||
sem <- struct{}{}
|
||||
go func(j ProbeJob) {
|
||||
defer wg.Done()
|
||||
defer func() { <-sem }()
|
||||
e.probeJob(j)
|
||||
}(j)
|
||||
}
|
||||
wg.Wait()
|
||||
}
|
||||
|
||||
// syncPlanLocked rebuilds the plan from the current box and specs, and RESETS the
|
||||
// cursor only if the resulting work actually differs.
|
||||
//
|
||||
// "The same plan" is defined as measurement-for-measurement identity: same jobs, in the
|
||||
// same order, each dialling the same tag with the same probe URL and recording into the
|
||||
// same set of tags (plansEqual). That is the right comparison rather than, say, hashing
|
||||
// the config, because the cursor's only meaning is a position in THIS list — if the list
|
||||
// is identical the position is still valid, and if any measurement changed the position
|
||||
// is meaningless whatever the config hash says.
|
||||
//
|
||||
// The plan is derived data (box outbounds + group membership + specs), so rebuilding and
|
||||
// comparing is also self-correcting: there is no separate "is it stale" bookkeeping that
|
||||
// could itself go wrong. The cost is one walk of the outbound list per call — the same
|
||||
// walk a tick already does at cycle end.
|
||||
//
|
||||
// Caller holds sweepMu.
|
||||
func (e *Engine) syncPlanLocked(cfg SweepConfig) {
|
||||
next := e.buildPlan(cfg)
|
||||
e.sweep.planHash = e.Hash()
|
||||
if e.sweep.plan != nil && plansEqual(next, e.sweep.plan) {
|
||||
return // identical work: keep walking where we were
|
||||
}
|
||||
e.sweep.plan = next
|
||||
e.sweep.cursor = 0
|
||||
}
|
||||
|
||||
// plansEqual reports whether two plans describe exactly the same measurements.
|
||||
func plansEqual(a, b []ProbeJob) bool {
|
||||
if len(a) != len(b) {
|
||||
return false
|
||||
}
|
||||
for i := range a {
|
||||
if a[i].Dial != b[i].Dial || a[i].URL != b[i].URL || len(a[i].Store) != len(b[i].Store) {
|
||||
return false
|
||||
}
|
||||
for j := range a[i].Store {
|
||||
if a[i].Store[j] != b[i].Store[j] {
|
||||
return false
|
||||
}
|
||||
}
|
||||
}
|
||||
return true
|
||||
}
|
||||
|
||||
// sweepShouldProbe reports whether a job is worth running now: yes when ANY tag it
|
||||
// covers has no observation at all or one older than freshness.
|
||||
//
|
||||
// "any", not "all", on purpose — the job writes to every tag it covers, so a single
|
||||
// stale tag is reason enough, and a job is only skipped when every tag it would fill in
|
||||
// is already current.
|
||||
func sweepShouldProbe(view HealthView, j ProbeJob, freshness time.Duration) bool {
|
||||
limit := int64(freshness.Seconds())
|
||||
for _, tag := range j.Store {
|
||||
_, _, age := view.State(tag)
|
||||
if age < 0 || age >= limit {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// SweepTickForTest exposes the tick period the sweep is currently configured with, so a
|
||||
// test in another package can assert that a configured interval actually reached the
|
||||
// engine rather than collapsing onto the default. Not part of the operational API.
|
||||
func (e *Engine) SweepTickForTest() time.Duration {
|
||||
e.sweepMu.Lock()
|
||||
defer e.sweepMu.Unlock()
|
||||
return e.sweep.cfg.Interval
|
||||
}
|
||||
@@ -1,275 +0,0 @@
|
||||
package engine
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
)
|
||||
|
||||
// The freshness gate is what stops the scheduled sweep from suppressing a group's OWN
|
||||
// health checking: testNodes skips a member whose history is younger than the group's
|
||||
// interval, so a sweep that kept refreshing a failover member would keep that group
|
||||
// from running its 30s check. A tag anyone is actively probing must therefore be
|
||||
// invisible to the sweep.
|
||||
func TestSweepFreshnessGate(t *testing.T) {
|
||||
now := time.Now()
|
||||
hist := urltest.NewHistoryStorage()
|
||||
hist.StoreURLTestHistory("fresh", &adapter.URLTestHistory{Time: now.Add(-20 * time.Second), Delay: 30})
|
||||
hist.StoreURLTestHistory("stale", &adapter.URLTestHistory{Time: now.Add(-30 * time.Minute), Delay: 30})
|
||||
dead := map[string]time.Time{
|
||||
"fresh-dead": now.Add(-10 * time.Second),
|
||||
"stale-dead": now.Add(-30 * time.Minute),
|
||||
}
|
||||
view := newHealthView(hist, dead, now)
|
||||
|
||||
cases := []struct {
|
||||
name string
|
||||
job ProbeJob
|
||||
want bool
|
||||
}{
|
||||
{"a member a group is actively probing is left alone", ProbeJob{Store: []string{"fresh"}}, false},
|
||||
{"a stale measurement is refreshed", ProbeJob{Store: []string{"stale"}}, true},
|
||||
{"never observed at all is always probed", ProbeJob{Store: []string{"unknown"}}, true},
|
||||
{"a recent DEAD verdict is fresh data too", ProbeJob{Store: []string{"fresh-dead"}}, false},
|
||||
{"an old dead verdict is re-checked", ProbeJob{Store: []string{"stale-dead"}}, true},
|
||||
// "any", not "all": the job writes to every tag it covers, so one stale tag is
|
||||
// reason enough to run it.
|
||||
{"one stale tag among fresh ones still runs", ProbeJob{Store: []string{"fresh", "stale"}}, true},
|
||||
{"all fresh is skipped", ProbeJob{Store: []string{"fresh", "fresh-dead"}}, false},
|
||||
}
|
||||
for _, c := range cases {
|
||||
if got := sweepShouldProbe(view, c.job, sweepFreshness); got != c.want {
|
||||
t.Errorf("%s: sweepShouldProbe = %v, want %v", c.name, got, c.want)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The sweep must defer to a manual "Test all nodes" run rather than compete with it for
|
||||
// the uplink — and it must do so WITHOUT claiming the manual run's guard, which would
|
||||
// make a pressed button answer "already running" because of an invisible background
|
||||
// task.
|
||||
func TestSweepDefersToManualRun(t *testing.T) {
|
||||
e := New()
|
||||
e.ConfigureSweep(SweepConfig{Enabled: true, Specs: nil, FallbackURL: ""})
|
||||
defer e.StopSweep()
|
||||
|
||||
// A manual run is in flight.
|
||||
e.testRunning.Store(true)
|
||||
e.sweepTick()
|
||||
e.sweepMu.Lock()
|
||||
cursor := e.sweep.cursor
|
||||
planned := e.sweep.plan != nil
|
||||
e.sweepMu.Unlock()
|
||||
if cursor != 0 || planned {
|
||||
t.Fatalf("sweep ran during a manual test (cursor=%d, planned=%v)", cursor, planned)
|
||||
}
|
||||
|
||||
// And the manual run's guard is still free for the operator's next press: the sweep
|
||||
// never took it.
|
||||
e.testRunning.Store(false)
|
||||
if !e.TestAllNodes(nil, "") {
|
||||
t.Fatal("a manual run was refused; the sweep must never hold testRunning")
|
||||
}
|
||||
}
|
||||
|
||||
// ConfigureSweep is called on every apply: it must be idempotent, must be able to
|
||||
// re-tune in place, and must stop cleanly. StopSweep is safe when nothing runs.
|
||||
func TestSweepLifecycle(t *testing.T) {
|
||||
e := New()
|
||||
|
||||
if enabled, _, _, _ := e.SweepStatus(); enabled {
|
||||
t.Fatal("a fresh engine must not be sweeping")
|
||||
}
|
||||
e.StopSweep() // idempotent when never started
|
||||
|
||||
e.ConfigureSweep(SweepConfig{Enabled: true})
|
||||
if enabled, _, _, _ := e.SweepStatus(); !enabled {
|
||||
t.Fatal("sweep did not start")
|
||||
}
|
||||
// Re-configuring while running must not spawn a second loop.
|
||||
e.ConfigureSweep(SweepConfig{Enabled: true, FallbackURL: "https://other.example/204"})
|
||||
e.sweepMu.Lock()
|
||||
loops := e.sweep.stop != nil
|
||||
url := e.sweep.cfg.FallbackURL
|
||||
e.sweepMu.Unlock()
|
||||
if !loops || url != "https://other.example/204" {
|
||||
t.Fatalf("re-configure did not re-tune in place (running=%v, url=%q)", loops, url)
|
||||
}
|
||||
|
||||
e.ConfigureSweep(SweepConfig{Enabled: false})
|
||||
if enabled, _, _, _ := e.SweepStatus(); enabled {
|
||||
t.Fatal("sweep did not stop on Enabled=false")
|
||||
}
|
||||
e.StopSweep()
|
||||
e.StopSweep() // still idempotent
|
||||
}
|
||||
|
||||
// Defaults must be filled in for every unset knob, so a caller can set only Enabled.
|
||||
func TestSweepConfigDefaults(t *testing.T) {
|
||||
got := SweepConfig{Enabled: true}.withDefaults()
|
||||
if got.Interval != sweepInterval || got.Batch != sweepBatch || got.Concurrency != sweepConcurrency {
|
||||
t.Fatalf("withDefaults = %+v, want the package defaults", got)
|
||||
}
|
||||
custom := SweepConfig{Enabled: true, Interval: time.Minute, Batch: 3, Concurrency: 2}.withDefaults()
|
||||
if custom.Interval != time.Minute || custom.Batch != 3 || custom.Concurrency != 2 {
|
||||
t.Fatalf("withDefaults overwrote explicit values: %+v", custom)
|
||||
}
|
||||
}
|
||||
|
||||
// A tick against a stopped engine plans nothing and must not panic or spin.
|
||||
func TestSweepTickStoppedEngine(t *testing.T) {
|
||||
e := New()
|
||||
e.ConfigureSweep(SweepConfig{Enabled: true})
|
||||
defer e.StopSweep()
|
||||
e.sweepTick()
|
||||
if _, cursor, total, _ := e.SweepStatus(); cursor != 0 || total != 0 {
|
||||
t.Fatalf("stopped engine: cursor=%d total=%d, want 0/0", cursor, total)
|
||||
}
|
||||
}
|
||||
|
||||
// D1 REGRESSION (release blocker). ConfigureSweep is called from applyLocked on every
|
||||
// successful apply, and cron reconciles this router every MINUTE. The first version
|
||||
// rebuilt the plan and zeroed the cursor on each of those calls, so a ~6-minute cycle
|
||||
// was restarted every 60s and never once completed: observed on the stand walking
|
||||
// 96 -> 24 -> 72 with cycles stuck at 0, leaving everything past the cursor forever
|
||||
// "untested" — the exact condition the sweep exists to remove.
|
||||
//
|
||||
// This replays that: walk part-way, then reconfigure repeatedly with an unchanged
|
||||
// config, as cron does.
|
||||
func TestConfigureSweepPreservesProgressWhenPlanUnchanged(t *testing.T) {
|
||||
plan := []ProbeJob{
|
||||
{Dial: "n1", URL: "u", Store: []string{"n1"}},
|
||||
{Dial: "n2", URL: "u", Store: []string{"n2"}},
|
||||
{Dial: "n3", URL: "u", Store: []string{"n3"}},
|
||||
{Dial: "n4", URL: "u", Store: []string{"n4"}},
|
||||
}
|
||||
e := New()
|
||||
e.sweepMu.Lock()
|
||||
// A fresh slice every call, exactly as a real rebuild would produce.
|
||||
e.sweep.planFn = func(map[string]GroupProbeSpec, string) []ProbeJob {
|
||||
return append([]ProbeJob(nil), plan...)
|
||||
}
|
||||
e.sweepMu.Unlock()
|
||||
|
||||
cfg := SweepConfig{Enabled: true, FallbackURL: "u"}
|
||||
e.ConfigureSweep(cfg)
|
||||
defer e.StopSweep()
|
||||
|
||||
// Walk half the cycle.
|
||||
e.sweepMu.Lock()
|
||||
e.sweep.cursor = 2
|
||||
e.sweepMu.Unlock()
|
||||
|
||||
// Five no-op reconciles, one "minute" apart.
|
||||
for i := 0; i < 5; i++ {
|
||||
e.ConfigureSweep(cfg)
|
||||
if _, cursor, total, cycles := e.SweepStatus(); cursor != 2 || total != 4 || cycles != 0 {
|
||||
t.Fatalf("reconcile #%d reset the walk: cursor=%d total=%d cycles=%d, want 2/4/0",
|
||||
i+1, cursor, total, cycles)
|
||||
}
|
||||
}
|
||||
|
||||
// A GENUINE change must still restart the walk — the old cursor indexes a list
|
||||
// that no longer exists.
|
||||
plan = append(plan, ProbeJob{Dial: "n5", URL: "u", Store: []string{"n5"}})
|
||||
e.ConfigureSweep(cfg)
|
||||
if _, cursor, total, _ := e.SweepStatus(); cursor != 0 || total != 5 {
|
||||
t.Fatalf("a changed plan must restart: cursor=%d total=%d, want 0/5", cursor, total)
|
||||
}
|
||||
}
|
||||
|
||||
// The cursor must also survive the tick path's own rebuild trigger (an Apply swap
|
||||
// changed the engine hash) when the resulting work is identical.
|
||||
func TestSweepTickRebuildPreservesProgress(t *testing.T) {
|
||||
plan := []ProbeJob{
|
||||
{Dial: "n1", URL: "u", Store: []string{"n1"}},
|
||||
{Dial: "n2", URL: "u", Store: []string{"n2"}},
|
||||
{Dial: "n3", URL: "u", Store: []string{"n3"}},
|
||||
}
|
||||
e := New()
|
||||
e.sweepMu.Lock()
|
||||
e.sweep.planFn = func(map[string]GroupProbeSpec, string) []ProbeJob {
|
||||
return append([]ProbeJob(nil), plan...)
|
||||
}
|
||||
e.sweepMu.Unlock()
|
||||
e.ConfigureSweep(SweepConfig{Enabled: true, Batch: 1, Concurrency: 1})
|
||||
defer e.StopSweep()
|
||||
|
||||
e.sweepMu.Lock()
|
||||
e.sweep.cursor = 1
|
||||
e.sweep.planHash = "stale-hash" // force the tick's rebuild branch
|
||||
e.sweepMu.Unlock()
|
||||
|
||||
e.sweepTick()
|
||||
|
||||
// The tick rebuilds (hash mismatch), finds identical work, keeps the cursor, and
|
||||
// then consumes its batch of 1 — so 1 -> 2, never back to 0.
|
||||
if _, cursor, _, cycles := e.SweepStatus(); cursor != 2 || cycles != 0 {
|
||||
t.Fatalf("cursor=%d cycles=%d, want 2/0 (kept its place, then advanced one batch)", cursor, cycles)
|
||||
}
|
||||
}
|
||||
|
||||
// plansEqual is the definition of "the same work", so its edges are a contract: a
|
||||
// different tag, a different probe URL, or a different set of recorded tags all mean a
|
||||
// different plan — and a different plan legitimately restarts the walk.
|
||||
func TestPlansEqual(t *testing.T) {
|
||||
base := []ProbeJob{
|
||||
{Dial: "n1", URL: "u", Store: []string{"n1", "group-a-m0-n1"}},
|
||||
{Dial: "n2", URL: "u", Store: []string{"n2"}},
|
||||
}
|
||||
if !plansEqual(base, append([]ProbeJob(nil), base...)) {
|
||||
t.Fatal("a copy must compare equal")
|
||||
}
|
||||
cases := map[string][]ProbeJob{
|
||||
"shorter": base[:1],
|
||||
"different dial tag": {
|
||||
{Dial: "n9", URL: "u", Store: []string{"n1", "group-a-m0-n1"}},
|
||||
{Dial: "n2", URL: "u", Store: []string{"n2"}},
|
||||
},
|
||||
"different probe URL": {
|
||||
{Dial: "n1", URL: "other", Store: []string{"n1", "group-a-m0-n1"}},
|
||||
{Dial: "n2", URL: "u", Store: []string{"n2"}},
|
||||
},
|
||||
"different store set": {
|
||||
{Dial: "n1", URL: "u", Store: []string{"n1"}},
|
||||
{Dial: "n2", URL: "u", Store: []string{"n2"}},
|
||||
},
|
||||
"different store member": {
|
||||
{Dial: "n1", URL: "u", Store: []string{"n1", "group-b-m0-n1"}},
|
||||
{Dial: "n2", URL: "u", Store: []string{"n2"}},
|
||||
},
|
||||
"empty": {},
|
||||
"nil": nil,
|
||||
}
|
||||
for name, other := range cases {
|
||||
if plansEqual(base, other) {
|
||||
t.Errorf("%s: compared EQUAL to the base plan, want different", name)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The one place the cursor is legitimately rewound is the completion of a cycle, and
|
||||
// that must also be the one place `cycles` is bumped.
|
||||
func TestSweepCycleCompletionRewinds(t *testing.T) {
|
||||
e := New()
|
||||
e.ConfigureSweep(SweepConfig{Enabled: true})
|
||||
defer e.StopSweep()
|
||||
|
||||
e.sweepMu.Lock()
|
||||
e.sweep.plan = []ProbeJob{{Dial: "n1", Store: []string{"n1"}}}
|
||||
e.sweep.cursor = 1 // walked to the end
|
||||
e.sweep.planHash = e.Hash()
|
||||
e.sweepMu.Unlock()
|
||||
|
||||
e.sweepTick()
|
||||
|
||||
_, cursor, _, cycles := e.SweepStatus()
|
||||
if cycles != 1 {
|
||||
t.Errorf("cycles = %d, want 1 after finishing a pass", cycles)
|
||||
}
|
||||
if cursor != 0 {
|
||||
t.Errorf("cursor = %d, want 0 at the start of the next pass", cursor)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,206 @@
|
||||
//go:build linux
|
||||
|
||||
// Integration tests for the chain health contract (plan §6-S6, scenario 2).
|
||||
//
|
||||
// Scenario 2 — a dead chain hop is fail-closed:
|
||||
// - the generated chain config (generate.Build) is accepted by engine.Apply
|
||||
// (box.New validate + Start), so the per-hop Detour-linked copies and the
|
||||
// rule's route-to-exit are legal sing-box;
|
||||
// - the rule routes to the chain's EXIT tag (chain-<n>-hN), never a silent direct
|
||||
// fallback — a dead path blocks rather than leaking over the plain WAN;
|
||||
// - the observatory, configured on the same engine, probes that exit end-to-end
|
||||
// and records a dead verdict on the shared board for chain-<n>-hN; and a dial
|
||||
// through the exit returns an error (no quiet success).
|
||||
//
|
||||
// Linux-only (box.New validates the loop-guard RoutingMark only on linux); the
|
||||
// OpenWrt VM / linux CI runs these, Windows/macOS dev gets GOOS=linux go vet/build.
|
||||
// The observatory reachability plan (scenario 3) is portable and lives in
|
||||
// observatory_reach_integration_test.go.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"context"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// chainHealthModel is a 2-hop chain referenced by a rule. node a (L1) and node b
|
||||
// (the EXIT, L2) dial TEST-NET addresses, so a real probe through the exit fails —
|
||||
// exactly the "dead chain hop" the contract exercises.
|
||||
func chainHealthModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.ProbeURL = "https://www.gstatic.com/generate_204"
|
||||
g.ProbeInterval = "60s"
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12402, TCP: true, UDP: true},
|
||||
},
|
||||
Nodes: []model.Node{
|
||||
{Name: "a", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#a"},
|
||||
{Name: "b", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#b"},
|
||||
},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "two", Hops: []string{"node:a", "node:b"}},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-chain", Enabled: true, Order: 10, DstPort: "443", Target: "chain:two"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// chainExitTag is the EXIT wrapper tag generate emits for a 2-hop chain "two":
|
||||
// chain-<name>-h<N> with N = hop count.
|
||||
const chainExitTag = "chain-two-h2"
|
||||
|
||||
// chainInterval is the global probe interval chainHealthModel sets, reused when the
|
||||
// observatory is configured directly (engine.Apply does not configure it — only
|
||||
// apply.Applier does, which these tests bypass to drive the engine themselves).
|
||||
const chainInterval = 60 * time.Second
|
||||
|
||||
// TestChainConfigAppliesAndRoutesToExit is the box.New+Start leg of scenario 2:
|
||||
// the generated 2-hop chain is legal sing-box the engine starts, and the rule
|
||||
// routes to the chain's EXIT wrapper — never a silent direct fallback. A dead path
|
||||
// must block (fail-closed), not leak over the default WAN.
|
||||
func TestChainConfigAppliesAndRoutesToExit(t *testing.T) {
|
||||
opts, warns, changed := applyAndClose(t, chainHealthModel())
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true (warnings: %v)", warns)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("unexpected warnings: %v", warns)
|
||||
}
|
||||
// The rule routes into the chain EXIT (h2 = node b). Anything else — direct in
|
||||
// particular — would be a silent fail-OPEN leak on a dead chain.
|
||||
got, ok := generalRouteOutbound(opts.Route)
|
||||
if !ok || got != chainExitTag {
|
||||
t.Fatalf("rule routes to %q (ok=%v), want exit %q (fail-closed, no direct fallback)", got, ok, chainExitTag)
|
||||
}
|
||||
// The exit wrapper survived box.New and detours through h1 (node a).
|
||||
ob := obByTag(opts, chainExitTag)
|
||||
if ob == nil {
|
||||
t.Fatalf("chain exit %q missing after Apply; outbounds=%v", chainExitTag, outboundTags(opts))
|
||||
}
|
||||
if d := outboundDetour(t, ob); d != "chain-two-h1" {
|
||||
t.Fatalf("%s.detour = %q, want chain-two-h1", chainExitTag, d)
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDeadExitMarkedByObservatory is the board leg of scenario 2: with the
|
||||
// observatory configured on the applied engine, the chain's exit tag is probed
|
||||
// end-to-end and — because both hops dial unreachable TEST-NET addresses — the
|
||||
// probe fails and a DEAD verdict lands on the shared board for chain-two-h2. The
|
||||
// same board is what selection and the panel read, so the death is visible
|
||||
// immediately, not on a private overlay.
|
||||
func TestChainDeadExitMarkedByObservatory(t *testing.T) {
|
||||
m := chainHealthModel()
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("unexpected warnings: %v", warns)
|
||||
}
|
||||
eng := engine.New()
|
||||
t.Cleanup(func() {
|
||||
eng.StopObservatory()
|
||||
_ = eng.Close()
|
||||
})
|
||||
if _, err := eng.Apply(opts); err != nil {
|
||||
t.Fatalf("engine.Apply (box.New + start): %v", err)
|
||||
}
|
||||
|
||||
eng.ConfigureObservatory(engine.ObservatoryConfig{
|
||||
Enabled: true,
|
||||
Options: opts,
|
||||
ProbeURL: m.Globals.ProbeURL,
|
||||
ProbeInterval: chainInterval,
|
||||
})
|
||||
if enabled, planned, _ := eng.ObservatoryStatus(); !enabled {
|
||||
t.Fatal("ConfigureObservatory did not enable the observatory")
|
||||
} else if planned == 0 {
|
||||
t.Fatalf("observatory plan is empty — the chain exit %q must be a probe target", chainExitTag)
|
||||
}
|
||||
|
||||
// The observatory's first pass runs immediately (cold start). Poll the board for
|
||||
// the exit's dead verdict — the probe is bounded by probeTimeout (5s) and the
|
||||
// TEST-NET dial fails fast, so this resolves well inside the deadline.
|
||||
hist := eng.URLTestHistory()
|
||||
if hist == nil {
|
||||
t.Fatal("engine has no shared URLTestHistory board")
|
||||
}
|
||||
ttl := verdictTTL(chainInterval)
|
||||
deadline := time.Now().Add(20 * time.Second)
|
||||
for {
|
||||
if v := hist.Verdict(chainExitTag, ttl); v == urltest.VerdictDead {
|
||||
break
|
||||
}
|
||||
if time.Now().After(deadline) {
|
||||
v := hist.Verdict(chainExitTag, ttl)
|
||||
t.Fatalf("exit %q verdict = %v after 20s, want dead — the observatory did not mark the dead chain", chainExitTag, v)
|
||||
}
|
||||
time.Sleep(200 * time.Millisecond)
|
||||
}
|
||||
// The dead verdict is information, not an invitation: the entry is kept, never
|
||||
// deleted on failure (a dead node stays distinguishable from a never-measured one).
|
||||
if h := hist.LoadURLTestHistory(chainExitTag); h == nil {
|
||||
t.Fatal("exit history entry was deleted; a dead verdict must keep it")
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainDeadExitDialFailsClosed is the dial leg of scenario 2: a dial THROUGH
|
||||
// the chain exit returns an error (the path is dead), and it is never a quiet
|
||||
// success. The exit outbound is the real one box.New built, dialled with the SAME
|
||||
// urltest.URLTest primitive the observatory and the exit test use — so a success
|
||||
// here would mean a leak, and an error is the honest fail-closed outcome.
|
||||
func TestChainDeadExitDialFailsClosed(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(chainHealthModel())
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("unexpected warnings: %v", warns)
|
||||
}
|
||||
eng := engine.New()
|
||||
t.Cleanup(func() { _ = eng.Close() })
|
||||
if _, err := eng.Apply(opts); err != nil {
|
||||
t.Fatalf("engine.Apply: %v", err)
|
||||
}
|
||||
inst := eng.Instance()
|
||||
if inst == nil {
|
||||
t.Fatal("engine has no running box instance")
|
||||
}
|
||||
om := inst.Outbound()
|
||||
if om == nil {
|
||||
t.Fatal("box has no outbound manager")
|
||||
}
|
||||
exitOb, ok := om.Outbound(chainExitTag)
|
||||
if !ok {
|
||||
t.Fatalf("exit outbound %q not in the running box", chainExitTag)
|
||||
}
|
||||
// A direct outbound is the exit-address trap the grouptest guard exists for. The
|
||||
// chain exit is a real proxy hop, so it must NOT resolve to direct — that would
|
||||
// be a silent fail-OPEN leak over the default WAN.
|
||||
if exitOb.Type() == C.TypeDirect {
|
||||
t.Fatalf("exit %q resolved to a DIRECT outbound — the chain leaked to the default WAN", chainExitTag)
|
||||
}
|
||||
// Dial the exit through the same probe primitive the observatory uses. The path
|
||||
// is dead (both hops dial unreachable TEST-NET addresses), so this MUST error —
|
||||
// a success here would mean a quiet fallback rather than fail-closed.
|
||||
probeCtx, probeCancel := context.WithTimeout(context.Background(), 3*time.Second)
|
||||
defer probeCancel()
|
||||
if _, err := urltest.URLTest(probeCtx, "", exitOb); err == nil {
|
||||
t.Fatalf("dial through dead chain exit %q succeeded — fail-closed is broken (silent fallback)", chainExitTag)
|
||||
}
|
||||
}
|
||||
@@ -149,9 +149,10 @@ func TestChainSingleHopResolvesToHop(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainUndefinedWarnsSkipped: a rule targeting an undefined chain warns and
|
||||
// is skipped — no wrapper outbounds, no general route rule, no panic.
|
||||
func TestChainUndefinedWarnsSkipped(t *testing.T) {
|
||||
// TestChainUndefinedWarnsBlocks: a rule targeting an undefined chain warns and is
|
||||
// emitted routing to block (fail-closed Rule.Kill) — no wrapper outbounds, no
|
||||
// fall-through to the default, no panic.
|
||||
func TestChainUndefinedWarnsBlocks(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Rules: []model.Rule{
|
||||
@@ -171,13 +172,14 @@ func TestChainUndefinedWarnsSkipped(t *testing.T) {
|
||||
if !warned {
|
||||
t.Fatalf("expected an undefined-chain warning, got %v", warns)
|
||||
}
|
||||
if _, ok := generalRouteOutbound(opts.Route); ok {
|
||||
t.Fatalf("undefined chain must not emit a route rule; rules=%+v", opts.Route.Rules)
|
||||
if got, ok := generalRouteOutbound(opts.Route); !ok || got != tagBlock {
|
||||
t.Fatalf("undefined chain must emit the rule routing to block, got %q (ok=%v); rules=%+v", got, ok, opts.Route.Rules)
|
||||
}
|
||||
}
|
||||
|
||||
// TestChainEmptyHopsWarnsSkipped: a defined chain with no hops warns and skips.
|
||||
func TestChainEmptyHopsWarnsSkipped(t *testing.T) {
|
||||
// TestChainEmptyHopsWarnsBlocks: a defined chain with no hops warns; the rule
|
||||
// targeting it is emitted routing to block (fail-closed Rule.Kill).
|
||||
func TestChainEmptyHopsWarnsBlocks(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Chains: []model.Chain{{Name: "empty", Hops: nil}},
|
||||
@@ -198,8 +200,8 @@ func TestChainEmptyHopsWarnsSkipped(t *testing.T) {
|
||||
if !warned {
|
||||
t.Fatalf("expected a no-hops warning, got %v", warns)
|
||||
}
|
||||
if _, ok := generalRouteOutbound(opts.Route); ok {
|
||||
t.Fatalf("empty chain must not emit a route rule")
|
||||
if got, ok := generalRouteOutbound(opts.Route); !ok || got != tagBlock {
|
||||
t.Fatalf("empty chain must emit the rule routing to block, got %q (ok=%v)", got, ok)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -235,10 +235,10 @@ func (b *builder) endpointResolver() string {
|
||||
}
|
||||
b.endpointResolverComputed = true
|
||||
|
||||
// Profile overrides live on the builder only after applyProfilesAndPresets; it is
|
||||
// Profile overrides live on the builder only after applyProfiles; it is
|
||||
// idempotent, so calling it here makes endpointResolver safe to invoke from either
|
||||
// buildRoute or buildDNS regardless of order.
|
||||
b.applyProfilesAndPresets()
|
||||
b.applyProfiles()
|
||||
|
||||
name := b.profileEndpointResolver
|
||||
src := "active profile"
|
||||
|
||||
@@ -515,7 +515,6 @@ func TestOfflineBootStartsEngine(t *testing.T) {
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: true}},
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "geo", Source: "geosite", Categories: []string{"youtube"}},
|
||||
{Name: "keep", Source: "inline", Entries: []string{"local.example"}},
|
||||
|
||||
@@ -212,24 +212,10 @@ func TestFailoverBuildsPriorityPool(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestFailoverIntervalRespectsOperator: an interval the operator set — on the
|
||||
// group or in globals — wins over our substituted default, and is not warned
|
||||
// about (they already know what they chose).
|
||||
// TestFailoverIntervalRespectsOperator: an interval the operator set in globals
|
||||
// wins over our substituted default, and is not warned about (they already know
|
||||
// what they chose).
|
||||
func TestFailoverIntervalRespectsOperator(t *testing.T) {
|
||||
t.Run("group", func(t *testing.T) {
|
||||
m := twoNodeGroupModel("failover")
|
||||
m.Groups[0].ProbeInterval = "10s"
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if d := time.Duration(urltestOf(t, obByTag(opts, "g")).Interval); d != 10*time.Second {
|
||||
t.Fatalf("Interval = %s, want 10s", d)
|
||||
}
|
||||
if got := warnMatching(warns, "health-check interval"); len(got) != 0 {
|
||||
t.Fatalf("must not warn about a substituted interval: %v", got)
|
||||
}
|
||||
})
|
||||
t.Run("globals", func(t *testing.T) {
|
||||
m := twoNodeGroupModel("failover")
|
||||
m.Globals.ProbeInterval = "45s"
|
||||
|
||||
@@ -40,8 +40,8 @@
|
||||
// - DstRuleset matchers: referencing an undefined rule_set aborts box.New, and
|
||||
// materialising rulesets (inline/file/url) is Phase-2b — DstRuleset is
|
||||
// skipped with a warning.
|
||||
// - Schedules / presets / WAN-profiles: evaluated by the control-plane at
|
||||
// gen/reconcile time, not here (the engine has no time match).
|
||||
// - Schedules / WAN-profiles: evaluated by the control-plane at gen/reconcile
|
||||
// time, not here (the engine has no time match).
|
||||
// - Per-client DNS scoping beyond MatchSrc/MatchDomain is minimal.
|
||||
//
|
||||
// # Not deferred — NOT SUPPORTED: MAC / iface: / zone: rule sources
|
||||
@@ -181,16 +181,12 @@ type builder struct {
|
||||
dohIPCIDRs []string
|
||||
dohComputed bool
|
||||
|
||||
// Profiles/presets (evaluated in generate the same way schedules are, via
|
||||
// b.now). applyProfilesAndPresets fills these once, guarded by
|
||||
// effectiveComputed; buildRoute consumes them. For a model with no
|
||||
// profiles/presets effectiveRules is an exact copy of b.m.Rules (identical
|
||||
// generate output) and the default overrides are empty.
|
||||
effectiveRules []model.Rule // preset packs prepended + active-profile enable/disable applied
|
||||
profileDefaultTarget string // active profile's DefaultTarget override ("" => none)
|
||||
profileDefaultEgress string // active profile's DefaultEgress override ("" => none)
|
||||
// Profiles. applyProfiles fills these once, guarded by effectiveComputed;
|
||||
// buildRoute consumes them. For a model with no profiles effectiveRules is an
|
||||
// exact copy of b.m.Rules (identical generate output).
|
||||
effectiveRules []model.Rule // active-profile enable/disable applied
|
||||
profileEndpointResolver string // active profile's EndpointResolver override ("" => none)
|
||||
effectiveComputed bool // applyProfilesAndPresets has run
|
||||
effectiveComputed bool // applyProfiles has run
|
||||
|
||||
// Endpoint resolver (route.default_domain_resolver): the bootstrap-direct DNS
|
||||
// server used ONLY to resolve proxy outbounds' server DOMAINS. Computed once by
|
||||
|
||||
@@ -163,7 +163,7 @@ func TestGroupURLTest(t *testing.T) {
|
||||
{Name: "n2", Enabled: true, URI: "trojan://pw@203.0.113.6:443?sni=example.com#n2"},
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "auto", Source: "manual", Nodes: []string{"n1", "n2"}, Strategy: "leastping", ProbeInterval: "60s"},
|
||||
{Name: "auto", Source: "manual", Nodes: []string{"n1", "n2"}, Strategy: "leastping"},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "default", Enabled: true, Order: 100, Target: "group:auto"}, // catch-all -> group
|
||||
|
||||
+23
-32
@@ -221,7 +221,7 @@ const (
|
||||
// --- failover ---------------------------------------------------------------
|
||||
//
|
||||
// failoverProbeInterval is the health-check period a failover group gets when
|
||||
// neither the group nor globals sets one.
|
||||
// globals does not set one.
|
||||
//
|
||||
// The engine default (C.DefaultURLTestInterval) is 3 MINUTES, which is fine for
|
||||
// least_test — there the probe only re-ranks nodes that are all working — but
|
||||
@@ -290,13 +290,10 @@ func failoverBalancer() *option.URLTestBalancerOptions {
|
||||
}
|
||||
}
|
||||
|
||||
// failoverInterval is groupInterval with the failover default substituted for the
|
||||
// engine's 3-minute one. Pure (no diagnostics) because groupOutbound must stay
|
||||
// warning-free for chain copies; warnGroupStrategy reports the substitution.
|
||||
func failoverInterval(gl model.Globals, g model.Group) badoptionDuration {
|
||||
if d, ok := parseDuration(g.ProbeInterval); ok {
|
||||
return d
|
||||
}
|
||||
// failoverInterval is globalProbeInterval with the failover default substituted
|
||||
// for the engine's 3-minute one. Pure (no diagnostics) because groupOutbound must
|
||||
// stay warning-free for chain copies; warnGroupStrategy reports the substitution.
|
||||
func failoverInterval(gl model.Globals) badoptionDuration {
|
||||
if d, ok := parseDuration(gl.ProbeInterval); ok {
|
||||
return d
|
||||
}
|
||||
@@ -366,12 +363,10 @@ func (b *builder) warnGroupStrategy(g model.Group, members []string) {
|
||||
// prompt. Both are about the CONSEQUENCE (how long traffic stays broken), not
|
||||
// about configuration style.
|
||||
func (b *builder) warnFailoverGroup(g model.Group, members []string) {
|
||||
// 1. Substituted probe interval. Only when the operator set neither the group
|
||||
// nor the global interval — if they chose one, it is theirs and needs no notice.
|
||||
_, groupSet := parseDuration(g.ProbeInterval)
|
||||
_, globalSet := parseDuration(b.m.Globals.ProbeInterval)
|
||||
if !groupSet && !globalSet {
|
||||
b.warnf("group %q: strategy=failover sets the health-check interval to %s (the engine default of %s would leave a dead node accepting and failing connections for up to that long, because failover switches on the probe tick, not on a failed connection). Set probe_interval on the group or in globals to override",
|
||||
// 1. Substituted probe interval. Only when the operator did not set the global
|
||||
// interval — if they chose one, it is theirs and needs no notice.
|
||||
if _, globalSet := parseDuration(b.m.Globals.ProbeInterval); !globalSet {
|
||||
b.warnf("group %q: strategy=failover sets the health-check interval to %s (the engine default of %s would leave a dead node accepting and failing connections for up to that long, because failover switches on the probe tick, not on a failed connection). Set probe_interval in globals to override",
|
||||
g.Name, failoverProbeInterval, C.DefaultURLTestInterval)
|
||||
}
|
||||
|
||||
@@ -408,8 +403,8 @@ func (b *builder) groupOutbound(g model.Group, tag string, members []string) opt
|
||||
Tag: tag,
|
||||
Options: &option.URLTestOutboundOptions{
|
||||
Outbounds: members,
|
||||
URL: groupProbeURL(b.m.Globals, g),
|
||||
Interval: groupInterval(b.m.Globals, g),
|
||||
URL: globalProbeURL(b.m.Globals),
|
||||
Interval: globalProbeInterval(b.m.Globals),
|
||||
Mode: C.URLTestModeRoundRobin,
|
||||
},
|
||||
}
|
||||
@@ -438,8 +433,8 @@ func (b *builder) groupOutbound(g model.Group, tag string, members []string) opt
|
||||
Tag: tag,
|
||||
Options: &option.URLTestOutboundOptions{
|
||||
Outbounds: members,
|
||||
URL: groupProbeURL(b.m.Globals, g),
|
||||
Interval: groupInterval(b.m.Globals, g),
|
||||
URL: globalProbeURL(b.m.Globals),
|
||||
Interval: globalProbeInterval(b.m.Globals),
|
||||
Mode: C.URLTestModeRandom,
|
||||
Balancer: &option.URLTestBalancerOptions{
|
||||
Pool: len(members),
|
||||
@@ -458,8 +453,8 @@ func (b *builder) groupOutbound(g model.Group, tag string, members []string) opt
|
||||
Tag: tag,
|
||||
Options: &option.URLTestOutboundOptions{
|
||||
Outbounds: members,
|
||||
URL: groupProbeURL(b.m.Globals, g),
|
||||
Interval: failoverInterval(b.m.Globals, g),
|
||||
URL: globalProbeURL(b.m.Globals),
|
||||
Interval: failoverInterval(b.m.Globals),
|
||||
Mode: C.URLTestModeRoundRobin,
|
||||
Balancer: failoverBalancer(),
|
||||
},
|
||||
@@ -470,8 +465,8 @@ func (b *builder) groupOutbound(g model.Group, tag string, members []string) opt
|
||||
Tag: tag,
|
||||
Options: &option.URLTestOutboundOptions{
|
||||
Outbounds: members,
|
||||
URL: groupProbeURL(b.m.Globals, g),
|
||||
Interval: groupInterval(b.m.Globals, g),
|
||||
URL: globalProbeURL(b.m.Globals),
|
||||
Interval: globalProbeInterval(b.m.Globals),
|
||||
Mode: C.URLTestModeLeastTest,
|
||||
},
|
||||
}
|
||||
@@ -622,22 +617,18 @@ func matchAnyRegexp(rs []*regexp.Regexp, s string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// groupProbeURL / groupInterval pick the group-level probe settings, falling
|
||||
// back to globals, then to sane defaults.
|
||||
func groupProbeURL(gl model.Globals, g model.Group) string {
|
||||
if g.ProbeURL != "" {
|
||||
return g.ProbeURL
|
||||
}
|
||||
// globalProbeURL / globalProbeInterval pick the probe settings from globals,
|
||||
// falling back to sane defaults. There are deliberately no per-group overrides:
|
||||
// one node in two groups with different URLs would make the stored delays
|
||||
// incomparable, so every measurement uses the one global instrument.
|
||||
func globalProbeURL(gl model.Globals) string {
|
||||
if gl.ProbeURL != "" {
|
||||
return gl.ProbeURL
|
||||
}
|
||||
return "https://www.gstatic.com/generate_204"
|
||||
}
|
||||
|
||||
func groupInterval(gl model.Globals, g model.Group) badoptionDuration {
|
||||
if d, ok := parseDuration(g.ProbeInterval); ok {
|
||||
return d
|
||||
}
|
||||
func globalProbeInterval(gl model.Globals) badoptionDuration {
|
||||
if d, ok := parseDuration(gl.ProbeInterval); ok {
|
||||
return d
|
||||
}
|
||||
|
||||
@@ -317,8 +317,9 @@ func TestGroupEgressDPIPresetWarnsIgnored(t *testing.T) {
|
||||
|
||||
// TestGroupEgressDeadGroupStaysFailClosed is the D17 invariant: when every member of
|
||||
// an egress-bound group is unusable the group is skipped, the rule targeting it is
|
||||
// skipped, and the kill-switch Final BLOCKS — the traffic must never fall through to
|
||||
// direct just because the group could not be built.
|
||||
// emitted routing to block (fail-closed Rule.Kill — the traffic must never fall
|
||||
// through to direct just because the group could not be built), and the
|
||||
// kill-switch Final BLOCKS.
|
||||
func TestGroupEgressDeadGroupStaysFailClosed(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(), // KillSwitch defaults to "closed"
|
||||
@@ -345,9 +346,10 @@ func TestGroupEgressDeadGroupStaysFailClosed(t *testing.T) {
|
||||
if !warnsHaveSub(warns, `group "tunneled"`) {
|
||||
t.Fatalf("expected a warning about the empty group, got %v", warns)
|
||||
}
|
||||
// No route rule may send traffic anywhere, and the Final must be block.
|
||||
if got, ok := generalRouteOutbound(opts.Route); ok {
|
||||
t.Fatalf("a rule survived pointing at %q; it must have been skipped", got)
|
||||
// The rule stays emitted, routing to block — an unresolved target never
|
||||
// drops the rule to fall through to the default — and the Final is block.
|
||||
if got, ok := generalRouteOutbound(opts.Route); !ok || got != tagBlock {
|
||||
t.Fatalf("rule to the dead group routes to %q (ok=%v), want block", got, ok)
|
||||
}
|
||||
if opts.Route == nil || opts.Route.Final != tagBlock {
|
||||
t.Fatalf("route Final = %q, want %q (fail-closed, D17)", finalOf(opts), tagBlock)
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
//go:build linux
|
||||
|
||||
// The box.New+Start leg of scenario 1 (plan §6-S6): the generated least_test
|
||||
// urltest group is legal sing-box the engine accepts and starts, with both members
|
||||
// wired in and the rule routing into the group. The retry/selection BEHAVIOUR the
|
||||
// health board enables is pinned by the portable health_retry_integration_test.go
|
||||
// (generate.Build + group.NewURLTest + stub members — no box.New needed, so it runs
|
||||
// on every host); this file adds the engine.Apply validation that only linux can do
|
||||
// (box.New validates the loop-guard RoutingMark only on linux).
|
||||
//
|
||||
// Runs on the OpenWrt VM / linux CI; Windows/macOS dev gets GOOS=linux go vet/build.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
)
|
||||
|
||||
// TestHealthGroupConfigApplies is the box.New+Start leg of scenario 1: the
|
||||
// generated least_test urltest group is legal sing-box the engine starts, with
|
||||
// both members wired in config order and the rule routing into the group (never a
|
||||
// silent direct fallback). A passing Apply means the tag schema, interval and
|
||||
// member wiring the retry tests dial against are real engine-validated sing-box,
|
||||
// not merely well-formed Go structs.
|
||||
func TestHealthGroupConfigApplies(t *testing.T) {
|
||||
opts, warns, changed := applyAndClose(t, urltestHealthModel())
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true (warnings: %v)", warns)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("unexpected warnings: %v", warns)
|
||||
}
|
||||
ut := urlTestOptsOf(t, opts, "ha")
|
||||
if ut.Mode != C.URLTestModeLeastTest {
|
||||
t.Fatalf("Mode = %q, want %q (leastping)", ut.Mode, C.URLTestModeLeastTest)
|
||||
}
|
||||
if len(ut.Outbounds) != 2 || ut.Outbounds[0] != "primary" || ut.Outbounds[1] != "backup" {
|
||||
t.Fatalf("members = %v, want [primary backup] in config order", ut.Outbounds)
|
||||
}
|
||||
if got := time.Duration(ut.Interval); got != 60*time.Second {
|
||||
t.Fatalf("Interval = %s, want 60s (from globals.probe_interval)", got)
|
||||
}
|
||||
// The rule routes into the group, not a silent direct fallback.
|
||||
if got, ok := generalRouteOutbound(opts.Route); !ok || got != "ha" {
|
||||
t.Fatalf("rule routes to %q (ok=%v), want ha", got, ok)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,317 @@
|
||||
// Integration tests for the live-proxy-pool retry contract (plan §6-S6, scenario 1).
|
||||
//
|
||||
// A generated urltest-group config (generate.Build) drives an in-process URLTest
|
||||
// over stub members — the real SS nodes in the model dial TEST-NET addresses
|
||||
// (unreachable by design), so a genuinely "alive" member has to be a stub. The
|
||||
// stubs share the SAME health board a running box would, so the retry and the
|
||||
// mark-fail -> selection read-through are observed as behaviour (dial result + board
|
||||
// verdict), not internals.
|
||||
//
|
||||
// Portable: generate.Build and group.NewURLTest run on every platform, so the retry
|
||||
// behaviour is verified on the Windows dev host too; the box.New+Start leg of
|
||||
// scenario 1 (the config is legal sing-box that starts) lives in the linux-only
|
||||
// health_lx_linux_test.go alongside the chain scenarios.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"context"
|
||||
"errors"
|
||||
"net"
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
"github.com/sagernet/sing-box/adapter"
|
||||
"github.com/sagernet/sing-box/common/urltest"
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/log"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
"github.com/sagernet/sing-box/protocol/group"
|
||||
"github.com/sagernet/sing/service"
|
||||
M "github.com/sagernet/sing/common/metadata"
|
||||
N "github.com/sagernet/sing/common/network"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// --- in-process outbound stubs --------------------------------------------------
|
||||
//
|
||||
// The group's own tests (protocol/group/urltest_health_lx_test.go) use a healthNode
|
||||
// fake outbound; those are package-internal. These are the generate-package
|
||||
// equivalents: a dialable fake that succeeds or fails per the fail flag and records
|
||||
// the dial order, so the retry/selection contracts can be observed as behaviour
|
||||
// (dial result + board verdict) rather than internals.
|
||||
|
||||
// stubNode is a dialable fake outbound for one urltest member.
|
||||
type stubNode struct {
|
||||
adapter.Outbound
|
||||
tag string
|
||||
fail bool
|
||||
dialed *[]string // shared record of which members were actually dialled
|
||||
}
|
||||
|
||||
func (n *stubNode) Type() string { return C.TypeVLESS }
|
||||
func (n *stubNode) Tag() string { return n.tag }
|
||||
func (n *stubNode) Network() []string { return []string{N.NetworkTCP, N.NetworkUDP} }
|
||||
func (n *stubNode) Dependencies() []string { return nil }
|
||||
|
||||
func (n *stubNode) DialContext(ctx context.Context, network string, destination M.Socksaddr) (net.Conn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
}
|
||||
if n.fail {
|
||||
return nil, errors.New("dial refused (stub)")
|
||||
}
|
||||
left, right := net.Pipe()
|
||||
_ = right.Close()
|
||||
return left, nil
|
||||
}
|
||||
|
||||
func (n *stubNode) ListenPacket(ctx context.Context, destination M.Socksaddr) (net.PacketConn, error) {
|
||||
if n.dialed != nil {
|
||||
*n.dialed = append(*n.dialed, n.tag)
|
||||
}
|
||||
if n.fail {
|
||||
return nil, errors.New("listen refused (stub)")
|
||||
}
|
||||
return net.ListenPacket("udp", "127.0.0.1:0")
|
||||
}
|
||||
|
||||
// stubOutboundManager resolves member tags to stub nodes (only Outbound is used by
|
||||
// the urltest group's Start/selectBalanced path). Embedding adapter.OutboundManager
|
||||
// gives nil implementations for the unused manager methods.
|
||||
type stubOutboundManager struct {
|
||||
adapter.OutboundManager
|
||||
nodes map[string]adapter.Outbound
|
||||
}
|
||||
|
||||
func (m *stubOutboundManager) Outbound(tag string) (adapter.Outbound, bool) {
|
||||
n, ok := m.nodes[tag]
|
||||
return n, ok
|
||||
}
|
||||
|
||||
// newHealthGroup builds a started *group.URLTest from the URLTestOutboundOptions the
|
||||
// generator emitted for tag, dialling the stub members through mgr and sharing hist
|
||||
// as the health board — exactly the wiring box.New would build, minus the real
|
||||
// network and the group's own probe ticker (PostStart is deliberately NOT called:
|
||||
// no checker runs, so the board stays under the test's control). Touch() is a no-op
|
||||
// until PostStart, so DialContext's retry path runs against the board alone.
|
||||
func newHealthGroup(t *testing.T, hist *urltest.HistoryStorage, mgr adapter.OutboundManager, tag string, ut option.URLTestOutboundOptions) *group.URLTest {
|
||||
t.Helper()
|
||||
ctx := service.ContextWithPtr[urltest.HistoryStorage](context.Background(), hist)
|
||||
ctx = service.ContextWith[adapter.OutboundManager](ctx, mgr)
|
||||
ob, err := group.NewURLTest(ctx, nil, log.NewNOPFactory().Logger(), tag, ut)
|
||||
if err != nil {
|
||||
t.Fatalf("NewURLTest(%s): %v", tag, err)
|
||||
}
|
||||
g, ok := ob.(*group.URLTest)
|
||||
if !ok {
|
||||
t.Fatalf("NewURLTest returned %T, want *group.URLTest", ob)
|
||||
}
|
||||
// Start resolves the member stubs through mgr and builds the URLTestGroup. It
|
||||
// does NOT start the probe loop (that is PostStart); leaving it un-started keeps
|
||||
// every verdict on the board under the test's control.
|
||||
if err := g.Start(); err != nil {
|
||||
t.Fatalf("URLTest(%s).Start: %v", tag, err)
|
||||
}
|
||||
return g
|
||||
}
|
||||
|
||||
// storeAlive records a fresh success 1s in the past: inside the TTL, but strictly
|
||||
// older than any failure the test stamps afterwards (a coarse monotonic clock can
|
||||
// otherwise tie LastOK == LastFail within one tick).
|
||||
func storeAlive(hist *urltest.HistoryStorage, tag string, delay uint16) {
|
||||
hist.StoreURLTestHistory(tag, &adapter.URLTestHistory{LastOK: time.Now().Add(-time.Second), Delay: delay})
|
||||
}
|
||||
|
||||
// verdictTTL mirrors protocol/group's healthTTL floor (3 x interval, never below
|
||||
// 10m) for board assertions in this package. The generator's least_test interval
|
||||
// is the global probe interval (or the 3m engine default), so 10m is the floor.
|
||||
func verdictTTL(interval time.Duration) time.Duration {
|
||||
ttl := 3 * interval
|
||||
if ttl < 10*time.Minute {
|
||||
ttl = 10 * time.Minute
|
||||
}
|
||||
return ttl
|
||||
}
|
||||
|
||||
// dialDest is the destination the retry dials are aimed at; the stubs ignore it.
|
||||
func dialDest() M.Socksaddr { return M.Socksaddr{Fqdn: "example.com"} }
|
||||
|
||||
// --- Scenario 1 fixtures --------------------------------------------------------
|
||||
|
||||
// urltestHealthModel is a minimal but realistic engine config: a tproxy inbound,
|
||||
// a DoH resolver, two SS members on TEST-NET, and a least_test urltest group the
|
||||
// rule routes into. The generator turns this into a real urltest outbound whose
|
||||
// tag schema, interval and member list drive the stub-driven dial checks below.
|
||||
func urltestHealthModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.ProbeInterval = "60s" // a short, explicit interval -> verdictTTL = 3m floor 10m
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{
|
||||
{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12401, TCP: true, UDP: true},
|
||||
},
|
||||
Nodes: []model.Node{
|
||||
{Name: "primary", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#primary"},
|
||||
{Name: "backup", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#backup"},
|
||||
},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "ha", Source: "manual", Nodes: []string{"primary", "backup"}, Strategy: "leastping"},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-ha", Enabled: true, Order: 10, DstPort: "443", Target: "group:ha"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// urlTestOptsOf extracts the urltest options the generator emitted for tag.
|
||||
func urlTestOptsOf(t *testing.T, opts option.Options, tag string) option.URLTestOutboundOptions {
|
||||
t.Helper()
|
||||
ob := obByTag(opts, tag)
|
||||
if ob == nil {
|
||||
t.Fatalf("group outbound %q not emitted; outbounds=%v", tag, outboundTags(opts))
|
||||
}
|
||||
ut, ok := ob.Options.(*option.URLTestOutboundOptions)
|
||||
if !ok {
|
||||
t.Fatalf("outbound %q is %T, want *URLTestOutboundOptions", tag, ob.Options)
|
||||
}
|
||||
return *ut
|
||||
}
|
||||
|
||||
// haGroupOpts generates urltestHealthModel and returns the urltest options for the
|
||||
// "ha" group — the tag schema, interval and member list the retry tests dial against.
|
||||
func haGroupOpts(t *testing.T) option.URLTestOutboundOptions {
|
||||
t.Helper()
|
||||
opts, warns, err := GenerateWithWarnings(urltestHealthModel())
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("unexpected warnings: %v", warns)
|
||||
}
|
||||
return urlTestOptsOf(t, opts, "ha")
|
||||
}
|
||||
|
||||
// haInterval returns the effective probe interval of the generated "ha" group,
|
||||
// falling back to the engine default when the generator left it 0.
|
||||
func haInterval(ut option.URLTestOutboundOptions) time.Duration {
|
||||
if d := time.Duration(ut.Interval); d > 0 {
|
||||
return d
|
||||
}
|
||||
return C.DefaultURLTestInterval
|
||||
}
|
||||
|
||||
// --- Scenario 1: retry heals a dead current node --------------------------------
|
||||
|
||||
// TestHealthRetryHealsDeadCurrentNode is the retry contract (plan §5.B / Д1): with
|
||||
// the checker's pick (primary) dead on the wire but backup alive, ONE DialContext
|
||||
// through the urltest group succeeds — primary is marked dead on the board and the
|
||||
// dial re-picks backup. The dial order [primary, backup] is the observable proof
|
||||
// the retry happened; the kept history entry is the proof mark-fail did not delete.
|
||||
func TestHealthRetryHealsDeadCurrentNode(t *testing.T) {
|
||||
ut := haGroupOpts(t)
|
||||
ttl := verdictTTL(haInterval(ut))
|
||||
|
||||
hist := urltest.NewHistoryStorage()
|
||||
t.Cleanup(func() { _ = hist.Close() })
|
||||
var dialed []string
|
||||
primary := &stubNode{tag: "primary", fail: true, dialed: &dialed}
|
||||
backup := &stubNode{tag: "backup", dialed: &dialed}
|
||||
mgr := &stubOutboundManager{nodes: map[string]adapter.Outbound{"primary": primary, "backup": backup}}
|
||||
|
||||
g := newHealthGroup(t, hist, mgr, "ha", ut)
|
||||
// The checker had picked primary (lower delay); it dies between ticks.
|
||||
storeAlive(hist, "primary", 10)
|
||||
storeAlive(hist, "backup", 100)
|
||||
|
||||
conn, err := g.DialContext(context.Background(), N.NetworkTCP, dialDest())
|
||||
if err != nil {
|
||||
t.Fatalf("DialContext failed despite a live member: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
|
||||
if len(dialed) != 2 || dialed[0] != "primary" || dialed[1] != "backup" {
|
||||
t.Fatalf("dial order = %v, want [primary backup] (primary marked dead, backup retried)", dialed)
|
||||
}
|
||||
if v := hist.Verdict("primary", ttl); v != urltest.VerdictDead {
|
||||
t.Fatalf("verdict(primary) = %v, want dead after the failed dial marked it", v)
|
||||
}
|
||||
// mark-fail keeps the entry (plan §2 Д5): a dead node stays distinguishable from
|
||||
// a never-measured one, and its last-known delay is preserved for display.
|
||||
if h := hist.LoadURLTestHistory("primary"); h == nil {
|
||||
t.Fatalf("primary history entry was deleted; mark-fail must keep it")
|
||||
}
|
||||
}
|
||||
|
||||
// TestHealthMarkFailedMovesSelectionOffDead is the second leg of scenario 1: after
|
||||
// a mark-fail (the observatory or the group's own checker learning primary died),
|
||||
// the NEXT connection skips the dead member entirely — it goes straight to the
|
||||
// alive one, never re-dialling the known-dead node. This is the board -> selection
|
||||
// read-through that made the old "alive forever" history entry stop mattering.
|
||||
func TestHealthMarkFailedMovesSelectionOffDead(t *testing.T) {
|
||||
ut := haGroupOpts(t)
|
||||
ttl := verdictTTL(haInterval(ut))
|
||||
|
||||
hist := urltest.NewHistoryStorage()
|
||||
t.Cleanup(func() { _ = hist.Close() })
|
||||
var dialed []string
|
||||
primary := &stubNode{tag: "primary", fail: true, dialed: &dialed}
|
||||
backup := &stubNode{tag: "backup", dialed: &dialed}
|
||||
mgr := &stubOutboundManager{nodes: map[string]adapter.Outbound{"primary": primary, "backup": backup}}
|
||||
|
||||
g := newHealthGroup(t, hist, mgr, "ha", ut)
|
||||
storeAlive(hist, "primary", 10)
|
||||
storeAlive(hist, "backup", 100)
|
||||
// The observatory / native checker learned primary is down and marked it.
|
||||
hist.MarkFailed("primary")
|
||||
if v := hist.Verdict("primary", ttl); v != urltest.VerdictDead {
|
||||
t.Fatalf("precondition: verdict(primary) = %v, want dead", v)
|
||||
}
|
||||
|
||||
conn, err := g.DialContext(context.Background(), N.NetworkTCP, dialDest())
|
||||
if err != nil {
|
||||
t.Fatalf("DialContext failed despite alive backup: %v", err)
|
||||
}
|
||||
_ = conn.Close()
|
||||
|
||||
// The dead member is never dialled — selection read the board verdict and moved
|
||||
// off it before the first attempt. (A regression to the old "has an entry" logic
|
||||
// would dial primary first and fail.)
|
||||
if len(dialed) != 1 || dialed[0] != "backup" {
|
||||
t.Fatalf("dial order = %v, want [backup] only — the board-dead primary must be skipped", dialed)
|
||||
}
|
||||
}
|
||||
|
||||
// TestHealthAllDeadDialFailsClosed is the fail-closed guard for the group path:
|
||||
// when EVERY member is dead, DialContext returns an error instead of silently
|
||||
// succeeding through some fallback. There is no quiet "direct" escape hatch — the
|
||||
// connection fails honestly, which is what lets the rule's kill policy decide.
|
||||
func TestHealthAllDeadDialFailsClosed(t *testing.T) {
|
||||
ut := haGroupOpts(t)
|
||||
|
||||
hist := urltest.NewHistoryStorage()
|
||||
t.Cleanup(func() { _ = hist.Close() })
|
||||
var dialed []string
|
||||
primary := &stubNode{tag: "primary", fail: true, dialed: &dialed}
|
||||
backup := &stubNode{tag: "backup", fail: true, dialed: &dialed}
|
||||
mgr := &stubOutboundManager{nodes: map[string]adapter.Outbound{"primary": primary, "backup": backup}}
|
||||
|
||||
g := newHealthGroup(t, hist, mgr, "ha", ut)
|
||||
storeAlive(hist, "primary", 10)
|
||||
storeAlive(hist, "backup", 100)
|
||||
|
||||
_, err := g.DialContext(context.Background(), N.NetworkTCP, dialDest())
|
||||
if err == nil {
|
||||
t.Fatal("DialContext succeeded with every member dead — fail-closed is broken")
|
||||
}
|
||||
// The retry tried both members (up to dialAttemptsMax) before giving up; neither
|
||||
// is a quiet success.
|
||||
if len(dialed) != 2 {
|
||||
t.Fatalf("dial order = %v, want both members tried before the honest failure", dialed)
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
// Integration test for the observatory reachability plan over a GENERATED config
|
||||
// (plan §6-S6, scenario 3). Portable: BuildObservatoryPlan is a pure function of
|
||||
// the applied option.Options and generate.Build runs on every platform, so this
|
||||
// runs everywhere — including the Windows dev host — and the linux CI re-runs it
|
||||
// alongside the box.New+Start scenarios.
|
||||
package generate
|
||||
|
||||
import (
|
||||
"testing"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/engine"
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// egressReachModel carries the three reachability cases scenario 3 asserts against
|
||||
// the REAL generated options: a used urltest group "auto" over two nodes, an
|
||||
// UNREFERENCED selector "idle", and an egress-bound group "tunneled" whose members
|
||||
// are per-group COPIES (group-tunneled-m0-...) recorded under the copy AND the base
|
||||
// node alias. This is the generate-side counterpart of engine/probeplan_test.go's
|
||||
// fixture-driven table: here the options come out of GenerateWithWarnings, so the
|
||||
// tag schema (per-group copy names, egress detours, the rule->group routing) is the
|
||||
// one the running router actually produces.
|
||||
func egressReachModel() *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ResolverDefault = "cf"
|
||||
g.ProbeURL = "https://probe.example/204"
|
||||
g.ProbeInterval = "60s"
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Egresses: []model.Egress{
|
||||
{Name: "wan2", Type: "direct"},
|
||||
},
|
||||
Nodes: []model.Node{
|
||||
{Name: "n1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#n1"},
|
||||
{Name: "n2", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.2:8388#n2"},
|
||||
{Name: "n3", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.3:8388#n3"},
|
||||
{Name: "idle1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.9:8388#idle1"},
|
||||
},
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://1.1.1.1/dns-query", Detour: "direct"},
|
||||
},
|
||||
Chains: []model.Chain{
|
||||
{Name: "hop", Hops: []string{"node:n1", "node:n3"}}, // exit = chain-hop-h2 (n3)
|
||||
},
|
||||
Groups: []model.Group{
|
||||
{Name: "auto", Source: "manual", Nodes: []string{"n1", "n2"}, Strategy: "leastping"},
|
||||
{Name: "tunneled", Source: "manual", Nodes: []string{"n1"}, Strategy: "leastping", Egress: "wan2"},
|
||||
{Name: "idle", Source: "manual", Nodes: []string{"idle1"}, Strategy: "single"},
|
||||
},
|
||||
Rules: []model.Rule{
|
||||
{Name: "via-auto", Enabled: true, Order: 10, DstPort: "443", Target: "group:auto"},
|
||||
{Name: "via-tunneled", Enabled: true, Order: 20, DstPort: "8080", Target: "group:tunneled"},
|
||||
{Name: "via-chain", Enabled: true, Order: 30, DstPort: "8443", Target: "chain:hop"},
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// jobDials reports whether the plan probes the given dial tag.
|
||||
func jobDials(jobs []engine.ProbeJob, dial string) bool {
|
||||
for _, j := range jobs {
|
||||
if j.Dial == dial {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// jobStore returns the store aliases of the job dialling dial, or nil if absent.
|
||||
func jobStore(jobs []engine.ProbeJob, dial string) []string {
|
||||
for _, j := range jobs {
|
||||
if j.Dial == dial {
|
||||
return j.Store
|
||||
}
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// dialsOfJobs lists the dial tags of a plan, for diagnostics.
|
||||
func dialsOfJobs(jobs []engine.ProbeJob) []string {
|
||||
out := make([]string, 0, len(jobs))
|
||||
for _, j := range jobs {
|
||||
out = append(out, j.Dial)
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// containsTag reports whether store holds tag.
|
||||
func containsTag(store []string, tag string) bool {
|
||||
for _, s := range store {
|
||||
if s == tag {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// TestObservatoryPlanFromGeneratedConfig is scenario 3: BuildObservatoryPlan over
|
||||
// the REAL generated options holds exactly the reachable tags — the unreferenced
|
||||
// "idle" group and its member are absent, the used "auto" group's members are
|
||||
// probed, and the egress-bound "tunneled" group's per-group COPY is probed and
|
||||
// recorded under the copy AND the base node "n1" alias, while the base "n1"
|
||||
// outbound (a different dial path) is never itself dialled.
|
||||
func TestObservatoryPlanFromGeneratedConfig(t *testing.T) {
|
||||
m := egressReachModel()
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("unexpected warnings: %v", warns)
|
||||
}
|
||||
jobs, used := engine.BuildObservatoryPlan(opts, m.Globals.ProbeURL)
|
||||
|
||||
// The used urltest group's members are probed (plain dial paths).
|
||||
for _, want := range []string{"n1", "n2"} {
|
||||
if !jobDials(jobs, want) {
|
||||
t.Errorf("plan does not probe used member %q; jobs=%v", want, dialsOfJobs(jobs))
|
||||
}
|
||||
}
|
||||
// No dial tag appears twice — the plan is deduplicated by dial path, so the
|
||||
// egress copy and any plain-path probe of the same base node stay one job each.
|
||||
seen := make(map[string]int, len(jobs))
|
||||
for _, j := range jobs {
|
||||
seen[j.Dial]++
|
||||
}
|
||||
for dial, n := range seen {
|
||||
if n != 1 {
|
||||
t.Errorf("plan dials %q %d times — each dial path is one job", dial, n)
|
||||
}
|
||||
}
|
||||
// The unreferenced group and its member are NOT probed (reachability filters them).
|
||||
for _, absent := range []string{"idle", "idle1"} {
|
||||
if jobDials(jobs, absent) {
|
||||
t.Errorf("plan probes unreferenced %q — only rule-reachable tags are probed", absent)
|
||||
}
|
||||
if used[absent] {
|
||||
t.Errorf("used-set claims unreferenced %q", absent)
|
||||
}
|
||||
}
|
||||
// The egress-bound group's per-group COPY is the dial target (group-<g>-m<i>-<member>).
|
||||
copyTag := "group-tunneled-m0-n1"
|
||||
if !jobDials(jobs, copyTag) {
|
||||
t.Errorf("plan does not probe egress copy %q; jobs=%v", copyTag, dialsOfJobs(jobs))
|
||||
} else {
|
||||
// The copy's measurement is recorded under the copy AND the base node alias,
|
||||
// but the base "n1" outbound itself is never dialled (it is a different path).
|
||||
store := jobStore(jobs, copyTag)
|
||||
if !containsTag(store, copyTag) || !containsTag(store, "n1") {
|
||||
t.Errorf("egress copy %q store = %v, want both the copy and the base alias n1", copyTag, store)
|
||||
}
|
||||
}
|
||||
// The chain's EXIT wrapper is probed end-to-end (one probe covers the whole
|
||||
// L1..Ln path); its store is only itself — a chain prefix is nobody else's
|
||||
// path, so the base exit node n3 is NOT recorded as an alias here.
|
||||
chainExit := "chain-hop-h2"
|
||||
if !jobDials(jobs, chainExit) {
|
||||
t.Errorf("plan does not probe chain exit %q; jobs=%v", chainExit, dialsOfJobs(jobs))
|
||||
} else {
|
||||
store := jobStore(jobs, chainExit)
|
||||
if len(store) != 1 || store[0] != chainExit {
|
||||
t.Errorf("chain exit %q store = %v, want only itself (a chain prefix is nobody else's path)", chainExit, store)
|
||||
}
|
||||
}
|
||||
// The chain's intermediate node hop n1 is used (the exit detours through it) but
|
||||
// NOT probed separately — its death is visible in the exit probe, and no
|
||||
// selection depends on it. n1 IS probed via the plain "auto" group, so the
|
||||
// assertion is "no SEPARATE chain-hop-h1 job", not "n1 never dialled".
|
||||
if jobDials(jobs, "chain-hop-h1") {
|
||||
t.Errorf("plan dials intermediate chain hop chain-hop-h1 — only the exit is probed end-to-end")
|
||||
}
|
||||
if !used[chainExit] || !used["chain-hop-h1"] {
|
||||
t.Errorf("used-set is missing chain hop tags; used=%v", used)
|
||||
}
|
||||
// The used groups themselves are in the used-set but never dialled (a group is
|
||||
// expanded into its members, never probed directly).
|
||||
for _, u := range []string{"auto", "tunneled"} {
|
||||
if !used[u] {
|
||||
t.Errorf("used-set is missing referenced group %q", u)
|
||||
}
|
||||
if jobDials(jobs, u) {
|
||||
t.Errorf("plan dials group %q — a group is expanded, never dialled", u)
|
||||
}
|
||||
}
|
||||
}
|
||||
+21
-242
@@ -6,49 +6,28 @@ import (
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// This file wires WAN-profiles (`config profile`) and preset packs
|
||||
// (`config preset`) into generate. Like time-scheduled rules, they are evaluated
|
||||
// HERE at gen/reconcile time against the injected clock (b.now) rather than left
|
||||
// to the control-plane: the only condition generate can decide deterministically
|
||||
// (no live network state) is a profile's SCHEDULE window. iface/probe-conditioned
|
||||
// auto-switching stays control-plane-only (Phase-2b) and is skipped by
|
||||
// auto-select with a warning; a profile named explicitly via Globals.ActiveProfile
|
||||
// is honored regardless of its conditions.
|
||||
// This file wires WAN-profiles (`config profile`) into generate. The active
|
||||
// profile is resolved HERE at gen/reconcile time: iface-conditioned
|
||||
// auto-switching stays control-plane-only (cmd/shaterd's WAN watcher) and is
|
||||
// skipped by auto-select, while a profile named explicitly via
|
||||
// Globals.ActiveProfile is honored regardless of its condition.
|
||||
//
|
||||
// Everything here is FAIL-OPEN: an unknown rule/profile/preset name, an
|
||||
// unresolvable override target, or an inert geoip pack degrades to a warning and
|
||||
// never aborts box.New.
|
||||
// Everything here is FAIL-OPEN: an unknown rule/profile name degrades to a
|
||||
// warning and never aborts box.New.
|
||||
|
||||
// Preset pack default sort keys. The rule loop sorts by (Order, insertion), so
|
||||
// negative defaults place preset rules AHEAD of typical user rules (Order >= 0)
|
||||
// while still being overridable per-preset via Preset.Order. Among the packs the
|
||||
// order is private (local traffic direct first) -> block-ads -> ru-bypass.
|
||||
const (
|
||||
presetOrderPrivate = -30
|
||||
presetOrderBlockAds = -20
|
||||
presetOrderRuBypass = -10
|
||||
)
|
||||
|
||||
// applyProfilesAndPresets computes the effective rule set (preset packs prepended
|
||||
// + active-profile enable/disable applied) and the active profile's default
|
||||
// target/egress overrides, exactly once (guarded by effectiveComputed). It never
|
||||
// mutates the caller's *model.Model: preset rules are freshly built and the user
|
||||
// rules are shallow-copied into a new slice before any Enabled flag is toggled.
|
||||
func (b *builder) applyProfilesAndPresets() {
|
||||
// applyProfiles computes the effective rule set (active-profile enable/disable
|
||||
// applied), exactly once (guarded by effectiveComputed). It never mutates the
|
||||
// caller's *model.Model: the user rules are shallow-copied into a new slice
|
||||
// (and model.ApplyProfileRuleOverrides copies again before toggling Enabled),
|
||||
// so the profile can never reach back into b.m.Rules (only the inner slice
|
||||
// headers are shared, and those are never mutated here).
|
||||
func (b *builder) applyProfiles() {
|
||||
if b.effectiveComputed {
|
||||
return
|
||||
}
|
||||
b.effectiveComputed = true
|
||||
|
||||
// Preset packs first, then a shallow copy of the user rules. append copies each
|
||||
// model.Rule struct value into the new backing array (and
|
||||
// model.ApplyProfileRuleOverrides below copies again before toggling Enabled),
|
||||
// so the profile can never reach back into b.m.Rules (only the inner slice
|
||||
// headers are shared, and those are never mutated here).
|
||||
presets := b.presetRules()
|
||||
eff := make([]model.Rule, 0, len(presets)+len(b.m.Rules))
|
||||
eff = append(eff, presets...)
|
||||
eff = append(eff, b.m.Rules...)
|
||||
eff := append([]model.Rule(nil), b.m.Rules...)
|
||||
|
||||
if prof := b.resolveActiveProfile(); prof != nil {
|
||||
// The override application is model's (single source of truth with the
|
||||
@@ -58,8 +37,6 @@ func (b *builder) applyProfilesAndPresets() {
|
||||
for _, w := range owarns {
|
||||
b.warnf("%s", w.Error())
|
||||
}
|
||||
b.profileDefaultTarget = strings.TrimSpace(prof.DefaultTarget)
|
||||
b.profileDefaultEgress = strings.TrimSpace(prof.DefaultEgress)
|
||||
// Per-profile endpoint-resolver override (by the active WAN profile). Consumed
|
||||
// by endpointResolver() with priority OVER Globals.EndpointResolver.
|
||||
b.profileEndpointResolver = strings.TrimSpace(prof.EndpointResolver)
|
||||
@@ -69,217 +46,19 @@ func (b *builder) applyProfilesAndPresets() {
|
||||
}
|
||||
|
||||
// resolveActiveProfile delegates to model.ResolveActiveProfile — the SINGLE
|
||||
// source of truth for "which profile is active at b.now", shared with the
|
||||
// netplane divert plan (nftPlanRules) so the engine's route rules and the nft
|
||||
// source of truth for "which profile is active", shared with the netplane
|
||||
// divert plan (nftPlanRules) so the engine's route rules and the nft
|
||||
// tproxy/fail-closed coverage can never disagree about a profile-driven
|
||||
// enable/disable — folding the model warnings into b.warnings.
|
||||
//
|
||||
// Selection semantics (manual pin honored regardless of conditions; auto-select
|
||||
// = highest-Priority enabled profile whose schedule window holds, iface-driven
|
||||
// profiles skipped silently because shaterd's WAN watcher owns that condition
|
||||
// and expresses its verdict as the Globals.ActiveProfile pin) are documented on
|
||||
// model.ResolveActiveProfile.
|
||||
//
|
||||
// There is no probe condition, and there is no longer a probe FIELD: model.Profile
|
||||
// dropped ProbeURL/ProbeMode entirely. Nothing in this repository ever performed
|
||||
// the connectivity probe they promised (the only surviving ProbeURL is the
|
||||
// unrelated Group.ProbeURL urltest field), and they did worse than nothing — a
|
||||
// profile carrying a probe URL was skipped by this selector, so adding the option
|
||||
// that claimed to ADD a condition silently deleted the working schedule condition
|
||||
// next to it. The removal diagnostic that used to live here went with them: a
|
||||
// warning about an option the parser no longer reads can never fire.
|
||||
// = highest-Priority enabled profile, iface-driven profiles skipped silently
|
||||
// because shaterd's WAN watcher owns that condition and expresses its verdict
|
||||
// as the Globals.ActiveProfile pin) are documented on model.ResolveActiveProfile.
|
||||
func (b *builder) resolveActiveProfile() *model.Profile {
|
||||
prof, warns := model.ResolveActiveProfile(b.m, b.now)
|
||||
prof, warns := model.ResolveActiveProfile(b.m)
|
||||
for _, w := range warns {
|
||||
b.warnf("%s", w.Error())
|
||||
}
|
||||
return prof
|
||||
}
|
||||
|
||||
// presetRules materialises the enabled preset packs into concrete model.Rule
|
||||
// values, fed through the SAME route loop as user rules (no parallel path). Each
|
||||
// rule is tagged `preset:<name>` for legible warnings. An unknown preset name is
|
||||
// warned and skipped. Preset.Target overrides the pack default target;
|
||||
// Preset.Order overrides the pack default sort key (0 => pack default).
|
||||
func (b *builder) presetRules() []model.Rule {
|
||||
var out []model.Rule
|
||||
for _, p := range b.m.Presets {
|
||||
if !p.Enabled {
|
||||
continue
|
||||
}
|
||||
name := strings.ToLower(strings.TrimSpace(p.Name))
|
||||
order := p.Order
|
||||
switch name {
|
||||
case "block-ads":
|
||||
if order == 0 {
|
||||
order = presetOrderBlockAds
|
||||
}
|
||||
out = append(out, model.Rule{
|
||||
Name: "preset:block-ads",
|
||||
Enabled: true,
|
||||
Order: order,
|
||||
DstDomain: blockAdsSuffixes(),
|
||||
Target: presetTarget(p.Target, "block"),
|
||||
})
|
||||
case "ru-bypass":
|
||||
if order == 0 {
|
||||
order = presetOrderRuBypass
|
||||
}
|
||||
// The pack used to carry DstIP=["geoip:ru"], which is INERT in this engine
|
||||
// (route-rule geoip was removed): ruleMatchers dropped it, the rule ended
|
||||
// up matcher-less and self-skipped, and the operator got a toggle that did
|
||||
// nothing. It now references a rule-set instead, materialised by
|
||||
// registerPresetRuleSets from the SAME official sing-geoip .srs path a
|
||||
// user-defined `config ruleset` with source=geoip uses.
|
||||
out = append(out, model.Rule{
|
||||
Name: "preset:ru-bypass",
|
||||
Enabled: true,
|
||||
Order: order,
|
||||
DstRuleset: []string{ruBypassRulesetName},
|
||||
Target: presetTarget(p.Target, "direct"),
|
||||
})
|
||||
case "private":
|
||||
if order == 0 {
|
||||
order = presetOrderPrivate
|
||||
}
|
||||
out = append(out, model.Rule{
|
||||
Name: "preset:private",
|
||||
Enabled: true,
|
||||
Order: order,
|
||||
DstIP: privateCIDRs(),
|
||||
Target: presetTarget(p.Target, "direct"),
|
||||
})
|
||||
default:
|
||||
b.warnf("preset %q: unknown built-in pack, skipped", p.Name)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// The ru-bypass pack's synthetic rule-set. It is NOT a `config ruleset` the user
|
||||
// declares — the preset owns it — but it is materialised through the very same
|
||||
// geoip path (GeoRuleSetURL + remoteRuleSet, ruleset.go), so the list the preset
|
||||
// routes on is byte-identical to what a hand-written source=geoip/category=ru
|
||||
// ruleset would fetch, and it shares the rs-<name>-<category> tag scheme.
|
||||
const (
|
||||
ruBypassPresetName = "ru-bypass"
|
||||
ruBypassRulesetName = "preset-ru-bypass"
|
||||
ruBypassCountry = "ru"
|
||||
)
|
||||
|
||||
// ruBypassRuleSetTag is the box tag of the preset's geoip rule-set.
|
||||
func ruBypassRuleSetTag() string {
|
||||
return routeRulesetTagPrefix + ruBypassRulesetName + "-" + ruBypassCountry
|
||||
}
|
||||
|
||||
// presetEnabled reports whether a built-in pack is switched on.
|
||||
func (b *builder) presetEnabled(name string) bool {
|
||||
for _, p := range b.m.Presets {
|
||||
if p.Enabled && strings.EqualFold(strings.TrimSpace(p.Name), name) {
|
||||
return true
|
||||
}
|
||||
}
|
||||
return false
|
||||
}
|
||||
|
||||
// registerPresetRuleSets materialises the rule-sets the built-in preset packs
|
||||
// need, and must run AFTER buildRoutingRuleSets — that function RESETS
|
||||
// routeRulesetTags/rulesetDefined, so anything registered earlier would be wiped.
|
||||
// It also has to be separate because buildRoutingRuleSets walks b.m.Rules (the
|
||||
// user's declared rules), and preset rules exist only in b.effectiveRules.
|
||||
//
|
||||
// Today that is just ru-bypass -> the official sing-geoip geoip-ru.srs.
|
||||
//
|
||||
// FAIL-OPEN, and deliberately in the SAFE direction: if the router cannot fetch
|
||||
// the list, Russian destinations simply keep following whatever rule matches them
|
||||
// next (normally the proxy) instead of being bypassed. The failure mode is "more
|
||||
// traffic through the tunnel", never "traffic leaks direct".
|
||||
func (b *builder) registerPresetRuleSets() {
|
||||
if !b.presetEnabled(ruBypassPresetName) {
|
||||
return
|
||||
}
|
||||
// Declared, so a dangling reference is never reported as "not defined".
|
||||
b.rulesetDefined[ruBypassRulesetName] = true
|
||||
|
||||
// A user-declared `config ruleset` of the same name already materialised: reuse
|
||||
// it verbatim rather than emitting a competing set under a colliding tag.
|
||||
if len(b.routeRulesetTags[ruBypassRulesetName]) > 0 {
|
||||
return
|
||||
}
|
||||
|
||||
url, err := GeoRuleSetURL("geoip", ruBypassCountry)
|
||||
if err != nil {
|
||||
b.warnf("preset ru-bypass: %v; the pack has no effect", err)
|
||||
return
|
||||
}
|
||||
tag := ruBypassRuleSetTag()
|
||||
if b.ruleSetTagTaken(tag) {
|
||||
// Some other list already claims the tag; point the preset at it instead of
|
||||
// emitting a duplicate (a duplicate rule-set tag aborts the whole config).
|
||||
b.routeRulesetTags[ruBypassRulesetName] = []string{tag}
|
||||
return
|
||||
}
|
||||
set, ok := b.remoteRuleSet(tag, url, "", "preset \"ru-bypass\"")
|
||||
if !ok {
|
||||
// Unreachable right now (remoteRuleSet warned). Register NOTHING: the preset
|
||||
// rule then has no materialised rule-set, loses its only matcher and is
|
||||
// skipped — which is exactly the fail-open direction documented above, with
|
||||
// Russian destinations following the next matching rule (normally the proxy)
|
||||
// rather than leaking direct. Picked up on the next reconcile.
|
||||
return
|
||||
}
|
||||
b.routeRuleSets = append(b.routeRuleSets, set)
|
||||
b.routeRulesetTags[ruBypassRulesetName] = []string{tag}
|
||||
b.warnf("preset ru-bypass: routing Russian addresses direct via the official sing-geoip list %s — it is downloaded on first start and cached, so the router needs working DNS + internet the first time the preset is enabled", url)
|
||||
}
|
||||
|
||||
// presetTarget returns the operator's Target override when set, else the pack
|
||||
// default.
|
||||
func presetTarget(override, def string) string {
|
||||
if t := strings.TrimSpace(override); t != "" {
|
||||
return t
|
||||
}
|
||||
return def
|
||||
}
|
||||
|
||||
// blockAdsSuffixes is the curated ad/tracker domain-suffix pack. Entries use the
|
||||
// `suffix:` classifier (see ruleMatchers) so each matches the apex AND every
|
||||
// subdomain (label-aware sing-box domain_suffix), e.g. doubleclick.net and
|
||||
// ads.doubleclick.net both match.
|
||||
func blockAdsSuffixes() []string {
|
||||
doms := []string{
|
||||
"doubleclick.net",
|
||||
"googlesyndication.com",
|
||||
"google-analytics.com",
|
||||
"googletagmanager.com",
|
||||
"googletagservices.com",
|
||||
"adservice.google.com",
|
||||
"ads.yahoo.com",
|
||||
"adnxs.com",
|
||||
"criteo.com",
|
||||
"scorecardresearch.com",
|
||||
"moatads.com",
|
||||
"amplitude.com",
|
||||
"doubleverify.com",
|
||||
"taboola.com",
|
||||
"outbrain.com",
|
||||
}
|
||||
out := make([]string, len(doms))
|
||||
for i, d := range doms {
|
||||
out[i] = "suffix:" + d
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// privateCIDRs is the RFC1918 + loopback + link-local dst-IP pack (v4 + v6).
|
||||
func privateCIDRs() []string {
|
||||
return []string{
|
||||
"10.0.0.0/8",
|
||||
"172.16.0.0/12",
|
||||
"192.168.0.0/16",
|
||||
"127.0.0.0/8",
|
||||
"169.254.0.0/16",
|
||||
"fc00::/7",
|
||||
"::1/128",
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
//go:build linux
|
||||
|
||||
// box.New validation (Phase-2 gate, linux-only like generate_test.go) of an
|
||||
// active profile + an enabled preset pack driven through engine.Apply.
|
||||
// active profile driven through engine.Apply.
|
||||
package generate
|
||||
|
||||
import (
|
||||
@@ -10,12 +10,10 @@ import (
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// TestProfilePresetAppliesCleanly drives a model carrying an explicit active
|
||||
// profile (force-disable a rule + override the default target) AND an enabled
|
||||
// preset pack (private RFC1918 -> direct) through box.New + Start. It must Apply
|
||||
// cleanly, the default target override must land on Final, the disabled rule must
|
||||
// be gone, and the injected preset rule must be present.
|
||||
func TestProfilePresetAppliesCleanly(t *testing.T) {
|
||||
// TestProfileAppliesCleanly drives a model carrying an explicit active profile
|
||||
// (force-disable a rule) through box.New + Start. It must Apply cleanly, the
|
||||
// disabled rule must be gone, and the surviving rule must keep routing.
|
||||
func TestProfileAppliesCleanly(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
g.ActiveProfile = "home"
|
||||
@@ -23,13 +21,12 @@ func TestProfilePresetAppliesCleanly(t *testing.T) {
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12370, TCP: true, UDP: true}},
|
||||
Nodes: []model.Node{{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.5:8388#ss1"}},
|
||||
Presets: []model.Preset{{Name: "private", Enabled: true}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "lan-proxy", Enabled: true, Order: 10, Src: []string{"192.168.1.0/24"}, Target: "node:ss1"},
|
||||
{Name: "adblock", Enabled: true, Order: 20, DstDomain: []string{"ads.example"}, Target: "block"},
|
||||
},
|
||||
Profiles: []model.Profile{
|
||||
{Name: "home", Enabled: true, Priority: 1, DisableRules: []string{"adblock"}, DefaultTarget: "node:ss1"},
|
||||
{Name: "home", Enabled: true, Priority: 1, DisableRules: []string{"adblock"}},
|
||||
},
|
||||
}
|
||||
|
||||
@@ -37,15 +34,8 @@ func TestProfilePresetAppliesCleanly(t *testing.T) {
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true (warnings: %v)", warns)
|
||||
}
|
||||
if opts.Route == nil || opts.Route.Final != "ss1" {
|
||||
t.Fatalf("DefaultTarget override => Final=ss1, got %+v", opts.Route)
|
||||
}
|
||||
// Preset 'private' RFC1918 rule injected and routed direct.
|
||||
if tgt := ipcidrRuleTarget(opts.Route, "10.0.0.0/8"); tgt != tagDirect {
|
||||
t.Fatalf("preset private 10.0.0.0/8 => direct, got %q", tgt)
|
||||
}
|
||||
// Profile-disabled 'adblock' rule must be absent.
|
||||
if hasDomainRule(opts.Route, "ads.example") {
|
||||
if opts.Route == nil || hasDomainRule(opts.Route, "ads.example") {
|
||||
t.Fatalf("profile-disabled rule 'adblock' should not be emitted")
|
||||
}
|
||||
// The lan-proxy rule still routes to ss1.
|
||||
|
||||
+45
-319
@@ -1,9 +1,7 @@
|
||||
// WAN-profiles (`config profile`) + preset packs (`config preset`) wired into
|
||||
// generate. These are pure codegen assertions (no box.New), so they build/run on
|
||||
// every platform; the box.New/engine validation of a profile+preset config runs
|
||||
// in the linux suite (profile_linux_test.go). The clock is injected via b.now so
|
||||
// schedule-driven auto-select is deterministic; windows are evaluated in UTC
|
||||
// (SchedUTCOffset defaults to 0 = UTC).
|
||||
// WAN-profiles (`config profile`) wired into generate. These are pure codegen
|
||||
// assertions (no box.New), so they build/run on every platform; the
|
||||
// box.New/engine validation of a profile config runs in the linux suite
|
||||
// (profile_linux_test.go).
|
||||
package generate
|
||||
|
||||
import (
|
||||
@@ -12,13 +10,12 @@ import (
|
||||
"testing"
|
||||
"time"
|
||||
|
||||
C "github.com/sagernet/sing-box/constant"
|
||||
"github.com/sagernet/sing-box/option"
|
||||
|
||||
"github.com/sagernet/sing-box/shater/model"
|
||||
)
|
||||
|
||||
// wed12 is a fixed Wednesday 12:00 UTC used as the injected clock.
|
||||
// wed12UTC is a fixed Wednesday 12:00 UTC used as the injected clock.
|
||||
var wed12UTC = time.Date(2026, 7, 15, 12, 0, 0, 0, time.UTC)
|
||||
|
||||
// buildRouteAt generates the route for m with the clock pinned to now, returning
|
||||
@@ -34,40 +31,6 @@ func buildRouteAt(m *model.Model, now time.Time) (*option.RouteOptions, *builder
|
||||
return b.buildRoute(), b
|
||||
}
|
||||
|
||||
// --- route inspection helpers ------------------------------------------------
|
||||
|
||||
// suffixRuleTarget returns the outbound tag of the first route rule carrying the
|
||||
// given domain_suffix matcher, or "" if none.
|
||||
func suffixRuleTarget(rt *option.RouteOptions, suffix string) string {
|
||||
if rt == nil {
|
||||
return ""
|
||||
}
|
||||
for _, r := range rt.Rules {
|
||||
for _, s := range r.DefaultOptions.RawDefaultRule.DomainSuffix {
|
||||
if s == suffix {
|
||||
return r.DefaultOptions.RuleAction.RouteOptions.Outbound
|
||||
}
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
// ipcidrRuleTarget returns the outbound tag of the first route rule carrying the
|
||||
// given ip_cidr matcher, or "" if none.
|
||||
func ipcidrRuleTarget(rt *option.RouteOptions, cidr string) string {
|
||||
if rt == nil {
|
||||
return ""
|
||||
}
|
||||
for _, r := range rt.Rules {
|
||||
for _, c := range r.DefaultOptions.RawDefaultRule.IPCIDR {
|
||||
if c == cidr {
|
||||
return r.DefaultOptions.RuleAction.RouteOptions.Outbound
|
||||
}
|
||||
}
|
||||
}
|
||||
return ""
|
||||
}
|
||||
|
||||
func hasWarning(b *builder, substr string) bool {
|
||||
for _, w := range b.warnings {
|
||||
if strings.Contains(w, substr) {
|
||||
@@ -77,13 +40,12 @@ func hasWarning(b *builder, substr string) bool {
|
||||
return false
|
||||
}
|
||||
|
||||
// --- A. regression guard: no profiles/presets => inert -----------------------
|
||||
// --- A. regression guard: no profiles => inert -------------------------------
|
||||
|
||||
// TestNoProfilesPresetsUnchanged proves the new path is inert for a model that
|
||||
// declares neither profiles nor presets: the effective rule set is exactly the
|
||||
// model's rules, no profile/preset default overrides fire, and no profile/preset
|
||||
// warnings are produced (byte-identical output guard).
|
||||
func TestNoProfilesPresetsUnchanged(t *testing.T) {
|
||||
// TestNoProfilesUnchanged proves the profile path is inert for a model that
|
||||
// declares no profiles: the effective rule set is exactly the model's rules and
|
||||
// no profile warnings are produced (byte-identical output guard).
|
||||
func TestNoProfilesUnchanged(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(), // kill-switch closed => Final "block"
|
||||
Nodes: []model.Node{{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.5:8388#ss1"}},
|
||||
@@ -95,10 +57,7 @@ func TestNoProfilesPresetsUnchanged(t *testing.T) {
|
||||
rt, b := buildRouteAt(m, wed12UTC)
|
||||
|
||||
if !reflect.DeepEqual(b.effectiveRules, m.Rules) {
|
||||
t.Fatalf("effectiveRules must equal m.Rules when no presets/profiles\n got: %+v\nwant: %+v", b.effectiveRules, m.Rules)
|
||||
}
|
||||
if b.profileDefaultTarget != "" || b.profileDefaultEgress != "" {
|
||||
t.Fatalf("no profile => no default overrides, got target=%q egress=%q", b.profileDefaultTarget, b.profileDefaultEgress)
|
||||
t.Fatalf("effectiveRules must equal m.Rules when no profiles\n got: %+v\nwant: %+v", b.effectiveRules, m.Rules)
|
||||
}
|
||||
if len(b.warnings) != 0 {
|
||||
t.Fatalf("unexpected warnings for a clean model: %v", b.warnings)
|
||||
@@ -116,10 +75,10 @@ func TestNoProfilesPresetsUnchanged(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// --- B. explicit ActiveProfile: force-disable a rule + default-target override -
|
||||
// --- B. explicit ActiveProfile: force-disable a rule -------------------------
|
||||
|
||||
func TestActiveProfileDisableAndDefaultTarget(t *testing.T) {
|
||||
g := model.DefaultGlobals() // kill-switch closed => Final "block" absent an override
|
||||
func TestActiveProfileDisablesRule(t *testing.T) {
|
||||
g := model.DefaultGlobals() // kill-switch closed => Final "block"
|
||||
g.ActiveProfile = "home"
|
||||
m := &model.Model{
|
||||
Globals: g,
|
||||
@@ -129,9 +88,8 @@ func TestActiveProfileDisableAndDefaultTarget(t *testing.T) {
|
||||
{Name: "keep", Enabled: true, Order: 20, DstDomain: []string{"keep.example"}, Target: "direct"},
|
||||
},
|
||||
Profiles: []model.Profile{
|
||||
// Manual pin: honored regardless of conditions. Disables "blockme", sets
|
||||
// the default catch-all to the proxy node.
|
||||
{Name: "home", Enabled: true, Priority: 1, DisableRules: []string{"blockme"}, DefaultTarget: "node:ss1"},
|
||||
// Manual pin: honored regardless of conditions. Disables "blockme".
|
||||
{Name: "home", Enabled: true, Priority: 1, DisableRules: []string{"blockme"}},
|
||||
},
|
||||
}
|
||||
rt, b := buildRouteAt(m, wed12UTC)
|
||||
@@ -142,75 +100,36 @@ func TestActiveProfileDisableAndDefaultTarget(t *testing.T) {
|
||||
if !hasDomainRule(rt, "keep.example") {
|
||||
t.Fatalf("rule 'keep' should be untouched")
|
||||
}
|
||||
if rt.Final != "ss1" {
|
||||
t.Fatalf("DefaultTarget override => Final=ss1, got %q", rt.Final)
|
||||
if rt.Final != tagBlock {
|
||||
t.Fatalf("a profile carries no default override => Final=%q, got %q", tagBlock, rt.Final)
|
||||
}
|
||||
if len(b.warnings) != 0 {
|
||||
t.Fatalf("unexpected warnings: %v", b.warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// TestDefaultEgressAndTargetPrecedence proves DefaultTarget wins over
|
||||
// DefaultEgress when a profile sets both (a proxied default is never downgraded
|
||||
// to a bare WAN egress), while an egress-only profile still redirects Final.
|
||||
func TestDefaultEgressAndTargetPrecedence(t *testing.T) {
|
||||
base := func(active string, prof model.Profile) *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.ActiveProfile = active
|
||||
return &model.Model{
|
||||
Globals: g,
|
||||
Nodes: []model.Node{{Name: "ss1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.5:8388#ss1"}},
|
||||
Egresses: []model.Egress{{Name: "wan2", Type: "direct"}},
|
||||
Profiles: []model.Profile{prof},
|
||||
}
|
||||
}
|
||||
// --- C. auto-select: highest priority; iface profiles skipped ----------------
|
||||
|
||||
// egress-only => Final routes through the egress outbound.
|
||||
rt, _ := buildRouteAt(base("p", model.Profile{Name: "p", Enabled: true, DefaultEgress: "egress:wan2"}), wed12UTC)
|
||||
if rt.Final != "egress-wan2" {
|
||||
t.Fatalf("egress-only default => Final=egress-wan2, got %q", rt.Final)
|
||||
}
|
||||
// both set => Target (proxy) wins.
|
||||
rt, _ = buildRouteAt(base("p", model.Profile{Name: "p", Enabled: true, DefaultTarget: "node:ss1", DefaultEgress: "egress:wan2"}), wed12UTC)
|
||||
if rt.Final != "ss1" {
|
||||
t.Fatalf("Target must win over Egress => Final=ss1, got %q", rt.Final)
|
||||
}
|
||||
// unresolvable target => warn + keep the un-overridden default (kill-switch block).
|
||||
rt, b := buildRouteAt(base("p", model.Profile{Name: "p", Enabled: true, DefaultTarget: "node:ghost"}), wed12UTC)
|
||||
if rt.Final != tagBlock {
|
||||
t.Fatalf("unresolvable default target => keep Final=block, got %q", rt.Final)
|
||||
}
|
||||
if !hasWarning(b, "default target") {
|
||||
t.Fatalf("expected a warning for the unresolved default target, got %v", b.warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// --- C. auto-select: highest-priority in-window; iface/probe skipped ---------
|
||||
|
||||
// TestAutoSelectHighestPriorityInWindow proves auto-select picks the highest
|
||||
// Priority profile whose SCHEDULE window currently holds — a higher-priority but
|
||||
// OUT-of-window profile loses to a lower-priority in-window one.
|
||||
func TestAutoSelectHighestPriorityInWindow(t *testing.T) {
|
||||
rules := []model.Rule{
|
||||
{Name: "rDay", Enabled: true, Order: 10, DstDomain: []string{"day.example"}, Target: "direct"},
|
||||
{Name: "rNight", Enabled: true, Order: 20, DstDomain: []string{"night.example"}, Target: "direct"},
|
||||
}
|
||||
// TestAutoSelectHighestPriority proves auto-select picks the highest-Priority
|
||||
// enabled profile.
|
||||
func TestAutoSelectHighestPriority(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Rules: rules,
|
||||
Rules: []model.Rule{
|
||||
{Name: "rLo", Enabled: true, Order: 10, DstDomain: []string{"lo.example"}, Target: "direct"},
|
||||
{Name: "rHi", Enabled: true, Order: 20, DstDomain: []string{"hi.example"}, Target: "direct"},
|
||||
},
|
||||
Profiles: []model.Profile{
|
||||
// In window at 12:00 UTC (09:00-17:00, offset 0 => UTC), lower priority.
|
||||
{Name: "day", Enabled: true, Priority: 5, SchedStart: "09:00", SchedEnd: "17:00", DisableRules: []string{"rDay"}},
|
||||
// Higher priority but OUT of window at 12:00 (overnight 22:00-07:00).
|
||||
{Name: "night", Enabled: true, Priority: 50, SchedStart: "22:00", SchedEnd: "07:00", DisableRules: []string{"rNight"}},
|
||||
{Name: "lo", Enabled: true, Priority: 5, DisableRules: []string{"rLo"}},
|
||||
{Name: "hi", Enabled: true, Priority: 50, DisableRules: []string{"rHi"}},
|
||||
},
|
||||
}
|
||||
rt, _ := buildRouteAt(m, wed12UTC)
|
||||
if hasDomainRule(rt, "day.example") {
|
||||
t.Fatalf("in-window profile 'day' should win and disable rDay")
|
||||
if hasDomainRule(rt, "hi.example") {
|
||||
t.Fatalf("the highest-priority profile must win and disable rHi")
|
||||
}
|
||||
if !hasDomainRule(rt, "night.example") {
|
||||
t.Fatalf("out-of-window profile 'night' must NOT apply (rNight should survive)")
|
||||
if !hasDomainRule(rt, "lo.example") {
|
||||
t.Fatalf("only the winning profile applies (rLo should survive)")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -275,220 +194,31 @@ func TestUnknownActiveProfileFallsBackToAuto(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// --- D. preset packs ---------------------------------------------------------
|
||||
|
||||
func TestPresetBlockAdsAndPrivateInjected(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Presets: []model.Preset{
|
||||
{Name: "block-ads", Enabled: true},
|
||||
{Name: "private", Enabled: true},
|
||||
},
|
||||
}
|
||||
rt, b := buildRouteAt(m, wed12UTC)
|
||||
|
||||
// block-ads: a curated suffix routed to block.
|
||||
if tgt := suffixRuleTarget(rt, "doubleclick.net"); tgt != tagBlock {
|
||||
t.Fatalf("block-ads suffix doubleclick.net target = %q, want %q", tgt, tagBlock)
|
||||
}
|
||||
// private: RFC1918 dst routed direct.
|
||||
if tgt := ipcidrRuleTarget(rt, "10.0.0.0/8"); tgt != tagDirect {
|
||||
t.Fatalf("private 10.0.0.0/8 target = %q, want %q", tgt, tagDirect)
|
||||
}
|
||||
if tgt := ipcidrRuleTarget(rt, "fc00::/7"); tgt != tagDirect {
|
||||
t.Fatalf("private fc00::/7 target = %q, want %q", tgt, tagDirect)
|
||||
}
|
||||
if len(b.warnings) != 0 {
|
||||
t.Fatalf("block-ads/private should emit no warnings, got %v", b.warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPresetTargetOverride proves Preset.Target overrides the pack default target.
|
||||
func TestPresetTargetOverride(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Presets: []model.Preset{
|
||||
{Name: "block-ads", Enabled: true, Target: "direct"}, // default is block
|
||||
{Name: "private", Enabled: true, Target: "block"}, // default is direct
|
||||
},
|
||||
}
|
||||
rt, _ := buildRouteAt(m, wed12UTC)
|
||||
if tgt := suffixRuleTarget(rt, "doubleclick.net"); tgt != tagDirect {
|
||||
t.Fatalf("block-ads Target override => direct, got %q", tgt)
|
||||
}
|
||||
if tgt := ipcidrRuleTarget(rt, "10.0.0.0/8"); tgt != tagBlock {
|
||||
t.Fatalf("private Target override => block, got %q", tgt)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPresetRuBypassRoutesViaGeoIPRuleSet is the R2 regression. The pack used to
|
||||
// emit DstIP=["geoip:ru"], which is INERT in this engine (route-rule geoip was
|
||||
// removed): the rule ended up matcher-less and self-skipped, so the operator got
|
||||
// a toggle in the panel that did precisely nothing. It must now produce a REAL
|
||||
// rule backed by the official sing-geoip geoip-ru.srs rule-set.
|
||||
func TestPresetRuBypassRoutesViaGeoIPRuleSet(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: true}},
|
||||
}
|
||||
rt, b := buildRouteAt(m, wed12UTC)
|
||||
|
||||
// sniff + hijack-dns + the real ru-bypass rule.
|
||||
if len(rt.Rules) != 3 {
|
||||
t.Fatalf("ru-bypass must emit a working rule; route rules = %d, want 3", len(rt.Rules))
|
||||
}
|
||||
tag := ruBypassRuleSetTag()
|
||||
var found bool
|
||||
for _, r := range rt.Rules {
|
||||
for _, rs := range r.DefaultOptions.RawDefaultRule.RuleSet {
|
||||
if rs != tag {
|
||||
continue
|
||||
}
|
||||
found = true
|
||||
if got := r.DefaultOptions.RuleAction.RouteOptions.Outbound; got != tagDirect {
|
||||
t.Fatalf("ru-bypass must route direct, got %q", got)
|
||||
}
|
||||
}
|
||||
}
|
||||
if !found {
|
||||
t.Fatalf("no rule references the ru-bypass rule-set %q", tag)
|
||||
}
|
||||
|
||||
// The rule-set itself is materialised, remote, and points at the OFFICIAL
|
||||
// sing-geoip URL — the same one a hand-written source=geoip ruleset resolves.
|
||||
wantURL, err := GeoRuleSetURL("geoip", "ru")
|
||||
if err != nil {
|
||||
t.Fatalf("GeoRuleSetURL: %v", err)
|
||||
}
|
||||
var rsFound bool
|
||||
for _, rs := range rt.RuleSet {
|
||||
if rs.Tag != tag {
|
||||
continue
|
||||
}
|
||||
rsFound = true
|
||||
if rs.Type != C.RuleSetTypeRemote {
|
||||
t.Fatalf("ru-bypass rule-set type = %q, want remote", rs.Type)
|
||||
}
|
||||
if rs.RemoteOptions.URL != wantURL {
|
||||
t.Fatalf("ru-bypass url = %q, want %q", rs.RemoteOptions.URL, wantURL)
|
||||
}
|
||||
}
|
||||
if !rsFound {
|
||||
t.Fatalf("ru-bypass rule-set %q not attached to Route.RuleSet", tag)
|
||||
}
|
||||
|
||||
// And the operator is told the pack depends on a download — never silent.
|
||||
if !hasWarning(b, "sing-geoip") {
|
||||
t.Fatalf("expected the ru-bypass download note, got %v", b.warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPresetRuBypassDisabledEmitsNothing: the pack must not leak its rule-set into
|
||||
// configs where it is switched off.
|
||||
func TestPresetRuBypassDisabledEmitsNothing(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: false}},
|
||||
}
|
||||
rt, _ := buildRouteAt(m, wed12UTC)
|
||||
if len(rt.Rules) != 2 {
|
||||
t.Fatalf("disabled preset must inject nothing; route rules = %d, want 2", len(rt.Rules))
|
||||
}
|
||||
for _, rs := range rt.RuleSet {
|
||||
if rs.Tag == ruBypassRuleSetTag() {
|
||||
t.Fatalf("disabled ru-bypass still emitted its rule-set")
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestPresetRuBypassTargetOverride: the operator's Target override still applies
|
||||
// to the rebuilt pack (e.g. send RU through a specific node instead of direct).
|
||||
func TestPresetRuBypassTargetOverride(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Nodes: []model.Node{{Name: "n1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#n1"}},
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: true, Target: "node:n1"}},
|
||||
}
|
||||
rt, _ := buildRouteAt(m, wed12UTC)
|
||||
tag := ruBypassRuleSetTag()
|
||||
for _, r := range rt.Rules {
|
||||
for _, rs := range r.DefaultOptions.RawDefaultRule.RuleSet {
|
||||
if rs == tag {
|
||||
if got := r.DefaultOptions.RuleAction.RouteOptions.Outbound; got != "n1" {
|
||||
t.Fatalf("Target override => n1, got %q", got)
|
||||
}
|
||||
return
|
||||
}
|
||||
}
|
||||
}
|
||||
t.Fatalf("ru-bypass rule not found")
|
||||
}
|
||||
|
||||
// TestPresetRuBypassReusesUserRuleset: if the operator happens to declare a
|
||||
// `config ruleset` with the preset's reserved name, the preset must reuse it
|
||||
// rather than emit a second set under a colliding tag (a duplicate rule-set tag
|
||||
// aborts the whole config).
|
||||
func TestPresetRuBypassReusesUserRuleset(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: true}},
|
||||
Rulesets: []model.Ruleset{{Name: ruBypassRulesetName, Source: "geoip", Categories: []string{"ru"}}},
|
||||
Rules: []model.Rule{
|
||||
{Name: "user", Enabled: true, Order: 5, DstRuleset: []string{ruBypassRulesetName}, Target: "direct"},
|
||||
},
|
||||
}
|
||||
rt, _ := buildRouteAt(m, wed12UTC)
|
||||
var n int
|
||||
for _, rs := range rt.RuleSet {
|
||||
if rs.Tag == ruBypassRuleSetTag() {
|
||||
n++
|
||||
}
|
||||
}
|
||||
if n != 1 {
|
||||
t.Fatalf("rule-set tag emitted %d times, want exactly 1", n)
|
||||
}
|
||||
}
|
||||
|
||||
// TestUnknownPresetSkipped proves an unknown preset name warns and is skipped.
|
||||
func TestUnknownPresetSkipped(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Presets: []model.Preset{{Name: "bogus-pack", Enabled: true}},
|
||||
}
|
||||
rt, b := buildRouteAt(m, wed12UTC)
|
||||
if len(rt.Rules) != 2 {
|
||||
t.Fatalf("unknown preset should inject nothing; route rules = %d, want 2", len(rt.Rules))
|
||||
}
|
||||
if !hasWarning(b, "unknown built-in pack") {
|
||||
t.Fatalf("expected an unknown-pack warning, got %v", b.warnings)
|
||||
}
|
||||
}
|
||||
|
||||
// --- value-space audit: which profile conditions exist -----------------------
|
||||
|
||||
// TestProfileScheduleIsSelectableAfterProbeRemoval is what is left of the two
|
||||
// TestPlainProfileIsSelectableAfterProbeRemoval is what is left of the two
|
||||
// probe tests, and it guards the thing that actually regressed.
|
||||
//
|
||||
// probe_url/probe_mode promised "activate when the connectivity check is up/down"
|
||||
// and nothing ever ran that check — but merely SETTING probe_url removed the
|
||||
// profile from auto-select, so the option that claimed to ADD a condition silently
|
||||
// deleted the working schedule condition beside it. The fields are now gone from
|
||||
// model.Profile, so the old tests cannot even be expressed; what must stay true is
|
||||
// that a schedule-only profile is selected, and that no probe machinery survives
|
||||
// to suppress it or to warn about it.
|
||||
func TestProfileScheduleIsSelectableAfterProbeRemoval(t *testing.T) {
|
||||
// profile from auto-select, so the option that claimed to ADD a condition
|
||||
// silently disabled the profile it sat on. The fields are now gone from
|
||||
// model.Profile, so the old tests cannot even be expressed; what must stay true
|
||||
// is that a plain profile is selected, and that no probe machinery survives to
|
||||
// suppress it or to warn about it.
|
||||
func TestPlainProfileIsSelectableAfterProbeRemoval(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Rules: []model.Rule{
|
||||
{Name: "rHi", Enabled: true, Order: 10, DstDomain: []string{"hi.example"}, Target: "direct"},
|
||||
},
|
||||
Profiles: []model.Profile{
|
||||
{Name: "sched", Enabled: true, Priority: 10, DisableRules: []string{"rHi"}},
|
||||
{Name: "plain", Enabled: true, Priority: 10, DisableRules: []string{"rHi"}},
|
||||
},
|
||||
}
|
||||
rt, b := buildRouteAt(m, wed12UTC)
|
||||
if hasDomainRule(rt, "hi.example") {
|
||||
t.Fatalf("a schedule-only profile must apply and disable rHi")
|
||||
t.Fatalf("a plain profile must apply and disable rHi")
|
||||
}
|
||||
// No leftover diagnostic: warning about an option the parser no longer reads
|
||||
// would be a permanent, unactionable line in the panel.
|
||||
@@ -500,18 +230,14 @@ func TestProfileScheduleIsSelectableAfterProbeRemoval(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestIfaceProfileSkipIsSilent: match_iface IS implemented — by shaterd's WAN
|
||||
// watcher, which pins the profile via Globals.ActiveProfile and evaluates its
|
||||
// schedule window too (model.ProfileScheduleActive, AND'd with the uplink match,
|
||||
// released at both window edges). Auto-select skipping such a profile is the
|
||||
// division of labour working, not a config defect, so it must NOT warn — and in
|
||||
// particular the old "the watcher does not look at the schedule / never
|
||||
// evaluated" text was a lie on HEAD and must stay gone.
|
||||
// watcher, which pins the profile via Globals.ActiveProfile. Auto-select
|
||||
// skipping such a profile is the division of labour working, not a config
|
||||
// defect, so it must NOT warn.
|
||||
func TestIfaceProfileSkipIsSilent(t *testing.T) {
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Profiles: []model.Profile{
|
||||
{Name: "lte", Enabled: true, Priority: 10, MatchIface: []string{"wwan0"},
|
||||
SchedDays: []string{"mon"}, SchedStart: "01:00", SchedEnd: "02:00"},
|
||||
{Name: "lte", Enabled: true, Priority: 10, MatchIface: []string{"wwan0"}},
|
||||
},
|
||||
}
|
||||
_, b := buildRouteAt(m, wed12UTC)
|
||||
|
||||
+28
-64
@@ -25,12 +25,10 @@ import (
|
||||
// open => "direct". A catch-all model rule (no matchers) overrides Final
|
||||
// with its resolved target so a "default -> group/node" egress works.
|
||||
func (b *builder) buildRoute() *option.RouteOptions {
|
||||
// Resolve the active WAN-profile and materialise preset packs ONCE (the same
|
||||
// gen-time, b.now-driven evaluation schedules already use). This yields the
|
||||
// effective rule set (preset packs prepended + profile enable/disable applied)
|
||||
// that the loop below consumes, plus the profile's default-target/egress
|
||||
// overrides applied to Final after the loop. Fail-open: never aborts box.New.
|
||||
b.applyProfilesAndPresets()
|
||||
// Resolve the active WAN-profile ONCE. This yields the effective rule set
|
||||
// (profile enable/disable applied) that the loop below consumes. Fail-open:
|
||||
// never aborts box.New.
|
||||
b.applyProfiles()
|
||||
|
||||
// Kill-switch default backstop.
|
||||
final := tagBlock
|
||||
@@ -43,12 +41,6 @@ func (b *builder) buildRoute() *option.RouteOptions {
|
||||
// its rs-<name> tag and know which references actually materialised.
|
||||
b.buildRoutingRuleSets()
|
||||
|
||||
// Rule-sets owned by the built-in preset packs (today: ru-bypass -> the official
|
||||
// sing-geoip geoip-ru.srs). Must come AFTER buildRoutingRuleSets, which resets
|
||||
// the ruleset bookkeeping maps, and it cannot live inside that function because
|
||||
// preset rules exist only in b.effectiveRules, never in b.m.Rules.
|
||||
b.registerPresetRuleSets()
|
||||
|
||||
// General model rules are collected separately so the leading sniff/hijack-dns
|
||||
// (and optional DoH-block reject) rules can be spliced in AHEAD of them while the
|
||||
// catch-all loop still computes the final default target.
|
||||
@@ -84,13 +76,11 @@ func (b *builder) buildRoute() *option.RouteOptions {
|
||||
target, ok := b.resolveTarget(want)
|
||||
if !ok {
|
||||
// The rule's target is unreachable (dead group, chain that would not
|
||||
// assemble, missing egress/node). Rule.Kill decides what happens to THIS
|
||||
// rule's traffic — see ruleKillFallback.
|
||||
fallback, keep := b.ruleKillFallback(r, want)
|
||||
if !keep {
|
||||
continue
|
||||
}
|
||||
target = fallback
|
||||
// assemble, missing egress/node). Rule.Kill decides where THIS rule's
|
||||
// traffic goes instead — see ruleKillFallback. It always returns an
|
||||
// outbound: a matched rule is never dropped, so its traffic can never
|
||||
// fall through to the broader rules below or to the default route.
|
||||
target = b.ruleKillFallback(r, want)
|
||||
}
|
||||
|
||||
if b.isCatchAll(r) {
|
||||
@@ -137,33 +127,6 @@ func (b *builder) buildRoute() *option.RouteOptions {
|
||||
})
|
||||
}
|
||||
|
||||
// Active-profile default overrides win over the base/catch-all default. Egress
|
||||
// is applied first, then Target, so when a profile sets BOTH the proxy Target
|
||||
// wins the engine Final and a proxied default is never silently downgraded to a
|
||||
// bare WAN egress (leak-safe). Unresolvable => warn + keep the un-overridden
|
||||
// default (never aborts box.New). Final is only a tag (no route-action), so a
|
||||
// DPI-carrying egress used as the default can't apply its preset — note it.
|
||||
if e := b.profileDefaultEgress; e != "" {
|
||||
if tag, ok := b.resolveTarget(e); ok {
|
||||
final = tag
|
||||
if _, dpi := b.egressDPI[tag]; dpi {
|
||||
b.warnf("active profile: default egress %q carries a dpi preset but route Final has no action; dpi ignored", e)
|
||||
}
|
||||
} else {
|
||||
b.warnf("active profile: default egress %q unresolved, keeping default %q", e, final)
|
||||
}
|
||||
}
|
||||
if t := b.profileDefaultTarget; t != "" {
|
||||
if tag, ok := b.resolveTarget(t); ok {
|
||||
final = tag
|
||||
if _, dpi := b.egressDPI[tag]; dpi {
|
||||
b.warnf("active profile: default target %q carries a dpi preset but route Final has no action; dpi ignored", t)
|
||||
}
|
||||
} else {
|
||||
b.warnf("active profile: default target %q unresolved, keeping default %q", t, final)
|
||||
}
|
||||
}
|
||||
|
||||
// Assemble: the leading sniff rule labels each connection's protocol; the
|
||||
// hijack-dns rule immediately after it steals every sniffed DNS query into the
|
||||
// engine's internal DNS resolver (D14). LAN :53 is diverted here by the
|
||||
@@ -233,38 +196,39 @@ func hijackDNSRule() option.Rule {
|
||||
// ruleKillFallback implements `Rule.Kill` — the per-rule policy for what happens
|
||||
// to THIS rule's traffic when its target turns out to be unreachable (a group
|
||||
// with no live members, a chain that would not assemble, a missing egress/node).
|
||||
// It returns the outbound tag to route to and keep=true, or keep=false to drop
|
||||
// the rule entirely.
|
||||
// It always returns an outbound tag: a matched rule is NEVER dropped. Dropping
|
||||
// it would let its traffic fall through to the broader rules below and finally
|
||||
// to the global default — a silent leak of exactly the traffic the operator
|
||||
// singled out ("send social media through the work VPN" degrading into a wider
|
||||
// "everything else -> direct" rule). The default route exists ONLY for traffic
|
||||
// no rule matched, so an unresolved target fails CLOSED here.
|
||||
//
|
||||
// ""/"default" — drop the rule. Its traffic falls through to the rules BELOW and
|
||||
// ultimately to the global Final (block while the kill-switch is
|
||||
// closed). This is the historical behaviour and stays the default.
|
||||
// "closed" — route this rule's traffic to `block` HERE. The point is that it
|
||||
// must NOT fall through: "send social media through the work VPN"
|
||||
// degrading into a broader "everything else -> direct" rule below
|
||||
// is exactly the silent leak an operator asked to prevent.
|
||||
// ""/"default" — route this rule's traffic to `block`, same as "closed". The
|
||||
// historical behaviour (drop the rule, fall through) is gone:
|
||||
// a rule states a policy, and an unresolved policy must never
|
||||
// silently become somebody else's policy.
|
||||
// "closed" — route this rule's traffic to `block` HERE, explicitly.
|
||||
// "open" — route this rule's traffic `direct` instead of blocking it. This
|
||||
// is a DELIBERATE kill-switch bypass for traffic the operator has
|
||||
// judged non-critical; it leaves the router's real IP exposed for
|
||||
// this rule, which is why it is never a default and always warned.
|
||||
//
|
||||
// An unrecognised value is warned and treated as "default" (never as "open" — an
|
||||
// An unrecognised value is warned and blocks as well (never "open" — an
|
||||
// unreadable policy must not silently open a bypass).
|
||||
func (b *builder) ruleKillFallback(r model.Rule, want string) (string, bool) {
|
||||
func (b *builder) ruleKillFallback(r model.Rule, want string) string {
|
||||
switch strings.ToLower(strings.TrimSpace(r.Kill)) {
|
||||
case "", "default":
|
||||
b.warnf("rule %q: unresolved target %q, skipped (kill=default: traffic falls through to the next rules and the global default)", r.Name, want)
|
||||
return "", false
|
||||
b.warnf("rule %q: unresolved target %q, blocking this rule's traffic (kill=default) — it never falls through to the default route", r.Name, want)
|
||||
return tagBlock
|
||||
case "closed":
|
||||
b.warnf("rule %q: unresolved target %q, blocking this rule's traffic (kill=closed)", r.Name, want)
|
||||
return tagBlock, true
|
||||
return tagBlock
|
||||
case "open":
|
||||
b.warnf("rule %q: unresolved target %q, sending this rule's traffic DIRECT (kill=open — a deliberate kill-switch bypass; this traffic leaves over the plain WAN)", r.Name, want)
|
||||
return tagDirect, true
|
||||
return tagDirect
|
||||
default:
|
||||
b.warnf("rule %q: unknown kill policy %q, treating as default", r.Name, r.Kill)
|
||||
b.warnf("rule %q: unresolved target %q, skipped (kill=default: traffic falls through to the next rules and the global default)", r.Name, want)
|
||||
return "", false
|
||||
b.warnf("rule %q: unknown kill policy %q, blocking this rule's traffic — an unreadable policy must not open a bypass", r.Name, r.Kill)
|
||||
return tagBlock
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -17,7 +17,7 @@ import (
|
||||
|
||||
// killModel builds a rule pointing at a group that cannot resolve (its only
|
||||
// member node is disabled), with the given Kill policy, plus a BROADER rule below
|
||||
// it that would swallow the traffic on fall-through.
|
||||
// it that would swallow the traffic if the rule were ever dropped.
|
||||
func killModel(kill string) *model.Model {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
@@ -44,16 +44,22 @@ func domainRuleTarget(rt *option.RouteOptions, domain string) (string, bool) {
|
||||
return "", false
|
||||
}
|
||||
|
||||
// TestRuleKillDefaultFallsThrough: kill="" / "default" keeps the historical
|
||||
// behaviour — the rule is dropped and its traffic follows the rules below.
|
||||
func TestRuleKillDefaultFallsThrough(t *testing.T) {
|
||||
// TestRuleKillDefaultBlocks: kill="" / "default" is fail-closed — the rule is
|
||||
// still emitted, routing its traffic to block. Dropping it would let its traffic
|
||||
// fall through to the broader rules below and the default route, and the default
|
||||
// route is only for traffic no rule matched.
|
||||
func TestRuleKillDefaultBlocks(t *testing.T) {
|
||||
for _, kill := range []string{"", "default", "DEFAULT"} {
|
||||
opts, warns, err := GenerateWithWarnings(killModel(kill))
|
||||
if err != nil {
|
||||
t.Fatalf("%q: Generate: %v", kill, err)
|
||||
}
|
||||
if _, ok := domainRuleTarget(opts.Route, "social.example"); ok {
|
||||
t.Fatalf("%q: rule must be dropped, but it was emitted", kill)
|
||||
got, ok := domainRuleTarget(opts.Route, "social.example")
|
||||
if !ok {
|
||||
t.Fatalf("%q: rule must be emitted routing to block, but it was dropped", kill)
|
||||
}
|
||||
if got != tagBlock {
|
||||
t.Fatalf("%q: target = %q, want block", kill, got)
|
||||
}
|
||||
if !routeWarnsHave(warns, "kill=default") {
|
||||
t.Fatalf("%q: expected a kill=default warning, got %v", kill, warns)
|
||||
@@ -61,10 +67,10 @@ func TestRuleKillDefaultFallsThrough(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRuleKillClosedBlocksHere is the point of the feature: kill=closed must
|
||||
// block THIS rule's traffic in place, so it cannot fall through into a broader
|
||||
// "everything else -> direct" rule below — the silent leak the operator was
|
||||
// trying to prevent by writing the rule at all.
|
||||
// TestRuleKillClosedBlocksHere: kill=closed must block THIS rule's traffic in
|
||||
// place, so it cannot fall through into a broader "everything else -> direct"
|
||||
// rule below — the silent leak the operator was trying to prevent by writing the
|
||||
// rule at all. Same emission as the default policy, just explicit.
|
||||
func TestRuleKillClosedBlocksHere(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(killModel("closed"))
|
||||
if err != nil {
|
||||
@@ -109,15 +115,16 @@ func TestRuleKillOpenGoesDirect(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRuleKillUnknownIsDefault: an unreadable policy must NOT silently open a
|
||||
// bypass — it degrades to default (drop) and warns about the bad value.
|
||||
func TestRuleKillUnknownIsDefault(t *testing.T) {
|
||||
// TestRuleKillUnknownBlocks: an unreadable policy must NOT silently open a
|
||||
// bypass — it blocks like the default and warns about the bad value.
|
||||
func TestRuleKillUnknownBlocks(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(killModel("yes-please"))
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if _, ok := domainRuleTarget(opts.Route, "social.example"); ok {
|
||||
t.Fatalf("unknown kill policy must behave as default (dropped)")
|
||||
got, ok := domainRuleTarget(opts.Route, "social.example")
|
||||
if !ok || got != tagBlock {
|
||||
t.Fatalf("unknown kill policy must block, got %q (ok=%v)", got, ok)
|
||||
}
|
||||
if !routeWarnsHave(warns, "unknown kill policy") {
|
||||
t.Fatalf("expected an unknown-policy warning, got %v", warns)
|
||||
@@ -130,7 +137,7 @@ func TestRuleKillUnknownIsDefault(t *testing.T) {
|
||||
func TestRuleKillPreservesFailClosedInvariant(t *testing.T) {
|
||||
for _, kill := range []string{"", "default", "closed"} {
|
||||
m := killModel(kill)
|
||||
m.Rules = m.Rules[:1] // drop the broad fall-through rule
|
||||
m.Rules = m.Rules[:1] // drop the broad second rule
|
||||
opts, _, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("%q: Generate: %v", kill, err)
|
||||
|
||||
@@ -144,7 +144,9 @@ func TestDokodemoTcpUdpBindsBoth(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestRuleKillPoliciesApply drives all three Rule.Kill policies through box.New +
|
||||
// Start: the emitted block/direct fallbacks must be real, resolvable outbounds.
|
||||
// Start: every rule with an unresolved target must still be emitted (block for
|
||||
// closed/default, direct for open) and the fallback tags must be real, resolvable
|
||||
// outbounds.
|
||||
func TestRuleKillPoliciesApply(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
@@ -164,51 +166,16 @@ func TestRuleKillPoliciesApply(t *testing.T) {
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true (warnings: %v)", warns)
|
||||
}
|
||||
if !hasDomainRule(opts.Route, "closed.example") {
|
||||
t.Fatalf("kill=closed must emit a rule")
|
||||
if got, ok := domainRuleTarget(opts.Route, "closed.example"); !ok || got != tagBlock {
|
||||
t.Fatalf("kill=closed must emit a rule routed to block, got %q (ok=%v)", got, ok)
|
||||
}
|
||||
if !hasDomainRule(opts.Route, "open.example") {
|
||||
t.Fatalf("kill=open must emit a rule")
|
||||
if got, ok := domainRuleTarget(opts.Route, "open.example"); !ok || got != tagDirect {
|
||||
t.Fatalf("kill=open must emit a rule routed to direct, got %q (ok=%v)", got, ok)
|
||||
}
|
||||
if hasDomainRule(opts.Route, "default.example") {
|
||||
t.Fatalf("kill=default must drop the rule")
|
||||
if got, ok := domainRuleTarget(opts.Route, "default.example"); !ok || got != tagBlock {
|
||||
t.Fatalf("kill=default must emit a rule routed to block, got %q (ok=%v)", got, ok)
|
||||
}
|
||||
if opts.Route.Final != tagBlock {
|
||||
t.Fatalf("Final = %q, want block", opts.Route.Final)
|
||||
}
|
||||
}
|
||||
|
||||
// TestPresetRuBypassApplies is the R2 box.New case: the rebuilt ru-bypass pack
|
||||
// must produce a config the engine accepts, with its geoip rule-set wired in.
|
||||
//
|
||||
// NOTE for the VM run: this Start()s a remote rule-set, so the box needs working
|
||||
// DNS + internet. That dependency is itself the finding worth watching — see the
|
||||
// report's note on RemoteRuleSet.StartContext failing box.Start.
|
||||
func TestPresetRuBypassApplies(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
m := &model.Model{
|
||||
Globals: g,
|
||||
Inbounds: []model.Inbound{{Name: "lan", Enabled: true, Type: "tproxy", TproxyPort: 12399, TCP: true, UDP: true}},
|
||||
Nodes: []model.Node{{Name: "n1", Enabled: true, URI: "ss://aes-256-gcm:secret@203.0.113.1:8388#n1"}},
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: true}},
|
||||
}
|
||||
|
||||
opts, warns, changed := applyAndClose(t, m)
|
||||
if !changed {
|
||||
t.Fatalf("expected Apply changed==true (warnings: %v)", warns)
|
||||
}
|
||||
tag := ruBypassRuleSetTag()
|
||||
var wired bool
|
||||
for _, rs := range opts.Route.RuleSet {
|
||||
if rs.Tag == tag {
|
||||
wired = true
|
||||
}
|
||||
}
|
||||
if !wired {
|
||||
t.Fatalf("ru-bypass rule-set %q not attached after Apply", tag)
|
||||
}
|
||||
if !hasRouteToOutbound(opts, tagDirect) {
|
||||
t.Fatalf("ru-bypass must route RU addresses direct")
|
||||
}
|
||||
}
|
||||
|
||||
@@ -244,11 +244,12 @@ func TestServiceRulesPrecedeUserRules(t *testing.T) {
|
||||
|
||||
// --- kill-switch / fail-closed ----------------------------------------------
|
||||
|
||||
// TestRuleToDeadGroupFallsToFinalBlock is the kill-switch case: a rule pointing at
|
||||
// a group whose every member node is disabled must NOT silently leak direct — the
|
||||
// group is dropped, the rule is skipped, and traffic falls through to Final, which
|
||||
// is "block" while the kill-switch is closed.
|
||||
func TestRuleToDeadGroupFallsToFinalBlock(t *testing.T) {
|
||||
// TestRuleToDeadGroupBlocksInPlace is the kill-switch case: a rule pointing at a
|
||||
// group whose every member node is disabled must NOT silently leak direct — the
|
||||
// group is dropped and the rule is emitted routing to block (default Rule.Kill is
|
||||
// fail-closed; a matched rule never falls through to the default route), with
|
||||
// Final staying "block" while the kill-switch is closed.
|
||||
func TestRuleToDeadGroupBlocksInPlace(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
m := &model.Model{
|
||||
@@ -266,8 +267,12 @@ func TestRuleToDeadGroupFallsToFinalBlock(t *testing.T) {
|
||||
if opts.Route.Final != tagBlock {
|
||||
t.Fatalf("dead group must fail CLOSED, Final = %q", opts.Route.Final)
|
||||
}
|
||||
if n := len(generalRules(opts.Route)); n != 0 {
|
||||
t.Fatalf("rule to a dead group must be skipped, got %d general rules", n)
|
||||
gen := generalRules(opts.Route)
|
||||
if len(gen) != 1 {
|
||||
t.Fatalf("rule to a dead group must be emitted routing to block, got %d general rules", len(gen))
|
||||
}
|
||||
if got := gen[0].DefaultOptions.RuleAction.RouteOptions.Outbound; got != tagBlock {
|
||||
t.Fatalf("rule to a dead group routes to %q, want block", got)
|
||||
}
|
||||
if !routeWarnsHave(warns, "unresolved target") {
|
||||
t.Fatalf("expected an unresolved-target warning, got %v", warns)
|
||||
@@ -548,8 +553,9 @@ func TestBareRuleEgressCarriesDPI(t *testing.T) {
|
||||
t.Fatalf("no rule routed to egress-frag")
|
||||
}
|
||||
|
||||
// TestMissingRuleEgressFailsClosed: a bare Egress naming nothing must skip the
|
||||
// rule (falling through to the fail-closed Final), never silently go direct.
|
||||
// TestMissingRuleEgressFailsClosed: a bare Egress naming nothing must emit the
|
||||
// rule routing to block (never silently go direct, never drop the rule so its
|
||||
// traffic falls through to the default).
|
||||
func TestMissingRuleEgressFailsClosed(t *testing.T) {
|
||||
opts, warns, err := GenerateWithWarnings(egressRuleModel(
|
||||
model.Rule{Name: "ghost", Enabled: true, Order: 1, DstPort: "443", Egress: "nope"},
|
||||
@@ -557,8 +563,8 @@ func TestMissingRuleEgressFailsClosed(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if n := len(generalRules(opts.Route)); n != 0 {
|
||||
t.Fatalf("want the rule skipped, got %d general rules", n)
|
||||
if got := ruleTargetByPort(opts.Route, 443); got != tagBlock {
|
||||
t.Fatalf("ghost-egress rule routes to %q, want block", got)
|
||||
}
|
||||
if opts.Route.Final != tagBlock {
|
||||
t.Fatalf("Final = %q, want block", opts.Route.Final)
|
||||
@@ -588,10 +594,11 @@ func TestCatchAllRuleEgressBecomesFinal(t *testing.T) {
|
||||
// (the sub-chain's hops spliced in place), a chain that references itself — directly
|
||||
// (self), through another chain (x -> y -> x), or as one of its own hops (mix) — is
|
||||
// a reference cycle that expandHops's `seen` guard must catch and refuse, failing the
|
||||
// whole chain CLOSED before any wrapper is materialised. No route rule is emitted, the
|
||||
// referring traffic falls through to the fail-closed Final, and — because the flatten
|
||||
// runs to completion before the wrapper loop — no orphan `chain-` outbound is ever
|
||||
// created (there is nothing to roll back).
|
||||
// whole chain CLOSED before any wrapper is materialised. Each referring rule is
|
||||
// emitted routing to block (an unresolved target never drops the rule to fall
|
||||
// through to the default), and — because the flatten runs to completion before
|
||||
// the wrapper loop — no orphan `chain-` outbound is ever created (there is
|
||||
// nothing to roll back).
|
||||
func TestChainCyclesTerminateFailClosed(t *testing.T) {
|
||||
g := model.DefaultGlobals()
|
||||
g.KillSwitch = "closed"
|
||||
@@ -614,8 +621,14 @@ func TestChainCyclesTerminateFailClosed(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if n := len(generalRules(opts.Route)); n != 0 {
|
||||
t.Fatalf("cyclic chains must yield no route rules, got %d", n)
|
||||
gen := generalRules(opts.Route)
|
||||
if len(gen) != 3 {
|
||||
t.Fatalf("want the three referring rules emitted routing to block, got %d", len(gen))
|
||||
}
|
||||
for _, r := range gen {
|
||||
if got := r.DefaultOptions.RuleAction.RouteOptions.Outbound; got != tagBlock {
|
||||
t.Fatalf("cyclic-chain rule routes to %q, want block", got)
|
||||
}
|
||||
}
|
||||
if opts.Route.Final != tagBlock {
|
||||
t.Fatalf("Final = %q, want block", opts.Route.Final)
|
||||
@@ -723,7 +736,9 @@ func TestChainFlattenInnerEgressAtEntry(t *testing.T) {
|
||||
// TestChainFlattenInnerEgressMidPathFailsClosed: the SAME egress-leading inner chain
|
||||
// spliced at a NON-entry position (outer=[node:a, chain:inner]) is a contradiction —
|
||||
// a chain that ENTERS over an egress cannot be a middle segment. It must be refused
|
||||
// fail-closed: no route rule, Final stays block, no orphan wrappers, and a warning.
|
||||
// fail-closed: the referring rule is emitted routing to block (an unresolved target
|
||||
// never drops the rule to fall through to the default), Final stays block, no
|
||||
// orphan wrappers, and a warning.
|
||||
func TestChainFlattenInnerEgressMidPathFailsClosed(t *testing.T) {
|
||||
m := chainFlattenModel()
|
||||
m.Egresses = []model.Egress{{Name: "wan2", Type: "direct"}}
|
||||
@@ -737,8 +752,12 @@ func TestChainFlattenInnerEgressMidPathFailsClosed(t *testing.T) {
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if n := len(generalRules(opts.Route)); n != 0 {
|
||||
t.Fatalf("mid-path egress-entry chain must yield no route rule, got %d", n)
|
||||
gen := generalRules(opts.Route)
|
||||
if len(gen) != 1 {
|
||||
t.Fatalf("want the referring rule emitted routing to block, got %d general rules", len(gen))
|
||||
}
|
||||
if got := gen[0].DefaultOptions.RuleAction.RouteOptions.Outbound; got != tagBlock {
|
||||
t.Fatalf("mid-path egress-entry chain rule routes to %q, want block", got)
|
||||
}
|
||||
if opts.Route.Final != tagBlock {
|
||||
t.Fatalf("Final = %q, want block", opts.Route.Final)
|
||||
|
||||
@@ -892,15 +892,14 @@ func (b *builder) buildRoutingRuleSets() {
|
||||
}
|
||||
|
||||
seen := map[string]bool{}
|
||||
// R6: walk the EFFECTIVE rules, not b.m.Rules. buildRoute calls
|
||||
// applyProfilesAndPresets before this, so b.effectiveRules is the set that will
|
||||
// actually be emitted: preset packs prepended and the active profile's
|
||||
// enable/disable applied. Walking the raw model instead meant a rule that the
|
||||
// active PROFILE switches on never got its dst_ruleset materialised — so
|
||||
// ruleMatchers found no tags, dropped the matcher and silently skipped the rule.
|
||||
// Switching profiles appeared to do nothing. (For a model with no
|
||||
// profiles/presets effectiveRules is an exact copy of b.m.Rules, so nothing
|
||||
// changes there.)
|
||||
// R6: walk the EFFECTIVE rules, not b.m.Rules. buildRoute calls applyProfiles
|
||||
// before this, so b.effectiveRules is the set that will actually be emitted:
|
||||
// the active profile's enable/disable applied. Walking the raw model instead
|
||||
// meant a rule that the active PROFILE switches on never got its dst_ruleset
|
||||
// materialised — so ruleMatchers found no tags, dropped the matcher and
|
||||
// silently skipped the rule. Switching profiles appeared to do nothing. (For a
|
||||
// model with no profiles effectiveRules is an exact copy of b.m.Rules, so
|
||||
// nothing changes there.)
|
||||
for _, i := range sortedRuleIndices(b.effectiveRules) {
|
||||
r := b.effectiveRules[i]
|
||||
if !r.Enabled {
|
||||
|
||||
@@ -728,9 +728,9 @@ func countRemoteRuleSets(opts option.Options) int {
|
||||
|
||||
// TestOfflineEmitsNoRemoteRuleSets is the core R5 guarantee. With no network, a
|
||||
// model stuffed with every remote-list flavour (routing url, routing geosite,
|
||||
// routing geoip, filter url, filter geosite, and the ru-bypass preset's geoip set)
|
||||
// must yield a config carrying NO remote rule-set at all — so there is nothing for
|
||||
// the engine to fetch and nothing that can fail its start.
|
||||
// routing geoip, filter url, filter geosite) must yield a config carrying NO
|
||||
// remote rule-set at all — so there is nothing for the engine to fetch and
|
||||
// nothing that can fail its start.
|
||||
//
|
||||
// Without this, RemoteRuleSet.StartContext returns an error, ruleSetStartGroup is
|
||||
// FastFail, router.Start fails, box.Start fails, engine.Apply fails, the engine
|
||||
@@ -747,7 +747,6 @@ func TestOfflineEmitsNoRemoteRuleSets(t *testing.T) {
|
||||
Resolvers: []model.Resolver{
|
||||
{Name: "cf", Type: "udp", Address: "1.1.1.1", Detour: "direct"},
|
||||
},
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: true}},
|
||||
Rulesets: []model.Ruleset{
|
||||
{Name: "u", Source: "url", URL: "https://example.invalid/list.srs"},
|
||||
{Name: "gs", Source: "geosite", Categories: []string{"youtube"}},
|
||||
@@ -871,41 +870,6 @@ func TestOfflineBlocklistBlocksNothing(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestOfflinePresetRuBypassFailsSafe: the ru-bypass pack routes Russian addresses
|
||||
// DIRECT. If its geoip list cannot load, the pack must simply not apply, so that
|
||||
// traffic keeps following the next matching rule (normally the proxy). The failure
|
||||
// mode must be "more traffic through the tunnel", never "traffic leaks direct".
|
||||
func TestOfflinePresetRuBypassFailsSafe(t *testing.T) {
|
||||
withRuleSetProbe(t, unreachableProbe)
|
||||
|
||||
m := &model.Model{
|
||||
Globals: model.DefaultGlobals(),
|
||||
Presets: []model.Preset{{Name: "ru-bypass", Enabled: true}},
|
||||
}
|
||||
opts, warns, err := GenerateWithWarnings(m)
|
||||
if err != nil {
|
||||
t.Fatalf("Generate: %v", err)
|
||||
}
|
||||
if countRemoteRuleSets(opts) != 0 {
|
||||
t.Fatalf("preset must not emit an unreachable remote rule-set")
|
||||
}
|
||||
if !warnsHaveSub(warns, "RULESET-NOT-APPLIED") {
|
||||
t.Fatalf("expected an unreachable warning, got %v", warns)
|
||||
}
|
||||
// No rule may route to direct off the back of the missing list.
|
||||
for _, r := range opts.Route.Rules {
|
||||
d := r.DefaultOptions
|
||||
if d.RuleAction.Action == C.RuleActionTypeRoute &&
|
||||
d.RuleAction.RouteOptions.Outbound == tagDirect && len(d.RuleSet) == 0 {
|
||||
t.Fatalf("ru-bypass leaked a matcher-less direct rule when its list was unavailable; rule=%+v", d)
|
||||
}
|
||||
}
|
||||
// Kill-switch Final must be untouched (fail-closed).
|
||||
if opts.Route.Final != tagBlock {
|
||||
t.Fatalf("Final must stay %q, got %q", tagBlock, opts.Route.Final)
|
||||
}
|
||||
}
|
||||
|
||||
// TestRemoteListPickedUpWhenReachable is the recovery half: the same model that
|
||||
// degraded offline must materialise fully once the network is back, which is what
|
||||
// the per-minute reconcile relies on.
|
||||
@@ -1033,7 +997,7 @@ func TestRuleSetProbeIsMemoised(t *testing.T) {
|
||||
|
||||
// TestProfileEnabledRuleMaterialisesRuleSet is the R6 regression.
|
||||
// buildRoutingRuleSets used to walk b.m.Rules (the raw model) instead of
|
||||
// b.effectiveRules (preset packs + the active profile's enable/disable). A rule
|
||||
// b.effectiveRules (the active profile's enable/disable applied). A rule
|
||||
// switched OFF in the config and switched ON by the active profile therefore never
|
||||
// had its dst_ruleset materialised: ruleMatchers found no tags, dropped the
|
||||
// matcher and silently skipped the rule. Switching profiles appeared to do nothing.
|
||||
|
||||
@@ -12,9 +12,8 @@ import (
|
||||
//
|
||||
// The semantics (days / HH:MM window / overnight wrap / fail-open on invalid
|
||||
// values / fixed UTC-offset anchor instead of tzdata) are owned by
|
||||
// model.ScheduleWindowActive — the SAME evaluator the WAN-profile watcher uses
|
||||
// (cmd/shaterd/profilewatch.go), so generate and the watcher cannot drift on
|
||||
// what "Tuesday" or a 22:00-07:00 window means. See model/schedule.go.
|
||||
// model.ScheduleWindowActive, so every consumer agrees on what "Tuesday" or a
|
||||
// 22:00-07:00 window means. See model/schedule.go.
|
||||
func (b *builder) scheduleActive(r model.Rule) bool {
|
||||
return b.scheduleWindowActive(fmt.Sprintf("rule %q", r.Name), r.SchedDays, r.SchedStart, r.SchedEnd, r.SchedUTCOffset)
|
||||
}
|
||||
@@ -22,8 +21,7 @@ func (b *builder) scheduleActive(r model.Rule) bool {
|
||||
// scheduleWindowActive adapts model.ScheduleWindowActive to the builder: it
|
||||
// evaluates the window against the injected clock (b.now) and routes the
|
||||
// evaluator's returned warnings into b.warnf, prefixed with the entity label
|
||||
// (e.g. `rule "x"` / `profile "y"`). Used by both rule schedules (scheduleActive)
|
||||
// and profile schedules (profileScheduleActive in profile.go).
|
||||
// (e.g. `rule "x"`).
|
||||
func (b *builder) scheduleWindowActive(label string, schedDays []string, schedStart, schedEnd string, schedUTCOffset int) bool {
|
||||
active, warns := model.ScheduleWindowActive(b.now, schedDays, schedStart, schedEnd, schedUTCOffset)
|
||||
for _, w := range warns {
|
||||
|
||||
@@ -1,21 +1,17 @@
|
||||
package model
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"strconv"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// Duration parsing for the interval knobs, and the resolution of the one interval
|
||||
// whose meaning is more than a number (Globals.SweepInterval).
|
||||
// Duration parsing for the interval knobs.
|
||||
|
||||
// ParseDuration parses a Go-style duration ("60s", "5m", "1h30m"). A BARE INTEGER is
|
||||
// accepted as SECONDS, which is what OpenWrt configs conventionally carry. Returns
|
||||
// ok=false for an empty or unparseable value — the caller decides what that means, and
|
||||
// the two callers deliberately decide differently (a missing probe_interval falls back
|
||||
// to the engine default; a missing sweep_interval does too, but a MALFORMED one is
|
||||
// warned about first).
|
||||
// ok=false for an empty or unparseable value — the caller decides what that means
|
||||
// (a missing probe_interval falls back to the engine default).
|
||||
//
|
||||
// This is the one algorithm every interval in the project is read with. It is defined
|
||||
// here, in the leaf package, because generate/, apply/ and engine/ all need it and all
|
||||
@@ -41,82 +37,3 @@ func ParseDuration(s string) (time.Duration, bool) {
|
||||
return d, true
|
||||
}
|
||||
|
||||
// SweepDisabledValues turn the background health sweep OFF.
|
||||
//
|
||||
// The vocabulary deliberately mirrors SilentLogLevels ("none"/"off"/"disabled"), for
|
||||
// the reason given there: an operator who has learned that "off" silences the log
|
||||
// should not have to discover that a different word is required to stop the sweep.
|
||||
// "silent" is dropped (it says nothing about a schedule) and "0" is added, because for
|
||||
// a value that is otherwise a duration, zero is the obvious way to spell "never".
|
||||
//
|
||||
// It is a closed set on purpose. Anything outside it that also fails to parse as a
|
||||
// duration is a MISTAKE, and SweepSchedule treats it as one — see there.
|
||||
var SweepDisabledValues = []string{"0", "off", "none", "disabled"}
|
||||
|
||||
// SweepIntervalMin is the floor for Globals.SweepInterval.
|
||||
//
|
||||
// It equals the per-probe timeout in shater/engine (probeAllTimeout, 5s), and that is
|
||||
// the principled reason for the number rather than a round guess: a tick shorter than
|
||||
// the time a SINGLE probe may take cannot complete a batch, so the extra ticks are
|
||||
// dropped by the sweeper's busy guard and buy nothing. What they do buy, whenever the
|
||||
// probes ARE fast, is load: the sweep's cost is batch/interval probes per second, so
|
||||
// honouring "1s" literally would run the default batch ten times faster than designed,
|
||||
// forever, on a router whose CPU and uplink belong to the user's traffic.
|
||||
//
|
||||
// A value below the floor is RAISED to it with a warning rather than rejected. Refusing
|
||||
// the config would be wildly disproportionate for a tuning knob — the kill-switch is
|
||||
// closed while the engine is down, so "refuse to start" means "the LAN is offline" —
|
||||
// and accepting it literally would be an invisible, permanent drain. Clamping is the
|
||||
// only option that is both safe and visible.
|
||||
const SweepIntervalMin = 5 * time.Second
|
||||
|
||||
// SweepSchedule resolves Globals.SweepInterval into what the engine should actually do.
|
||||
//
|
||||
// enabled false only when the value is explicitly one of SweepDisabledValues.
|
||||
// interval the tick period; 0 means "use the engine's own default", which is what an
|
||||
// EMPTY value (and any value we could not honour) resolves to. Returning 0
|
||||
// rather than a number keeps the default in exactly one place — engine's
|
||||
// sweep.go — instead of duplicating it here where it would drift.
|
||||
// warn operator-facing text when the value could not be taken at face value; ""
|
||||
// when it could.
|
||||
//
|
||||
// The three outcomes for a non-empty value are kept strictly apart:
|
||||
//
|
||||
// a disabling word -> off, no warning. The operator asked for this.
|
||||
// a parseable duration -> on at that tick, clamped up to SweepIntervalMin with a
|
||||
// warning if it is below the floor.
|
||||
// anything else -> on at the DEFAULT tick, with a warning. Note what this is not:
|
||||
// an unparseable value must never be read as "off". "I could not understand
|
||||
// you" and "you asked me to stop" are different statements, and collapsing
|
||||
// them would silently disable health data because of a typo — the operator
|
||||
// would then see their own value echoed back in UCI and the panel and
|
||||
// conclude the sweep was running.
|
||||
//
|
||||
// Pure, so both the validator (ValidateGlobals) and the consumer
|
||||
// (apply.configureSweep) resolve the value through this one function and cannot drift.
|
||||
func (g Globals) SweepSchedule() (interval time.Duration, enabled bool, warn string) {
|
||||
raw := strings.TrimSpace(g.SweepInterval)
|
||||
if raw == "" {
|
||||
return 0, true, ""
|
||||
}
|
||||
if inSet(raw, SweepDisabledValues) {
|
||||
return 0, false, ""
|
||||
}
|
||||
d, ok := ParseDuration(raw)
|
||||
if !ok {
|
||||
return 0, true, fmt.Sprintf(
|
||||
"sweep interval %q is not a duration and is applied as the default tick "+
|
||||
"(the background health sweep stays ON — an unreadable value is not "+
|
||||
"read as \"off\"); use a duration like 30s/5m/1h, or %s to turn the "+
|
||||
"sweep off.",
|
||||
g.SweepInterval, strings.Join(SweepDisabledValues, "/"))
|
||||
}
|
||||
if d < SweepIntervalMin {
|
||||
return SweepIntervalMin, true, fmt.Sprintf(
|
||||
"sweep interval %q is below the %s floor and is applied as %s; a tick "+
|
||||
"shorter than one probe's own timeout cannot finish a batch, and on a "+
|
||||
"router the extra ticks are pure load.",
|
||||
g.SweepInterval, SweepIntervalMin, SweepIntervalMin)
|
||||
}
|
||||
return d, true, ""
|
||||
}
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
package model
|
||||
|
||||
import (
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
@@ -37,93 +36,3 @@ func TestParseDuration(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// SweepSchedule keeps the three outcomes strictly apart. The one that matters most is
|
||||
// the last: an unreadable value must NOT be read as "off".
|
||||
func TestSweepSchedule(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
in string
|
||||
wantInterval time.Duration
|
||||
wantEnabled bool
|
||||
wantWarn bool
|
||||
}{
|
||||
{
|
||||
name: "unset means ON at the engine default (interval 0 = 'engine decides')",
|
||||
in: "", wantInterval: 0, wantEnabled: true,
|
||||
},
|
||||
{name: "explicit duration", in: "30s", wantInterval: 30 * time.Second, wantEnabled: true},
|
||||
{name: "minutes", in: "5m", wantInterval: 5 * time.Minute, wantEnabled: true},
|
||||
{name: "bare seconds", in: "45", wantInterval: 45 * time.Second, wantEnabled: true},
|
||||
{name: "exactly at the floor is honoured as-is", in: "5s", wantInterval: SweepIntervalMin, wantEnabled: true},
|
||||
|
||||
// Disabling: explicit, and silent (the operator asked for it).
|
||||
{name: "off by zero", in: "0", wantEnabled: false},
|
||||
{name: "off", in: "off", wantEnabled: false},
|
||||
{name: "none", in: "none", wantEnabled: false},
|
||||
{name: "disabled", in: "disabled", wantEnabled: false},
|
||||
{name: "disabling words are case- and space-insensitive", in: " OFF ", wantEnabled: false},
|
||||
|
||||
// Below the floor: clamped UP, still on, and said out loud.
|
||||
{
|
||||
name: "1s is raised to the floor with a warning",
|
||||
in: "1s", wantInterval: SweepIntervalMin, wantEnabled: true, wantWarn: true,
|
||||
},
|
||||
{
|
||||
name: "a bare 2 (seconds) is raised too",
|
||||
in: "2", wantInterval: SweepIntervalMin, wantEnabled: true, wantWarn: true,
|
||||
},
|
||||
|
||||
// Unparseable: warned, defaulted, and CRITICALLY still enabled.
|
||||
{
|
||||
name: "garbage warns and falls back to the default tick, still ON",
|
||||
in: "soon", wantInterval: 0, wantEnabled: true, wantWarn: true,
|
||||
},
|
||||
{
|
||||
name: "a plausible-looking typo is still not 'off'",
|
||||
in: "5 minutes", wantInterval: 0, wantEnabled: true, wantWarn: true,
|
||||
},
|
||||
}
|
||||
for _, c := range cases {
|
||||
interval, enabled, warn := Globals{SweepInterval: c.in}.SweepSchedule()
|
||||
if interval != c.wantInterval || enabled != c.wantEnabled {
|
||||
t.Errorf("%s: SweepSchedule(%q) = (%v, enabled=%v), want (%v, enabled=%v)",
|
||||
c.name, c.in, interval, enabled, c.wantInterval, c.wantEnabled)
|
||||
}
|
||||
if (warn != "") != c.wantWarn {
|
||||
t.Errorf("%s: SweepSchedule(%q) warn = %q, want warn=%v", c.name, c.in, warn, c.wantWarn)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// A disabling value must never produce a warning — the operator made a deliberate,
|
||||
// supported choice, and nagging about it is how a warning list becomes noise nobody
|
||||
// reads.
|
||||
func TestSweepScheduleDisablingIsQuiet(t *testing.T) {
|
||||
for _, v := range SweepDisabledValues {
|
||||
if _, enabled, warn := (Globals{SweepInterval: v}).SweepSchedule(); enabled || warn != "" {
|
||||
t.Errorf("%q: enabled=%v warn=%q, want disabled and quiet", v, enabled, warn)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// The warning surfaces through ValidateGlobals (which is what reaches the log and the
|
||||
// panel), and it must name the offending value and the way to turn the sweep off.
|
||||
func TestValidateGlobalsSweepInterval(t *testing.T) {
|
||||
ws := ValidateGlobals(Globals{SweepInterval: "soon"})
|
||||
if !hasWarning(ws, "globals", "soon") {
|
||||
t.Fatalf("no warning naming the bad value: %+v", ws)
|
||||
}
|
||||
joined := ""
|
||||
for _, w := range ws {
|
||||
joined += w.Message
|
||||
}
|
||||
if !strings.Contains(joined, "off") {
|
||||
t.Errorf("the warning must tell the operator how to disable the sweep: %q", joined)
|
||||
}
|
||||
// A good value, and a deliberate "off", are both silent.
|
||||
for _, v := range []string{"", "30s", "off"} {
|
||||
if ws := ValidateGlobals(Globals{SweepInterval: v}); len(ws) != 0 {
|
||||
t.Errorf("sweep_interval=%q warned unnecessarily: %+v", v, ws)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+43
-56
@@ -32,7 +32,7 @@
|
||||
// sing-box vless/vmess expose only `packet_encoding`
|
||||
// (xudp on by default for vless); there is nothing
|
||||
// for this value to mean.
|
||||
// Rule/Profile.SchedTZ (sched_tz) an optional IANA timezone name for the schedule
|
||||
// Rule.SchedTZ (sched_tz) an optional IANA timezone name for the schedule
|
||||
// window. The router binary embeds no tzdata and
|
||||
// bare OpenWrt ships none, so time.LoadLocation
|
||||
// failed for every real value and fell back to
|
||||
@@ -52,6 +52,19 @@
|
||||
// from the accepted set rather than implemented;
|
||||
// ValidateAlerts now flags an old config's
|
||||
// `list event 'node_down'` as an unknown event.
|
||||
// Group.ProbeURL/ProbeInterval per-group probe overrides (probe_url /
|
||||
// probe_interval on `config group`). One node in
|
||||
// two groups with different URLs made the stored
|
||||
// delays incomparable; probing is now configured
|
||||
// ONLY by Globals.ProbeURL/ProbeInterval, so every
|
||||
// measurement uses the same instrument. Old
|
||||
// options parse fine and drain on re-render.
|
||||
// Globals.SweepInterval tick period of the background health sweep
|
||||
// (sweep_interval). The sweep — "probe the whole
|
||||
// population in a circle" — is replaced by the
|
||||
// observatory, which probes only what the routing
|
||||
// rules can reach, on the global ProbeInterval.
|
||||
// The option parses fine and drains on re-render.
|
||||
//
|
||||
// Unknown UCI options are ignored by the parser (see TestUnknownSectionAndOptionIgnored),
|
||||
// so a config still carrying `option dns_mode` / `xudp_*` parses fine — the values
|
||||
@@ -81,7 +94,6 @@ type Model struct {
|
||||
Egresses []Egress
|
||||
Rulesets []Ruleset
|
||||
Rules []Rule
|
||||
Presets []Preset
|
||||
Profiles []Profile
|
||||
Resolvers []Resolver
|
||||
DNSRules []DNSRule
|
||||
@@ -173,24 +185,6 @@ type Globals struct {
|
||||
ActiveProfile string // last profile switched to (display bookkeeping)
|
||||
PanelPort int // admin-panel HTTP port; 0 = use the built-in default (8088)
|
||||
|
||||
// SweepInterval is the tick period of the background health sweep — the
|
||||
// scheduled walk that keeps every node's health fresh so the panel and the
|
||||
// group strategies have current data (see shater/engine/sweep.go).
|
||||
//
|
||||
// "" means ENABLED at the engine's default tick, not off. That default matters:
|
||||
// urltest groups probe only while they are being used and selector groups never
|
||||
// probe at all, so without the sweep a 376-node subscription reads almost
|
||||
// entirely "untested" and the per-group health numbers stay empty until somebody
|
||||
// presses "Test all nodes". Shipping the thing disabled would leave the problem
|
||||
// it exists to solve exactly as it was.
|
||||
//
|
||||
// Values: a duration ("30s", "5m", "1h") or a bare integer of seconds, parsed by
|
||||
// ParseDuration like every other interval in the model; any of
|
||||
// SweepDisabledValues to turn the sweep off; anything else is warned about and
|
||||
// falls back to the default (see SweepSchedule — an unparseable value is NOT
|
||||
// silently treated as "off"). Below SweepIntervalMin it is raised to that floor.
|
||||
SweepInterval string
|
||||
|
||||
// Geo-data source for `source=geosite`/`source=geoip` rule-sets. Resolved by
|
||||
// generate.SetGeoProvider / GeoRuleSetURL; this package only carries the values
|
||||
// (model must not import generate — generate imports model).
|
||||
@@ -220,11 +214,11 @@ type Globals struct {
|
||||
DNSFilter bool // master enable for the DNS bl/allow-list filter (D15); default false (opt-in)
|
||||
DNSIntercept bool // force ALL LAN plaintext DNS (:53) through the engine, incl. router-addressed queries; default false (opt-in)
|
||||
BlockDoH bool // block known public DoH resolvers (:443) so clients fall back to plaintext :53 (which the engine catches); default false (opt-in)
|
||||
// GroupHealth is the master-switch of OUR background group health-sweep (the scheduled
|
||||
// probe that keeps a group's member delays fresh, see apply.configureSweep); default TRUE
|
||||
// (opt-out). Off disables the sweep exactly like SweepInterval="off" while KEEPING the
|
||||
// interval value intact. It does NOT touch sing-box's own urltest probes inside a group —
|
||||
// those live their own life in generate/*; only our sweep is gated here.
|
||||
// GroupHealth is the master-switch of OUR background group health-probing (the
|
||||
// scheduled probing that keeps a group's member delays fresh, see
|
||||
// apply.configureSweep); default TRUE (opt-out). It does NOT touch sing-box's own
|
||||
// urltest probes inside a group — those live their own life in generate/*; only
|
||||
// our background probing is gated here.
|
||||
GroupHealth bool
|
||||
|
||||
// Untunnelable is the policy for LAN traffic that CANNOT be carried by the
|
||||
@@ -468,7 +462,10 @@ func (u UserInfo) Any() bool {
|
||||
return u.HasUpload || u.HasDownload || u.HasTotal || u.HasExpire
|
||||
}
|
||||
|
||||
// Node is a manual node (`config node`) or a subscription-cache node.
|
||||
// Node is a manual node or a subscription-cache node. Persistence is split by
|
||||
// FromSub: manual nodes ("") are `config node` UCI sections; subscription nodes
|
||||
// live in per-subscription JSON cache files (subcache.go) and are merged back
|
||||
// into Model.Nodes by ReadUCI, so consumers never see the difference.
|
||||
type Node struct {
|
||||
Name string
|
||||
Enabled bool
|
||||
@@ -523,8 +520,6 @@ type Group struct {
|
||||
FilterProto []string // additional group-level protocol filter
|
||||
FilterCountry []string // additional group-level country filter
|
||||
Dedup bool // drop duplicate members by fingerprint
|
||||
ProbeURL string
|
||||
ProbeInterval string
|
||||
|
||||
// Egress binds EVERY member of this group to a named `config egress`: each
|
||||
// member dials ITS OWN server through that egress outbound
|
||||
@@ -575,39 +570,32 @@ type Egress struct {
|
||||
}
|
||||
|
||||
// Profile is a `config profile` — a WAN-mode / failover conditional override.
|
||||
// When its conditions hold (active default-route interface, connectivity
|
||||
// probe up/down, and/or a time window — all AND'd), the highest-priority active
|
||||
// profile applies its overrides: force-enable / disable named rules and
|
||||
// optionally override the default catch-all target/egress. Evaluated at
|
||||
// gen/reconcile time; the cron re-applies when the active profile changes.
|
||||
// When its condition holds (active default-route interface), the highest-priority
|
||||
// active profile applies its overrides: force-enable / disable named rules and
|
||||
// optionally override the endpoint resolver. Evaluated at gen/reconcile time;
|
||||
// the cron re-applies when the active profile changes.
|
||||
type Profile struct {
|
||||
Name string
|
||||
Enabled bool
|
||||
Priority int
|
||||
|
||||
// conditions (only the specified ones are checked; all must hold)
|
||||
// condition (checked only when non-empty)
|
||||
//
|
||||
// There is no ProbeURL/ProbeMode here. They promised "activate this profile
|
||||
// while a connectivity probe succeeds/fails" and no code ever probed anything —
|
||||
// but they were worse than inert: the selector treated a profile carrying a
|
||||
// ProbeURL as having an unsatisfiable condition and SKIPPED it, so adding a
|
||||
// probe to a working schedule-driven profile silently switched that profile off.
|
||||
// A dead knob that also disables the live knob next to it is the worst version
|
||||
// of this defect, and it is why they were deleted rather than documented.
|
||||
// probe to a working profile silently switched that profile off. A dead knob
|
||||
// that also disables the live knob next to it is the worst version of this
|
||||
// defect, and it is why they were deleted rather than documented.
|
||||
//
|
||||
// MatchIface, by contrast, is real: cmd/shaterd/profilewatch.go reads the active
|
||||
// default-route device and pins the matching profile.
|
||||
MatchIface []string // active default-route dev in this set (e.g. wwan0, usb0)
|
||||
SchedDays []string // mon..sun; empty => every day
|
||||
SchedStart string // "HH:MM" at UTC+SchedUTCOffset; empty => 00:00
|
||||
SchedEnd string // "HH:MM" at UTC+SchedUTCOffset; empty/equal => all-day
|
||||
SchedUTCOffset int // minutes east of UTC (panel-captured); 0 => UTC
|
||||
MatchIface []string // active default-route dev in this set (e.g. wwan0, usb0)
|
||||
|
||||
// overrides applied while active
|
||||
EnableRules []string // rule names to force-enable
|
||||
DisableRules []string // rule names to disable
|
||||
DefaultTarget string // override the default catch-all target (group:/node:/chain:/direct/block)
|
||||
DefaultEgress string // override the default egress binding
|
||||
EnableRules []string // rule names to force-enable
|
||||
DisableRules []string // rule names to disable
|
||||
// EndpointResolver is a per-profile override of the endpoint resolver (keyed by
|
||||
// active WAN: SIM->yandex, WiFi->DoH). "" = no override. Consumed by the
|
||||
// generator (another agent); model contract only.
|
||||
@@ -657,7 +645,15 @@ type Rule struct {
|
||||
Proto string
|
||||
Target string // chain:|group:|node:|direct|block
|
||||
Egress string
|
||||
Kill string
|
||||
|
||||
// Kill is the per-rule policy for when the target cannot resolve at generate
|
||||
// time (dead group, chain that would not assemble, missing egress/node). The
|
||||
// rule is ALWAYS still emitted — its traffic never falls through to the
|
||||
// default route, which exists only for traffic no rule matched. "", "default",
|
||||
// "closed" and anything unrecognised route the traffic to block (fail-closed);
|
||||
// "open" is an explicit, warned, deliberate bypass that routes it direct. See
|
||||
// generate/route.go ruleKillFallback.
|
||||
Kill string // ''|default|closed|open
|
||||
|
||||
// Schedule: when SchedEnabled, the rule is only emitted while the wall clock
|
||||
// at UTC+SchedUTCOffset falls inside the window. Evaluated at gen/reconcile
|
||||
@@ -671,15 +667,6 @@ type Rule struct {
|
||||
SchedUTCOffset int // minutes east of UTC (panel-captured); 0 => UTC
|
||||
}
|
||||
|
||||
// Preset is a `config preset` — a toggle for a built-in curated rule pack
|
||||
// (block-ads / ru-bypass / private).
|
||||
type Preset struct {
|
||||
Name string // block-ads | ru-bypass | private
|
||||
Enabled bool
|
||||
Order int // sort key override; 0 => pack default
|
||||
Target string // target override; "" => pack default
|
||||
}
|
||||
|
||||
// Resolver is a `config resolver`.
|
||||
type Resolver struct {
|
||||
Name string
|
||||
|
||||
+15
-29
@@ -23,27 +23,25 @@ package model
|
||||
import (
|
||||
"fmt"
|
||||
"strings"
|
||||
"time"
|
||||
)
|
||||
|
||||
// ResolveActiveProfile picks the profile whose overrides apply at now, or nil:
|
||||
// ResolveActiveProfile picks the profile whose overrides apply, or nil:
|
||||
//
|
||||
// - Globals.ActiveProfile names an EXISTING, ENABLED profile => that one wins,
|
||||
// regardless of its schedule/iface conditions (an explicit manual pin is
|
||||
// always honored). A name that matches nothing usable (missing or disabled)
|
||||
// is warned and falls through to auto-select.
|
||||
// - Auto-select: among enabled profiles, the highest Priority whose SCHEDULE
|
||||
// window contains now (empty schedule => always). Priority ties break by
|
||||
// Name. Profiles with a MatchIface condition are skipped SILENTLY: the WAN
|
||||
// regardless of its iface condition (an explicit manual pin is always
|
||||
// honored). A name that matches nothing usable (missing or disabled) is
|
||||
// warned and falls through to auto-select.
|
||||
// - Auto-select: among enabled profiles, the highest Priority wins; ties break
|
||||
// by Name. Profiles with a MatchIface condition are skipped SILENTLY: the WAN
|
||||
// watcher (cmd/shaterd/profilewatch.go) owns the uplink condition — it reads
|
||||
// the active default-route device, evaluates the schedule too, and expresses
|
||||
// its verdict as the Globals.ActiveProfile pin honored above. An unpinned
|
||||
// iface profile is the watcher saying "this uplink is not active right now",
|
||||
// which is the feature working, not a config defect.
|
||||
// the active default-route device and expresses its verdict as the
|
||||
// Globals.ActiveProfile pin honored above. An unpinned iface profile is the
|
||||
// watcher saying "this uplink is not active right now", which is the feature
|
||||
// working, not a config defect.
|
||||
// - Nothing matches => nil (no active profile).
|
||||
//
|
||||
// The returned pointer aliases m.Profiles; callers must treat it as read-only.
|
||||
func ResolveActiveProfile(m *Model, now time.Time) (*Profile, []Warning) {
|
||||
func ResolveActiveProfile(m *Model) (*Profile, []Warning) {
|
||||
var warns []Warning
|
||||
profs := m.Profiles
|
||||
|
||||
@@ -67,19 +65,11 @@ func ResolveActiveProfile(m *Model, now time.Time) (*Profile, []Warning) {
|
||||
continue
|
||||
}
|
||||
// The uplink condition needs live network state this package does not have;
|
||||
// the daemon's WAN watcher owns it (schedule window included) and expresses
|
||||
// its verdict as the pin above. Deliberately BEFORE the schedule check, so an
|
||||
// iface profile never emits schedule warnings from auto-select either.
|
||||
// the daemon's WAN watcher owns it and expresses its verdict as the pin
|
||||
// above.
|
||||
if len(p.MatchIface) > 0 {
|
||||
continue
|
||||
}
|
||||
active, ws := ProfileScheduleActive(now, *p)
|
||||
for _, w := range ws {
|
||||
warns = append(warns, Warning{Section: "profile", Name: p.Name, Message: w})
|
||||
}
|
||||
if !active {
|
||||
continue
|
||||
}
|
||||
if best == nil || p.Priority > best.Priority ||
|
||||
(p.Priority == best.Priority && p.Name < best.Name) {
|
||||
best = p
|
||||
@@ -94,12 +84,8 @@ func ResolveActiveProfile(m *Model, now time.Time) (*Profile, []Warning) {
|
||||
// prof returns an unmodified copy. The input slice is never mutated — the
|
||||
// caller's model stays the desired state exactly as configured.
|
||||
//
|
||||
// Unknown names are reported as warnings, but note the rule universe is the
|
||||
// caller's: generate passes its preset rules + the user rules, while netplane
|
||||
// passes only the user rules (preset rules never carry an iface:/zone: source,
|
||||
// so they cannot contribute divert devices). A caller passing the narrower set
|
||||
// may therefore see — and is free to discard — "unknown rule" warnings for
|
||||
// names that exist only in the wider one.
|
||||
// Unknown names are reported as warnings; generate and netplane both pass the
|
||||
// user rule set, so the two consumers agree on what "unknown" means.
|
||||
func ApplyProfileRuleOverrides(rules []Rule, prof *Profile) ([]Rule, []Warning) {
|
||||
out := append([]Rule(nil), rules...)
|
||||
if prof == nil {
|
||||
|
||||
@@ -3,20 +3,16 @@ package model
|
||||
// ResolveActiveProfile / ApplyProfileRuleOverrides — the shared resolver both
|
||||
// generate (engine plan) and netplane (nft divert plan) consume. These tests
|
||||
// pin the selection semantics generate's profile tests established (manual pin
|
||||
// wins, auto-select by Priority AND schedule, iface profiles skipped by auto),
|
||||
// so moving the logic here cannot drift from the behaviour those tests verify
|
||||
// end-to-end through the route builder.
|
||||
// wins, auto-select by Priority, iface profiles skipped by auto), so moving the
|
||||
// logic here cannot drift from the behaviour those tests verify end-to-end
|
||||
// through the route builder.
|
||||
|
||||
import (
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
"time"
|
||||
)
|
||||
|
||||
// wed12UTC is a fixed Wednesday 12:00 UTC (matches generate's profile tests).
|
||||
var wed12UTC = time.Date(2026, 7, 15, 12, 0, 0, 0, time.UTC)
|
||||
|
||||
func warningsContain(ws []Warning, substr string) bool {
|
||||
for _, w := range ws {
|
||||
if strings.Contains(w.Error(), substr) {
|
||||
@@ -28,14 +24,13 @@ func warningsContain(ws []Warning, substr string) bool {
|
||||
|
||||
func TestResolveActiveProfileManualPin(t *testing.T) {
|
||||
profs := []Profile{
|
||||
// Out of window at 12:00 AND iface-conditioned — the pin must win anyway.
|
||||
{Name: "pinned", Enabled: true, Priority: 1, MatchIface: []string{"wwan0"},
|
||||
SchedStart: "22:00", SchedEnd: "23:00"},
|
||||
// Iface-conditioned — the pin must win anyway.
|
||||
{Name: "pinned", Enabled: true, Priority: 1, MatchIface: []string{"wwan0"}},
|
||||
{Name: "auto", Enabled: true, Priority: 99},
|
||||
}
|
||||
m := &Model{Globals: Globals{ActiveProfile: "pinned"}, Profiles: profs}
|
||||
|
||||
got, warns := ResolveActiveProfile(m, wed12UTC)
|
||||
got, warns := ResolveActiveProfile(m)
|
||||
if got == nil || got.Name != "pinned" {
|
||||
t.Fatalf("manual pin must be honored regardless of conditions, got %+v", got)
|
||||
}
|
||||
@@ -53,7 +48,7 @@ func TestResolveActiveProfilePinMissFallsBackAndWarns(t *testing.T) {
|
||||
{Name: "auto", Enabled: true, Priority: 1},
|
||||
},
|
||||
}
|
||||
got, warns := ResolveActiveProfile(m, wed12UTC)
|
||||
got, warns := ResolveActiveProfile(m)
|
||||
if got == nil || got.Name != "auto" {
|
||||
t.Fatalf("pin %q: expected fallback to auto-select, got %+v", pin, got)
|
||||
}
|
||||
@@ -67,30 +62,21 @@ func TestResolveActiveProfilePinMissFallsBackAndWarns(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
func TestResolveActiveProfileAutoPrioritySchedule(t *testing.T) {
|
||||
func TestResolveActiveProfileAutoPriority(t *testing.T) {
|
||||
m := &Model{
|
||||
Profiles: []Profile{
|
||||
// In window at 12:00 UTC, lower priority.
|
||||
{Name: "day", Enabled: true, Priority: 5, SchedStart: "09:00", SchedEnd: "17:00"},
|
||||
// Higher priority but OUT of window (overnight 22:00-07:00): schedule is
|
||||
// AND'd, so priority alone must not win.
|
||||
{Name: "night", Enabled: true, Priority: 50, SchedStart: "22:00", SchedEnd: "07:00"},
|
||||
{Name: "low", Enabled: true, Priority: 5},
|
||||
{Name: "high", Enabled: true, Priority: 50},
|
||||
// Highest priority but disabled.
|
||||
{Name: "dead", Enabled: false, Priority: 100},
|
||||
},
|
||||
}
|
||||
got, warns := ResolveActiveProfile(m, wed12UTC)
|
||||
if got == nil || got.Name != "day" {
|
||||
t.Fatalf("auto-select must pick the in-window profile, got %+v", got)
|
||||
got, warns := ResolveActiveProfile(m)
|
||||
if got == nil || got.Name != "high" {
|
||||
t.Fatalf("auto-select must pick the highest-priority enabled profile, got %+v", got)
|
||||
}
|
||||
if len(warns) != 0 {
|
||||
t.Fatalf("clean schedules must not warn, got %v", warns)
|
||||
}
|
||||
|
||||
// At 23:00 the overnight window holds and priority decides.
|
||||
got, _ = ResolveActiveProfile(m, wed12UTC.Add(11*time.Hour))
|
||||
if got == nil || got.Name != "night" {
|
||||
t.Fatalf("at 23:00 the overnight profile must win, got %+v", got)
|
||||
t.Fatalf("clean profiles must not warn, got %v", warns)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -101,7 +87,7 @@ func TestResolveActiveProfileTieBreaksByName(t *testing.T) {
|
||||
{Name: "alpha", Enabled: true, Priority: 7},
|
||||
},
|
||||
}
|
||||
got, _ := ResolveActiveProfile(m, wed12UTC)
|
||||
got, _ := ResolveActiveProfile(m)
|
||||
if got == nil || got.Name != "alpha" {
|
||||
t.Fatalf("priority tie must break by lowest Name, got %+v", got)
|
||||
}
|
||||
@@ -110,15 +96,13 @@ func TestResolveActiveProfileTieBreaksByName(t *testing.T) {
|
||||
func TestResolveActiveProfileSkipsIfaceProfilesSilently(t *testing.T) {
|
||||
m := &Model{
|
||||
Profiles: []Profile{
|
||||
// Highest priority, iface-conditioned, and with a schedule that would
|
||||
// WARN if evaluated (bad day token): auto-select must skip it BEFORE the
|
||||
// schedule, silently — the WAN watcher owns iface profiles.
|
||||
{Name: "lte", Enabled: true, Priority: 100, MatchIface: []string{"wwan0"},
|
||||
SchedDays: []string{"someday"}},
|
||||
// Highest priority but iface-conditioned: auto-select must skip it
|
||||
// silently — the WAN watcher owns iface profiles.
|
||||
{Name: "lte", Enabled: true, Priority: 100, MatchIface: []string{"wwan0"}},
|
||||
{Name: "plain", Enabled: true, Priority: 1},
|
||||
},
|
||||
}
|
||||
got, warns := ResolveActiveProfile(m, wed12UTC)
|
||||
got, warns := ResolveActiveProfile(m)
|
||||
if got == nil || got.Name != "plain" {
|
||||
t.Fatalf("auto-select must skip iface profiles, got %+v", got)
|
||||
}
|
||||
@@ -132,10 +116,9 @@ func TestResolveActiveProfileNothingMatches(t *testing.T) {
|
||||
Profiles: []Profile{
|
||||
{Name: "off", Enabled: false},
|
||||
{Name: "lte", Enabled: true, MatchIface: []string{"wwan0"}},
|
||||
{Name: "night", Enabled: true, SchedStart: "22:00", SchedEnd: "07:00"},
|
||||
},
|
||||
}
|
||||
if got, _ := ResolveActiveProfile(m, wed12UTC); got != nil {
|
||||
if got, _ := ResolveActiveProfile(m); got != nil {
|
||||
t.Fatalf("no applicable profile => nil, got %+v", got)
|
||||
}
|
||||
}
|
||||
|
||||
+14
-37
@@ -33,9 +33,12 @@ import (
|
||||
// ParseUCIExport already carries those defaults as concrete values, so the
|
||||
// panel's read→edit→write flow round-trips exactly. See render_test.go.
|
||||
//
|
||||
// Node.FromSub/Fingerprint/Stale are subscription-cache RUNTIME state, never
|
||||
// persisted to UCI (the parser never reads them); they are not emitted, so only
|
||||
// their zero values survive a round-trip.
|
||||
// Subscription-cache nodes (FromSub != "") are NOT emitted at all: they live in
|
||||
// per-subscription JSON cache files (see subcache.go) and are folded back in by
|
||||
// ReadUCI's MergeSubCaches. Only manual nodes become `config node` sections, and
|
||||
// the cache-only fields (from_sub/fingerprint/stale) are never written — the
|
||||
// parser still READS from_sub for the one-shot migration of old configs, which
|
||||
// drain here on their next write.
|
||||
func RenderUCIExport(m *Model) string {
|
||||
var w uciWriter
|
||||
w.b.WriteString("package shater\n")
|
||||
@@ -63,7 +66,6 @@ func RenderUCIExport(m *Model) string {
|
||||
w.strOpt("endpoint_resolver", g.EndpointResolver)
|
||||
w.strOpt("probe_url", g.ProbeURL)
|
||||
w.strOpt("probe_interval", g.ProbeInterval)
|
||||
w.strOpt("sweep_interval", g.SweepInterval)
|
||||
w.intOpt("schema_version", g.SchemaVersion)
|
||||
w.strOpt("active_profile", g.ActiveProfile)
|
||||
w.strOpt("geo_provider", g.GeoProvider)
|
||||
@@ -140,19 +142,18 @@ func RenderUCIExport(m *Model) string {
|
||||
}
|
||||
|
||||
for _, n := range m.Nodes {
|
||||
// Both manual nodes (FromSub=="") and subscription-cache nodes (FromSub!="")
|
||||
// are persisted as `config node`. A cache node additionally carries its owning
|
||||
// subscription (from_sub), its connection fingerprint, and a stale flag, so the
|
||||
// fetched cache survives reboot/reload without a re-fetch and `sub update` can
|
||||
// replace exactly its own set. Sub-cache nodes are re-fetched/replaced by
|
||||
// `shaterd sub update`; hand edits under a from_sub node are overwritten.
|
||||
// Only manual nodes (FromSub=="") are persisted as `config node`.
|
||||
// Subscription-cache nodes live in per-subscription JSON files
|
||||
// (subcache.go) and would bloat the overlay-backed UCI config — a
|
||||
// merged Model can be written back verbatim and its cache nodes
|
||||
// simply drain out of /etc/config/shater.
|
||||
if n.FromSub != "" {
|
||||
continue
|
||||
}
|
||||
w.startAnon("node")
|
||||
w.strOpt("name", n.Name)
|
||||
w.boolOpt("enabled", n.Enabled)
|
||||
w.strOpt("uri", n.URI)
|
||||
w.strOpt("from_sub", n.FromSub)
|
||||
w.strOpt("fingerprint", n.Fingerprint)
|
||||
w.boolOptTrue("stale", n.Stale)
|
||||
w.boolOpt("mux", n.Mux)
|
||||
w.intOpt("mux_concurrency", n.MuxConcurrency)
|
||||
w.uintOpt("sockopt_mark", n.Mark)
|
||||
@@ -173,8 +174,6 @@ func RenderUCIExport(m *Model) string {
|
||||
w.listOpt("filter_proto", gr.FilterProto)
|
||||
w.listOpt("filter_country", gr.FilterCountry)
|
||||
w.boolOpt("dedup", gr.Dedup)
|
||||
w.strOpt("probe_url", gr.ProbeURL)
|
||||
w.strOpt("probe_interval", gr.ProbeInterval)
|
||||
w.strOpt("egress", gr.Egress)
|
||||
}
|
||||
|
||||
@@ -227,28 +226,14 @@ func RenderUCIExport(m *Model) string {
|
||||
w.intOpt("sched_utc_offset", r.SchedUTCOffset)
|
||||
}
|
||||
|
||||
for _, p := range m.Presets {
|
||||
w.startAnon("preset")
|
||||
w.strOpt("name", p.Name)
|
||||
w.boolOpt("enabled", p.Enabled)
|
||||
w.intOpt("order", p.Order)
|
||||
w.strOpt("target", p.Target)
|
||||
}
|
||||
|
||||
for _, p := range m.Profiles {
|
||||
w.startAnon("profile")
|
||||
w.strOpt("name", p.Name)
|
||||
w.boolOpt("enabled", p.Enabled)
|
||||
w.intOpt("priority", p.Priority)
|
||||
w.listOpt("match_iface", p.MatchIface)
|
||||
w.listOpt("sched_day", p.SchedDays)
|
||||
w.strOpt("sched_start", p.SchedStart)
|
||||
w.strOpt("sched_end", p.SchedEnd)
|
||||
w.intOpt("sched_utc_offset", p.SchedUTCOffset)
|
||||
w.listOpt("enable_rule", p.EnableRules)
|
||||
w.listOpt("disable_rule", p.DisableRules)
|
||||
w.strOpt("default_target", p.DefaultTarget)
|
||||
w.strOpt("default_egress", p.DefaultEgress)
|
||||
w.strOpt("endpoint_resolver", p.EndpointResolver)
|
||||
}
|
||||
|
||||
@@ -350,14 +335,6 @@ func (w *uciWriter) boolOpt(k string, v bool) {
|
||||
fmt.Fprintf(&w.b, "\toption %s %s\n", k, renderQuote(s))
|
||||
}
|
||||
|
||||
// boolOptTrue emits the option only when v is true (a false flag stays absent so
|
||||
// it does not churn manual-node sections that never carry it, e.g. stale).
|
||||
func (w *uciWriter) boolOptTrue(k string, v bool) {
|
||||
if v {
|
||||
fmt.Fprintf(&w.b, "\toption %s %s\n", k, renderQuote("1"))
|
||||
}
|
||||
}
|
||||
|
||||
func (w *uciWriter) intOpt(k string, v int) {
|
||||
if v != 0 {
|
||||
fmt.Fprintf(&w.b, "\toption %s %s\n", k, renderQuote(strconv.Itoa(v)))
|
||||
|
||||
+17
-33
@@ -69,7 +69,7 @@ func richModel() *Model {
|
||||
Nodes: []string{"reality-nl"}, Strategy: "leastping",
|
||||
Include: []string{"US"}, Exclude: []string{"cn"},
|
||||
FilterProto: []string{"vless"}, FilterCountry: []string{"US"},
|
||||
Dedup: true, ProbeURL: "http://probe.local", ProbeInterval: "60s",
|
||||
Dedup: true,
|
||||
}},
|
||||
Chains: []Chain{{
|
||||
Name: "triple", Hops: []string{"group:main", "node:reality-nl"},
|
||||
@@ -90,17 +90,10 @@ func richModel() *Model {
|
||||
Kill: "default", SchedEnabled: true, SchedDays: []string{"mon", "tue"},
|
||||
SchedStart: "08:00", SchedEnd: "22:00", SchedUTCOffset: 180,
|
||||
}},
|
||||
Presets: []Preset{{
|
||||
Name: "block-ads", Enabled: true, Order: 5, Target: "block",
|
||||
}},
|
||||
Profiles: []Profile{{
|
||||
Name: "home", Enabled: true, Priority: 10,
|
||||
MatchIface: []string{"wwan0", "usb0"},
|
||||
SchedDays: []string{"sat", "sun"},
|
||||
// A negative offset must round-trip too (uci stores the minus sign).
|
||||
SchedStart: "00:00", SchedEnd: "23:59", SchedUTCOffset: -300,
|
||||
MatchIface: []string{"wwan0", "usb0"},
|
||||
EnableRules: []string{"pc"}, DisableRules: []string{"other"},
|
||||
DefaultTarget: "group:main", DefaultEgress: "frag",
|
||||
}},
|
||||
Resolvers: []Resolver{
|
||||
{Name: "cf", Type: "doh", Address: "https://cloudflare-dns.com/dns-query", Detour: "group:main"},
|
||||
@@ -293,8 +286,8 @@ func TestStatsBackendRoundTrip(t *testing.T) {
|
||||
}
|
||||
|
||||
// TestGroupHealthRoundTrip pins the group_health master-switch: an EXPLICIT false
|
||||
// (sweep off) survives WriteUCI->ReadUCI, and an ABSENT option falls back to the
|
||||
// DefaultGlobals seed true (opt-out), never accidentally off.
|
||||
// (background probing off) survives WriteUCI->ReadUCI, and an ABSENT option falls
|
||||
// back to the DefaultGlobals seed true (opt-out), never accidentally off.
|
||||
func TestGroupHealthRoundTrip(t *testing.T) {
|
||||
for _, v := range []bool{true, false} {
|
||||
// group_health is only meaningful over an otherwise-default globals section, so
|
||||
@@ -357,11 +350,13 @@ func TestRenderConvention(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestRenderPersistsSubCacheNodes proves subscription-cache nodes (FromSub!="")
|
||||
// ARE persisted to UCI as `config node` carrying `option from_sub` (plus
|
||||
// fingerprint/stale), so the fetched cache survives reboot/reload without a
|
||||
// re-fetch and round-trips through ParseUCIExport.
|
||||
func TestRenderPersistsSubCacheNodes(t *testing.T) {
|
||||
// TestRenderSkipsSubCacheNodes pins the persistence SPLIT: subscription-cache
|
||||
// nodes (FromSub!="") are NOT emitted to UCI at all — they live in the
|
||||
// per-subscription JSON cache files (subcache.go) and are merged back in by
|
||||
// ReadUCI. Only manual nodes become `config node` sections, and a legacy
|
||||
// from_sub section therefore drains out of /etc/config/shater on the next
|
||||
// write. Manual nodes never gain the cache-only options.
|
||||
func TestRenderSkipsSubCacheNodes(t *testing.T) {
|
||||
m := &Model{Nodes: []Node{
|
||||
{Name: "manual1", URI: "ss://x", Enabled: true},
|
||||
{Name: "cached1", URI: "ss://y", Enabled: true, FromSub: "qomar", Fingerprint: "fp", Stale: true},
|
||||
@@ -370,28 +365,17 @@ func TestRenderPersistsSubCacheNodes(t *testing.T) {
|
||||
if !strings.Contains(text, "option name 'manual1'") {
|
||||
t.Fatalf("manual node missing from render:\n%s", text)
|
||||
}
|
||||
if !strings.Contains(text, "option from_sub 'qomar'") {
|
||||
t.Fatalf("sub-cache node not persisted with from_sub:\n%s", text)
|
||||
}
|
||||
// A manual node must NOT gain a from_sub / stale option.
|
||||
if strings.Contains(text, "option from_sub ''") {
|
||||
t.Fatalf("manual node emitted an empty from_sub:\n%s", text)
|
||||
for _, forbidden := range []string{"cached1", "from_sub", "fingerprint", "stale"} {
|
||||
if strings.Contains(text, forbidden) {
|
||||
t.Fatalf("sub-cache node leaked into UCI (%q):\n%s", forbidden, text)
|
||||
}
|
||||
}
|
||||
got, err := ParseUCIExport(text)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if len(got.Nodes) != 2 {
|
||||
t.Fatalf("expected both nodes to round-trip, got %+v", got.Nodes)
|
||||
}
|
||||
var cached *Node
|
||||
for i := range got.Nodes {
|
||||
if got.Nodes[i].FromSub == "qomar" {
|
||||
cached = &got.Nodes[i]
|
||||
}
|
||||
}
|
||||
if cached == nil || cached.Name != "cached1" || cached.Fingerprint != "fp" || !cached.Stale {
|
||||
t.Fatalf("sub-cache node did not round-trip: %+v", got.Nodes)
|
||||
if len(got.Nodes) != 1 || got.Nodes[0].Name != "manual1" {
|
||||
t.Fatalf("expected only the manual node to round-trip, got %+v", got.Nodes)
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -8,6 +8,7 @@ package model
|
||||
// class of bug in this package.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
@@ -50,7 +51,12 @@ func fillNonZero(v reflect.Value, prefix string) {
|
||||
}
|
||||
|
||||
// fullModel builds a Model with every section present exactly once and every
|
||||
// field of every section set to a non-zero value.
|
||||
// field of every section set to a non-zero value — except Node's three
|
||||
// cache-only fields (FromSub/Fingerprint/Stale): a node with FromSub set is
|
||||
// deliberately NOT emitted to UCI at all (it persists in the per-subscription
|
||||
// JSON cache instead, see subcache.go), so they cannot take part in the UCI
|
||||
// round-trip. Their persistence is pinned by TestSubCacheNodeFieldsRoundTrip,
|
||||
// which walks the same fillNonZero Node through the JSON cache path.
|
||||
func fullModel() *Model {
|
||||
m := &Model{}
|
||||
mv := reflect.ValueOf(m).Elem()
|
||||
@@ -65,6 +71,9 @@ func fullModel() *Model {
|
||||
f.Set(reflect.Append(f, elem))
|
||||
}
|
||||
}
|
||||
m.Nodes[0].FromSub = ""
|
||||
m.Nodes[0].Fingerprint = ""
|
||||
m.Nodes[0].Stale = false
|
||||
return m
|
||||
}
|
||||
|
||||
@@ -113,7 +122,7 @@ func TestParseIsAFixpoint(t *testing.T) {
|
||||
sparse := "package shater\n"
|
||||
for _, typ := range []string{
|
||||
"globals", "inbound", "subscription", "node", "group", "chain", "egress",
|
||||
"ruleset", "rule", "preset", "profile", "resolver", "dns_rule",
|
||||
"ruleset", "rule", "profile", "resolver", "dns_rule",
|
||||
"blocklist", "allowlist", "device", "alert",
|
||||
} {
|
||||
sparse += "\nconfig " + typ + "\n\toption name 'x-" + typ + "'\n"
|
||||
@@ -213,3 +222,85 @@ config node
|
||||
t.Fatalf("known content lost around the unknown section: %+v", m)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSubCacheNodeFieldsRoundTrip is the JSON-cache counterpart of
|
||||
// TestEveryFieldRoundTrips: EVERY Node field — including the three cache-only
|
||||
// ones the UCI round-trip exempts — must survive SaveSubCache -> LoadSubCache.
|
||||
// A Node field added to model.go is automatically covered here by fillNonZero.
|
||||
func TestSubCacheNodeFieldsRoundTrip(t *testing.T) {
|
||||
t.Setenv("SHATER_SUBS_DIR", t.TempDir())
|
||||
var n Node
|
||||
fillNonZero(reflect.ValueOf(&n).Elem(), "c")
|
||||
n.FromSub = "mysub" // the storage key restores this on load
|
||||
|
||||
if err := SaveSubCache("mysub", []Node{n}); err != nil {
|
||||
t.Fatalf("SaveSubCache: %v", err)
|
||||
}
|
||||
got, ok, err := LoadSubCache("mysub")
|
||||
if err != nil || !ok {
|
||||
t.Fatalf("LoadSubCache: ok=%v err=%v", ok, err)
|
||||
}
|
||||
if len(got) != 1 {
|
||||
t.Fatalf("1 node written, %d read back", len(got))
|
||||
}
|
||||
wv, gv := reflect.ValueOf(n), reflect.ValueOf(got[0])
|
||||
for i := 0; i < wv.NumField(); i++ {
|
||||
field := wv.Type().Field(i).Name
|
||||
x, y := wv.Field(i).Interface(), gv.Field(i).Interface()
|
||||
if !reflect.DeepEqual(x, y) {
|
||||
t.Errorf("CACHE ROUND-TRIP HOLE Node.%s: wrote %#v, read back %#v", field, x, y)
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// TestLegacyFromSubNodesMigrateOnFirstRead pins the one-shot migration off the
|
||||
// old layout: a config still carrying 200 `from_sub` node sections (written
|
||||
// before the JSON cache existed) is imported into the subscription's cache file
|
||||
// by the FIRST MergeSubCaches, the merged Model is indistinguishable from
|
||||
// before, the next render drops the legacy sections from UCI, and a subsequent
|
||||
// read serves the same nodes from the cache file alone. Manual nodes ride
|
||||
// through every cycle untouched.
|
||||
func TestLegacyFromSubNodesMigrateOnFirstRead(t *testing.T) {
|
||||
dir := t.TempDir()
|
||||
t.Setenv("SHATER_SUBS_DIR", dir)
|
||||
|
||||
var b strings.Builder
|
||||
b.WriteString("package shater\n\nconfig subscription\n\toption name 'qomar'\n\toption url 'https://feed.example'\n\toption enabled '1'\n")
|
||||
b.WriteString("\nconfig node\n\toption name 'manual1'\n\toption enabled '1'\n\toption uri 'ss://manual'\n")
|
||||
const cached = 200
|
||||
for i := range cached {
|
||||
fmt.Fprintf(&b, "\nconfig node\n\toption name 'q-%d'\n\toption enabled '1'\n\toption uri 'vless://u@h:%d'\n\toption from_sub 'qomar'\n\toption fingerprint 'fp-%d'\n", i, 1000+i, i)
|
||||
}
|
||||
|
||||
m, err := ParseUCIExport(b.String())
|
||||
if err != nil {
|
||||
t.Fatalf("legacy parse: %v", err)
|
||||
}
|
||||
MergeSubCaches(m) // the "first read": imports the legacy nodes
|
||||
|
||||
if got := len(m.Nodes); got != cached+1 {
|
||||
t.Fatalf("merged model has %d nodes, want %d", got, cached+1)
|
||||
}
|
||||
if nodes, ok, _ := LoadSubCache("qomar"); !ok || len(nodes) != cached {
|
||||
t.Fatalf("migration did not fill the cache file: ok=%v nodes=%d", ok, len(nodes))
|
||||
}
|
||||
|
||||
// The next write drains the legacy sections out of UCI ...
|
||||
rendered := RenderUCIExport(m)
|
||||
if strings.Contains(rendered, "from_sub") {
|
||||
t.Fatalf("legacy from_sub sections survived the render:\n%.400s", rendered)
|
||||
}
|
||||
if !strings.Contains(rendered, "option name 'manual1'") {
|
||||
t.Fatal("manual node lost in the migration render")
|
||||
}
|
||||
|
||||
// ... and the read after that serves the identical node set from the cache.
|
||||
m2, err := ParseUCIExport(rendered)
|
||||
if err != nil {
|
||||
t.Fatalf("post-migration parse: %v", err)
|
||||
}
|
||||
MergeSubCaches(m2)
|
||||
if !reflect.DeepEqual(m.Nodes, m2.Nodes) {
|
||||
t.Fatalf("node set changed across the migration write:\nfirst: %d nodes\nsecond: %d nodes", len(m.Nodes), len(m2.Nodes))
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,18 +1,11 @@
|
||||
package model
|
||||
|
||||
// Schedule-window evaluation for the Sched* fields on Rule and Profile.
|
||||
// Schedule-window evaluation for the Sched* fields on Rule.
|
||||
//
|
||||
// This lives in model because model owns the FIELDS. It used to live only in
|
||||
// generate, which meant the code generator was the sole thing that could answer
|
||||
// "is this window open right now?" — and the WAN-profile watcher
|
||||
// (cmd/shaterd/profilewatch.go), which decides whether an iface-driven profile
|
||||
// applies, could not ask. It did not ask, so it ignored the schedule entirely:
|
||||
// a profile pinned to an uplink AND given a 22:00-06:00 window applied around the
|
||||
// clock, and the window was accepted, stored, displayed and inert.
|
||||
//
|
||||
// The alternative — a second copy of the evaluator in cmd/shaterd — is how the two
|
||||
// halves of a rule start disagreeing about what "Tuesday" or an overnight window
|
||||
// means. One implementation, in the package both sides already depend on.
|
||||
// This lives in model because model owns the FIELDS, and a stdlib-only leaf
|
||||
// package is the one place every consumer can share a single evaluator instead
|
||||
// of growing a second copy that disagrees about what "Tuesday" or an overnight
|
||||
// window means.
|
||||
//
|
||||
// # No tzdata, by design
|
||||
//
|
||||
@@ -30,8 +23,8 @@ package model
|
||||
// re-saved, and that is stated in the panel rather than papered over.
|
||||
//
|
||||
// Warnings are RETURNED rather than logged so each caller can prefix them with its
|
||||
// own entity label (`rule "x"` / `profile "y"`) and route them to the log or the
|
||||
// panel. The package stays a stdlib-only leaf.
|
||||
// own entity label (`rule "x"`) and route them to the log or the panel. The
|
||||
// package stays a stdlib-only leaf.
|
||||
|
||||
import (
|
||||
"fmt"
|
||||
@@ -67,8 +60,7 @@ const (
|
||||
// INVALID-TIME POLICY: an unparseable HH:MM is warned and the window is treated as
|
||||
// ALWAYS-ON (returns true). A typo must never SILENTLY switch a rule off — failing
|
||||
// open to "always active" keeps a protection rule enforcing while the admin fixes
|
||||
// the value. The same policy applies to a profile: it stays applicable rather than
|
||||
// silently dropping its overrides.
|
||||
// the value.
|
||||
func ScheduleWindowActive(now time.Time, days []string, start, end string, utcOffsetMin int) (active bool, warnings []string) {
|
||||
warn := func(format string, args ...any) {
|
||||
warnings = append(warnings, fmt.Sprintf(format, args...))
|
||||
@@ -133,13 +125,6 @@ func ScheduleWindowActive(now time.Time, days []string, start, end string, utcOf
|
||||
return nowMin >= startMin || nowMin < endMin, warnings
|
||||
}
|
||||
|
||||
// ProfileScheduleActive reports whether p's schedule window contains now. A
|
||||
// profile with no schedule is always active, so this is safe to AND into any
|
||||
// profile-selection decision.
|
||||
func ProfileScheduleActive(now time.Time, p Profile) (bool, []string) {
|
||||
return ScheduleWindowActive(now, p.SchedDays, p.SchedStart, p.SchedEnd, p.SchedUTCOffset)
|
||||
}
|
||||
|
||||
// RuleScheduleActive reports whether r's schedule window contains now. It is only
|
||||
// meaningful when r.SchedEnabled is set; the caller guards on that.
|
||||
func RuleScheduleActive(now time.Time, r Rule) (bool, []string) {
|
||||
|
||||
@@ -235,20 +235,17 @@ func TestScheduleWindowEqualStartEndIsAllDay(t *testing.T) {
|
||||
}
|
||||
}
|
||||
|
||||
// TestProfileAndRuleWrappersAgree: both wrappers must be the same evaluator, so a
|
||||
// window means the same thing wherever it is configured.
|
||||
func TestProfileAndRuleWrappersAgree(t *testing.T) {
|
||||
// TestRuleWrapperAgrees: the wrapper must be the same evaluator, so a window
|
||||
// means the same thing however it is invoked.
|
||||
func TestRuleWrapperAgrees(t *testing.T) {
|
||||
now := at(t, "2026-07-20 23:30 UTC")
|
||||
days, start, end, offset := []string{"mon"}, "22:00", "06:00", 180
|
||||
|
||||
p, _ := ProfileScheduleActive(now, Profile{
|
||||
SchedDays: days, SchedStart: start, SchedEnd: end, SchedUTCOffset: offset,
|
||||
})
|
||||
r, _ := RuleScheduleActive(now, Rule{
|
||||
SchedDays: days, SchedStart: start, SchedEnd: end, SchedUTCOffset: offset,
|
||||
})
|
||||
base, _ := ScheduleWindowActive(now, days, start, end, offset)
|
||||
if p != base || r != base {
|
||||
t.Fatalf("wrappers disagree with the evaluator: profile=%v rule=%v base=%v", p, r, base)
|
||||
if r != base {
|
||||
t.Fatalf("wrapper disagrees with the evaluator: rule=%v base=%v", r, base)
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,437 @@
|
||||
package model
|
||||
|
||||
// Subscription node cache — subscription-fetched nodes live OUTSIDE UCI.
|
||||
//
|
||||
// # Why
|
||||
//
|
||||
// A 200-node subscription persisted as `config node` sections bloats
|
||||
// /etc/config/shater and rewrites the whole file (on the overlay) on every
|
||||
// `sub update`. UCI now stores ONLY manual nodes (FromSub == "") plus the
|
||||
// `config subscription` sections themselves; each subscription's fetched node
|
||||
// set lives in its own JSON cache file that only that subscription's refresh
|
||||
// rewrites.
|
||||
//
|
||||
// # Format
|
||||
//
|
||||
// One file per subscription: <dir>/<escaped-sub-name>.json holding a JSON
|
||||
// OBJECT (not a bare array — an object carries the format version and the
|
||||
// authoritative subscription name, which a bare array cannot):
|
||||
//
|
||||
// {"version": 1, "sub": "<subscription name>", "nodes": [Node, ...]}
|
||||
//
|
||||
// Node uses its natural Go-field JSON encoding — the same shape the panel
|
||||
// already exchanges over GET/PUT /api/config. Unknown fields are ignored on
|
||||
// read (forward compatibility); an unknown version is skipped with a warning.
|
||||
// Files are written indented so `cat` on the router stays a usable debugger.
|
||||
//
|
||||
// # Where
|
||||
//
|
||||
// The persistent home is /etc/shater/subs (survives reboot). Like
|
||||
// generate/cache.go's DB, the overlay is tight, so when the persistent dir
|
||||
// cannot be written the cache degrades to tmpfs (/tmp/shater-subs) with a
|
||||
// warning — a reboot then costs one re-fetch, never a broken config. The
|
||||
// SHATER_SUBS_DIR env var overrides the directory outright (tests, dev hosts).
|
||||
// Writes are atomic: temp file in the target dir, then rename, so a reader
|
||||
// (daemon vs CLI) never observes a torn file. A refresh that produces the
|
||||
// byte-identical file is skipped entirely (no overlay churn).
|
||||
//
|
||||
// # The single merge point
|
||||
//
|
||||
// ReadUCI = ParseUCIExport + MergeSubCaches: every model consumer (apply,
|
||||
// generate, panel, `sub update`, stats) loads through ReadUCI and therefore
|
||||
// sees the same full Model as before — manual nodes from UCI plus cached
|
||||
// subscription nodes. MergeSubCaches also MIGRATES: legacy `from_sub` nodes
|
||||
// still present in UCI are imported into cache files on first read, and
|
||||
// RenderUCIExport no longer emits FromSub nodes, so they drain out of
|
||||
// /etc/config/shater on the next write.
|
||||
|
||||
import (
|
||||
"bytes"
|
||||
"encoding/json"
|
||||
"fmt"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"sort"
|
||||
"strings"
|
||||
)
|
||||
|
||||
const (
|
||||
// subCacheDirPersistent is the preferred (survives reboot) home for the
|
||||
// per-subscription node caches; subCacheDirFallback is the tmpfs degradation
|
||||
// used when the overlay cannot take the write (same policy as generate/cache.go).
|
||||
subCacheDirPersistent = "/etc/shater/subs"
|
||||
subCacheDirFallback = "/tmp/shater-subs"
|
||||
|
||||
// subCacheDirEnv overrides the cache directory entirely (no fallback chain):
|
||||
// tests point it at a temp dir; a dev host keeps /etc clean.
|
||||
subCacheDirEnv = "SHATER_SUBS_DIR"
|
||||
|
||||
// subCacheVersion is the on-disk format version. Readers skip (with a warning)
|
||||
// any file whose version they do not know rather than guessing at its shape.
|
||||
subCacheVersion = 1
|
||||
)
|
||||
|
||||
// subCacheFile is the on-disk JSON shape (see the package comment).
|
||||
type subCacheFile struct {
|
||||
Version int `json:"version"`
|
||||
Sub string `json:"sub"`
|
||||
Nodes []Node `json:"nodes"`
|
||||
}
|
||||
|
||||
// subCacheLogf reports cache degradations (tmpfs fallback, an unreadable or
|
||||
// unknown-version file). model deliberately has no logger dependency; stderr
|
||||
// reaches syslog via procd on the router and the terminal when run by hand.
|
||||
var subCacheLogf = func(format string, args ...any) {
|
||||
fmt.Fprintf(os.Stderr, "shater: subs cache: "+format+"\n", args...)
|
||||
}
|
||||
|
||||
// subCacheDirs returns the candidate directories in priority order: the env
|
||||
// override alone when set, else persistent-then-tmpfs.
|
||||
func subCacheDirs() []string {
|
||||
if d := os.Getenv(subCacheDirEnv); d != "" {
|
||||
return []string{d}
|
||||
}
|
||||
return []string{subCacheDirPersistent, subCacheDirFallback}
|
||||
}
|
||||
|
||||
// subCacheFileName maps a subscription name to a safe file name: bytes outside
|
||||
// [A-Za-z0-9_-] are %XX-escaped ('.' included, so no name can spell a dotfile
|
||||
// or ".."), so a name carrying '/', spaces or traversal syntax can never
|
||||
// escape — or even appear to escape — the cache directory.
|
||||
func subCacheFileName(sub string) string {
|
||||
var b strings.Builder
|
||||
for i := range len(sub) {
|
||||
c := sub[i]
|
||||
switch {
|
||||
case c >= 'a' && c <= 'z', c >= 'A' && c <= 'Z', c >= '0' && c <= '9',
|
||||
c == '_', c == '-':
|
||||
b.WriteByte(c)
|
||||
default:
|
||||
fmt.Fprintf(&b, "%%%02X", c)
|
||||
}
|
||||
}
|
||||
return b.String() + ".json"
|
||||
}
|
||||
|
||||
// marshalSubCache renders the canonical bytes for a subscription's cache file.
|
||||
// Deterministic (fixed field order, fixed indent) so "did it change" is a plain
|
||||
// byte comparison.
|
||||
func marshalSubCache(sub string, nodes []Node) ([]byte, error) {
|
||||
if nodes == nil {
|
||||
nodes = []Node{} // encode as [], never null
|
||||
}
|
||||
data, err := json.MarshalIndent(subCacheFile{Version: subCacheVersion, Sub: sub, Nodes: nodes}, "", "\t")
|
||||
if err != nil {
|
||||
return nil, err
|
||||
}
|
||||
return append(data, '\n'), nil
|
||||
}
|
||||
|
||||
// SaveSubCache atomically persists the node set of one subscription: temp file
|
||||
// + rename in the persistent dir, degrading to tmpfs with a warning when the
|
||||
// overlay refuses the write. Writing the byte-identical content is a no-op, so
|
||||
// a refresh that changed nothing costs no overlay churn.
|
||||
func SaveSubCache(sub string, nodes []Node) error {
|
||||
if strings.TrimSpace(sub) == "" {
|
||||
return fmt.Errorf("subs cache: empty subscription name")
|
||||
}
|
||||
data, err := marshalSubCache(sub, nodes)
|
||||
if err != nil {
|
||||
return fmt.Errorf("subs cache %q: marshal: %w", sub, err)
|
||||
}
|
||||
name := subCacheFileName(sub)
|
||||
dirs := subCacheDirs()
|
||||
var firstErr error
|
||||
for i, dir := range dirs {
|
||||
path := filepath.Join(dir, name)
|
||||
if existing, rerr := os.ReadFile(path); rerr == nil && bytes.Equal(existing, data) {
|
||||
return nil // unchanged — skip the write entirely
|
||||
}
|
||||
if werr := writeFileAtomic(dir, path, data); werr != nil {
|
||||
if firstErr == nil {
|
||||
firstErr = werr
|
||||
}
|
||||
continue
|
||||
}
|
||||
if i > 0 {
|
||||
// Same degradation policy as generate/cache.go: tmpfs keeps things
|
||||
// working, but the cache is gone after a reboot (one re-fetch).
|
||||
subCacheLogf("%s is not writable (%v); cached %q to tmpfs %s — the cache will not survive a reboot", dirs[0], firstErr, sub, path)
|
||||
}
|
||||
return nil
|
||||
}
|
||||
return fmt.Errorf("subs cache %q: %w", sub, firstErr)
|
||||
}
|
||||
|
||||
// writeFileAtomic writes data to path via a temp file + rename in the same
|
||||
// directory, creating the directory first. Rename is atomic on the same
|
||||
// filesystem, so a concurrent reader sees either the old file or the new one,
|
||||
// never a torn write.
|
||||
func writeFileAtomic(dir, path string, data []byte) error {
|
||||
if err := os.MkdirAll(dir, 0o755); err != nil {
|
||||
return err
|
||||
}
|
||||
tmp, err := os.CreateTemp(dir, ".subcache-*.tmp")
|
||||
if err != nil {
|
||||
return err
|
||||
}
|
||||
tmpName := tmp.Name()
|
||||
_, werr := tmp.Write(data)
|
||||
cerr := tmp.Close()
|
||||
if werr == nil {
|
||||
werr = cerr
|
||||
}
|
||||
if werr == nil {
|
||||
werr = os.Rename(tmpName, path)
|
||||
}
|
||||
if werr != nil {
|
||||
_ = os.Remove(tmpName)
|
||||
return werr
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// LoadSubCache reads one subscription's cached node set. ok=false means "no
|
||||
// cache file" (a fresh subscription, or a pre-migration config) — not an error.
|
||||
// A file that exists but cannot be decoded is reported and treated as absent:
|
||||
// the daemon must come up on whatever it finds, and the next refresh rewrites
|
||||
// the file anyway.
|
||||
func LoadSubCache(sub string) (nodes []Node, ok bool, err error) {
|
||||
name := subCacheFileName(sub)
|
||||
for _, dir := range subCacheDirs() {
|
||||
path := filepath.Join(dir, name)
|
||||
data, rerr := os.ReadFile(path)
|
||||
if rerr != nil {
|
||||
continue
|
||||
}
|
||||
f, derr := decodeSubCache(data)
|
||||
if derr != nil {
|
||||
subCacheLogf("%s: %v — ignoring the file (the next refresh rewrites it)", path, derr)
|
||||
continue
|
||||
}
|
||||
return f.Nodes, true, nil
|
||||
}
|
||||
return nil, false, nil
|
||||
}
|
||||
|
||||
// LoadAllSubCaches reads every cache file from the candidate dirs into a
|
||||
// sub-name -> nodes map. The persistent dir shadows tmpfs for the same
|
||||
// subscription. Undecodable files are reported and skipped.
|
||||
func LoadAllSubCaches() map[string][]Node {
|
||||
out := map[string][]Node{}
|
||||
for _, dir := range subCacheDirs() {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, e := range entries {
|
||||
if e.IsDir() || !strings.HasSuffix(e.Name(), ".json") {
|
||||
continue
|
||||
}
|
||||
path := filepath.Join(dir, e.Name())
|
||||
data, rerr := os.ReadFile(path)
|
||||
if rerr != nil {
|
||||
subCacheLogf("%s: %v — skipped", path, rerr)
|
||||
continue
|
||||
}
|
||||
f, derr := decodeSubCache(data)
|
||||
if derr != nil {
|
||||
subCacheLogf("%s: %v — skipped (the next refresh rewrites it)", path, derr)
|
||||
continue
|
||||
}
|
||||
if f.Sub == "" {
|
||||
subCacheLogf("%s: missing \"sub\" field — skipped", path)
|
||||
continue
|
||||
}
|
||||
if _, dup := out[f.Sub]; dup {
|
||||
continue // earlier (higher-priority) dir wins
|
||||
}
|
||||
out[f.Sub] = f.Nodes
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// decodeSubCache parses and version-checks one cache file body.
|
||||
func decodeSubCache(data []byte) (*subCacheFile, error) {
|
||||
var f subCacheFile
|
||||
if err := json.Unmarshal(data, &f); err != nil {
|
||||
return nil, fmt.Errorf("decode: %w", err)
|
||||
}
|
||||
if f.Version != subCacheVersion {
|
||||
return nil, fmt.Errorf("unknown cache version %d (want %d)", f.Version, subCacheVersion)
|
||||
}
|
||||
return &f, nil
|
||||
}
|
||||
|
||||
// MergeSubCaches is THE merge point: it folds the on-disk subscription caches
|
||||
// into a Model fresh out of ParseUCIExport, so every consumer of ReadUCI sees
|
||||
// the same full Model (manual nodes + subscription nodes) it always did.
|
||||
//
|
||||
// It also performs the one-shot MIGRATION off the old layout: `from_sub` nodes
|
||||
// still present in UCI (a config written before this cache existed) are
|
||||
// imported into cache files right here. When a cache file already exists for
|
||||
// the same subscription the file wins (it is the newer write path) and the
|
||||
// legacy UCI copies are dropped. Either way RenderUCIExport no longer emits
|
||||
// FromSub nodes, so the legacy sections drain out of /etc/config/shater on the
|
||||
// next write.
|
||||
//
|
||||
// Resulting node order is deterministic: manual nodes in UCI order, then each
|
||||
// subscription's cached nodes in m.Subscriptions order (file order within a
|
||||
// subscription = feed order), then caches for subscriptions no longer in the
|
||||
// config, sorted by name. That last group exists so a hand-edited config
|
||||
// (subscription section deleted with `uci` while nodes were still cached)
|
||||
// degrades to "nodes kept" rather than silently dropping them; the panel's
|
||||
// SyncSubCaches removes such orphan files when a subscription is deleted
|
||||
// through the UI.
|
||||
func MergeSubCaches(m *Model) {
|
||||
if m == nil {
|
||||
return
|
||||
}
|
||||
cached := LoadAllSubCaches()
|
||||
|
||||
// Partition UCI nodes: manual ones stay; legacy from_sub ones migrate.
|
||||
manual := make([]Node, 0, len(m.Nodes))
|
||||
legacy := map[string][]Node{}
|
||||
for _, n := range m.Nodes {
|
||||
if n.FromSub == "" {
|
||||
manual = append(manual, n)
|
||||
continue
|
||||
}
|
||||
legacy[n.FromSub] = append(legacy[n.FromSub], n)
|
||||
}
|
||||
for sub, nodes := range legacy {
|
||||
if _, exists := cached[sub]; exists {
|
||||
continue // the cache file is the newer write path — it wins
|
||||
}
|
||||
if err := SaveSubCache(sub, nodes); err != nil {
|
||||
// Import failed (disk full on both homes). Keep serving the legacy
|
||||
// copies this read; the next read retries the import.
|
||||
subCacheLogf("migrating %d legacy UCI node(s) of %q: %v", len(nodes), sub, err)
|
||||
}
|
||||
cached[sub] = nodes
|
||||
}
|
||||
|
||||
// Rebuild: manual + per-subscription caches in config order + orphans.
|
||||
merged := manual
|
||||
for _, sub := range m.Subscriptions {
|
||||
merged = appendSubNodes(merged, sub.Name, cached[sub.Name])
|
||||
delete(cached, sub.Name)
|
||||
}
|
||||
orphans := make([]string, 0, len(cached))
|
||||
for sub := range cached {
|
||||
orphans = append(orphans, sub)
|
||||
}
|
||||
sort.Strings(orphans)
|
||||
for _, sub := range orphans {
|
||||
merged = appendSubNodes(merged, sub, cached[sub])
|
||||
}
|
||||
m.Nodes = merged
|
||||
}
|
||||
|
||||
// appendSubNodes appends one subscription's cached nodes, restoring the
|
||||
// FromSub invariant on each (the file is the storage key's authority, not the
|
||||
// node's own field — a hand-edited file must not smuggle a node into another
|
||||
// subscription or into the manual set).
|
||||
func appendSubNodes(dst []Node, sub string, nodes []Node) []Node {
|
||||
for _, n := range nodes {
|
||||
n.FromSub = sub
|
||||
dst = append(dst, n)
|
||||
}
|
||||
return dst
|
||||
}
|
||||
|
||||
// NodesFromSub returns the nodes belonging to one subscription — the set
|
||||
// SaveSubCache persists after a refresh.
|
||||
func (m *Model) NodesFromSub(sub string) []Node {
|
||||
var out []Node
|
||||
for _, n := range m.Nodes {
|
||||
if n.FromSub == sub {
|
||||
out = append(out, n)
|
||||
}
|
||||
}
|
||||
return out
|
||||
}
|
||||
|
||||
// SyncSubCaches is the WRITE-side split for whole-model writers (the panel's
|
||||
// PUT /api/config): manual nodes go to UCI via WriteUCI (whose renderer skips
|
||||
// FromSub nodes), while this call reconciles the cache files to the FromSub
|
||||
// nodes carried in m —
|
||||
//
|
||||
// - every subscription present in m gets its cache rewritten to exactly its
|
||||
// node set (an emptied set persists as an empty file only when a file
|
||||
// already exists, so deleting the last node sticks without littering a
|
||||
// never-fetched subscription with an empty cache);
|
||||
// - cache files whose subscription is neither configured nor carried by any
|
||||
// node are removed (deleting a subscription in the panel deletes its cache).
|
||||
//
|
||||
// Unchanged files are skipped byte-for-byte by SaveSubCache, so an unrelated
|
||||
// config edit costs zero cache writes.
|
||||
func SyncSubCaches(m *Model) error {
|
||||
if m == nil {
|
||||
return nil
|
||||
}
|
||||
bySub := map[string][]Node{}
|
||||
for _, n := range m.Nodes {
|
||||
if n.FromSub != "" {
|
||||
bySub[n.FromSub] = append(bySub[n.FromSub], n)
|
||||
}
|
||||
}
|
||||
keep := map[string]bool{}
|
||||
var errs []string
|
||||
for _, sub := range m.Subscriptions {
|
||||
if sub.Name == "" {
|
||||
continue
|
||||
}
|
||||
keep[sub.Name] = true
|
||||
nodes := bySub[sub.Name]
|
||||
delete(bySub, sub.Name)
|
||||
if len(nodes) == 0 {
|
||||
if _, ok, _ := LoadSubCache(sub.Name); !ok {
|
||||
continue // never fetched — don't litter an empty file
|
||||
}
|
||||
}
|
||||
if err := SaveSubCache(sub.Name, nodes); err != nil {
|
||||
errs = append(errs, err.Error())
|
||||
}
|
||||
}
|
||||
// Nodes referencing a subscription that is not configured (an orphan kept by
|
||||
// MergeSubCaches): persist them too, so a PUT of a merged model never loses
|
||||
// what a read produced.
|
||||
for sub, nodes := range bySub {
|
||||
keep[sub] = true
|
||||
if err := SaveSubCache(sub, nodes); err != nil {
|
||||
errs = append(errs, err.Error())
|
||||
}
|
||||
}
|
||||
removeOrphanSubCaches(keep)
|
||||
if len(errs) > 0 {
|
||||
return fmt.Errorf("subs cache: %s", strings.Join(errs, "; "))
|
||||
}
|
||||
return nil
|
||||
}
|
||||
|
||||
// removeOrphanSubCaches deletes cache files (in every candidate dir) whose
|
||||
// subscription is not in keep. Best-effort: a file that cannot be removed is
|
||||
// reported and left; it only costs a few KB until the dir is writable again.
|
||||
func removeOrphanSubCaches(keep map[string]bool) {
|
||||
keepFiles := map[string]bool{}
|
||||
for sub := range keep {
|
||||
keepFiles[subCacheFileName(sub)] = true
|
||||
}
|
||||
for _, dir := range subCacheDirs() {
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil {
|
||||
continue
|
||||
}
|
||||
for _, e := range entries {
|
||||
if e.IsDir() || !strings.HasSuffix(e.Name(), ".json") || keepFiles[e.Name()] {
|
||||
continue
|
||||
}
|
||||
path := filepath.Join(dir, e.Name())
|
||||
if rerr := os.Remove(path); rerr != nil {
|
||||
subCacheLogf("removing orphan %s: %v", path, rerr)
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
package model
|
||||
|
||||
// Unit coverage for the per-subscription node cache (subcache.go): the on-disk
|
||||
// format, the write-skip and orphan-cleanup behaviour SyncSubCaches gives the
|
||||
// panel's PUT split, and the merge semantics ReadUCI relies on. All tests pin
|
||||
// the directory via SHATER_SUBS_DIR, so nothing touches /etc or /tmp.
|
||||
|
||||
import (
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"strings"
|
||||
"testing"
|
||||
)
|
||||
|
||||
func setSubsDir(t *testing.T) string {
|
||||
t.Helper()
|
||||
dir := t.TempDir()
|
||||
t.Setenv("SHATER_SUBS_DIR", dir)
|
||||
return dir
|
||||
}
|
||||
|
||||
// TestSubCacheSaveLoad: the basic contract — what is saved is loaded, an absent
|
||||
// cache is (nil, ok=false, nil), and LoadAllSubCaches sees every saved sub.
|
||||
func TestSubCacheSaveLoad(t *testing.T) {
|
||||
setSubsDir(t)
|
||||
nodes := []Node{
|
||||
{Name: "a", Enabled: true, URI: "ss://a", FromSub: "s1", Fingerprint: "fp-a"},
|
||||
{Name: "b", Enabled: false, URI: "vless://b", FromSub: "s1", Stale: true},
|
||||
}
|
||||
if err := SaveSubCache("s1", nodes); err != nil {
|
||||
t.Fatalf("save: %v", err)
|
||||
}
|
||||
got, ok, err := LoadSubCache("s1")
|
||||
if err != nil || !ok || !reflect.DeepEqual(got, nodes) {
|
||||
t.Fatalf("load: ok=%v err=%v got=%+v", ok, err, got)
|
||||
}
|
||||
if _, ok, err := LoadSubCache("nope"); ok || err != nil {
|
||||
t.Fatalf("absent cache must be (ok=false, err=nil), got ok=%v err=%v", ok, err)
|
||||
}
|
||||
if err := SaveSubCache("s2", []Node{{Name: "c", URI: "ss://c", FromSub: "s2"}}); err != nil {
|
||||
t.Fatalf("save s2: %v", err)
|
||||
}
|
||||
all := LoadAllSubCaches()
|
||||
if len(all) != 2 || len(all["s1"]) != 2 || len(all["s2"]) != 1 {
|
||||
t.Fatalf("LoadAllSubCaches = %+v", all)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSubCacheFileNameEscaped: a subscription name carrying path syntax must not
|
||||
// escape the cache dir — everything outside [A-Za-z0-9._-] is %XX-escaped, and
|
||||
// the round-trip still works because the sub name inside the file is authoritative.
|
||||
func TestSubCacheFileNameEscaped(t *testing.T) {
|
||||
dir := setSubsDir(t)
|
||||
hostile := "../evil/один два"
|
||||
if err := SaveSubCache(hostile, []Node{{Name: "n", URI: "ss://n", FromSub: hostile}}); err != nil {
|
||||
t.Fatalf("save: %v", err)
|
||||
}
|
||||
entries, err := os.ReadDir(dir)
|
||||
if err != nil || len(entries) != 1 {
|
||||
t.Fatalf("cache dir: entries=%v err=%v (the file must land INSIDE the dir)", entries, err)
|
||||
}
|
||||
name := entries[0].Name()
|
||||
if strings.ContainsAny(name, "/\\ ") || strings.Contains(name, "..") {
|
||||
t.Fatalf("unsafe cache file name %q", name)
|
||||
}
|
||||
if nodes, ok, _ := LoadSubCache(hostile); !ok || len(nodes) != 1 {
|
||||
t.Fatalf("hostile-named sub did not round-trip: ok=%v nodes=%+v", ok, nodes)
|
||||
}
|
||||
if all := LoadAllSubCaches(); len(all[hostile]) != 1 {
|
||||
t.Fatalf("LoadAllSubCaches lost the hostile-named sub: %+v", all)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSubCacheUnchangedWriteSkipped: saving the byte-identical set must not
|
||||
// rewrite the file (no overlay churn on a refresh that changed nothing).
|
||||
func TestSubCacheUnchangedWriteSkipped(t *testing.T) {
|
||||
dir := setSubsDir(t)
|
||||
nodes := []Node{{Name: "a", Enabled: true, URI: "ss://a", FromSub: "s1"}}
|
||||
if err := SaveSubCache("s1", nodes); err != nil {
|
||||
t.Fatalf("save: %v", err)
|
||||
}
|
||||
path := filepath.Join(dir, subCacheFileName("s1"))
|
||||
before, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
// Make any rewrite observable regardless of timestamp granularity.
|
||||
old := before.ModTime().Add(-1e9)
|
||||
if err := os.Chtimes(path, old, old); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := SaveSubCache("s1", nodes); err != nil {
|
||||
t.Fatalf("second save: %v", err)
|
||||
}
|
||||
after, err := os.Stat(path)
|
||||
if err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if !after.ModTime().Equal(old) {
|
||||
t.Fatal("identical save rewrote the file (should be a no-op)")
|
||||
}
|
||||
}
|
||||
|
||||
// TestSubCacheUnknownVersionSkipped: a cache file from a future format version
|
||||
// is skipped (treated as absent) rather than half-decoded.
|
||||
func TestSubCacheUnknownVersionSkipped(t *testing.T) {
|
||||
dir := setSubsDir(t)
|
||||
body := `{"version": 99, "sub": "s1", "nodes": [{"Name": "x"}]}`
|
||||
if err := os.WriteFile(filepath.Join(dir, subCacheFileName("s1")), []byte(body), 0o644); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if _, ok, err := LoadSubCache("s1"); ok || err != nil {
|
||||
t.Fatalf("future-version file must read as absent, got ok=%v err=%v", ok, err)
|
||||
}
|
||||
if all := LoadAllSubCaches(); len(all) != 0 {
|
||||
t.Fatalf("LoadAllSubCaches must skip a future-version file: %+v", all)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMergeSubCachesPrefersFileOverLegacyUCI: during a partial migration both a
|
||||
// cache file and legacy UCI sections can exist for the same sub — the file is
|
||||
// the newer write path and must win, without duplicating nodes.
|
||||
func TestMergeSubCachesPrefersFileOverLegacyUCI(t *testing.T) {
|
||||
setSubsDir(t)
|
||||
fresh := []Node{{Name: "new", Enabled: true, URI: "ss://new", FromSub: "s1"}}
|
||||
if err := SaveSubCache("s1", fresh); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m := &Model{
|
||||
Subscriptions: []Subscription{{Name: "s1", Enabled: true}},
|
||||
Nodes: []Node{
|
||||
{Name: "manual", URI: "ss://m"},
|
||||
{Name: "old", URI: "ss://old", FromSub: "s1"}, // legacy UCI copy
|
||||
},
|
||||
}
|
||||
MergeSubCaches(m)
|
||||
if len(m.Nodes) != 2 || m.Nodes[0].Name != "manual" || m.Nodes[1].Name != "new" {
|
||||
t.Fatalf("merge = %+v, want [manual new] (file wins over legacy UCI)", m.Nodes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestMergeSubCachesKeepsOrphanCache: nodes cached for a subscription that was
|
||||
// removed from UCI by hand still merge in (kept, sorted last) — MergeSubCaches
|
||||
// never silently drops nodes; only the panel's SyncSubCaches deletes a cache.
|
||||
func TestMergeSubCachesKeepsOrphanCache(t *testing.T) {
|
||||
setSubsDir(t)
|
||||
if err := SaveSubCache("gone", []Node{{Name: "g", URI: "ss://g", FromSub: "gone"}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
m := &Model{Nodes: []Node{{Name: "manual", URI: "ss://m"}}}
|
||||
MergeSubCaches(m)
|
||||
if len(m.Nodes) != 2 || m.Nodes[1].FromSub != "gone" {
|
||||
t.Fatalf("orphan cache dropped by merge: %+v", m.Nodes)
|
||||
}
|
||||
}
|
||||
|
||||
// TestSyncSubCaches is the panel-PUT split contract: each configured sub's cache
|
||||
// is rewritten to exactly its node set, deleting the last node of a sub sticks,
|
||||
// a never-fetched sub gains no empty file, and the cache of a subscription that
|
||||
// disappeared from the model is removed.
|
||||
func TestSyncSubCaches(t *testing.T) {
|
||||
dir := setSubsDir(t)
|
||||
// Pre-state: s1 and doomed have caches on disk.
|
||||
if err := SaveSubCache("s1", []Node{{Name: "old", URI: "ss://old", FromSub: "s1"}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
if err := SaveSubCache("doomed", []Node{{Name: "d", URI: "ss://d", FromSub: "doomed"}}); err != nil {
|
||||
t.Fatal(err)
|
||||
}
|
||||
|
||||
m := &Model{
|
||||
Subscriptions: []Subscription{
|
||||
{Name: "s1", Enabled: true},
|
||||
{Name: "fresh", Enabled: true}, // never fetched, no nodes
|
||||
},
|
||||
Nodes: []Node{
|
||||
{Name: "manual", URI: "ss://m"},
|
||||
{Name: "edited", Enabled: false, URI: "ss://e", FromSub: "s1"},
|
||||
},
|
||||
}
|
||||
if err := SyncSubCaches(m); err != nil {
|
||||
t.Fatalf("sync: %v", err)
|
||||
}
|
||||
|
||||
if nodes, ok, _ := LoadSubCache("s1"); !ok || len(nodes) != 1 || nodes[0].Name != "edited" || nodes[0].Enabled {
|
||||
t.Fatalf("s1 cache not rewritten to the PUT set: ok=%v %+v", ok, nodes)
|
||||
}
|
||||
if _, err := os.Stat(filepath.Join(dir, subCacheFileName("fresh"))); !os.IsNotExist(err) {
|
||||
t.Fatalf("never-fetched sub must not gain an empty cache file (stat err=%v)", err)
|
||||
}
|
||||
if _, ok, _ := LoadSubCache("doomed"); ok {
|
||||
t.Fatal("cache of a deleted subscription must be removed")
|
||||
}
|
||||
|
||||
// Deleting the LAST node of s1 must persist as an EMPTY cache, not resurrect.
|
||||
m.Nodes = m.Nodes[:1] // manual only
|
||||
if err := SyncSubCaches(m); err != nil {
|
||||
t.Fatalf("second sync: %v", err)
|
||||
}
|
||||
if nodes, ok, _ := LoadSubCache("s1"); !ok || len(nodes) != 0 {
|
||||
t.Fatalf("emptied sub must persist as an empty cache: ok=%v %+v", ok, nodes)
|
||||
}
|
||||
m2 := &Model{Subscriptions: m.Subscriptions, Nodes: []Node{{Name: "manual", URI: "ss://m"}}}
|
||||
MergeSubCaches(m2)
|
||||
if len(m2.Nodes) != 1 {
|
||||
t.Fatalf("deleted nodes resurrected on merge: %+v", m2.Nodes)
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user